Files
omarchy-pkgs/bin/auto-release
T
Marcelo Alcantara 91843ab099 Make aarch64 a first-class architecture in the scheduled pipeline
One list, PUBLISHED_ARCHES in helpers/paths.sh (default x86_64,
overridable with OMARCHY_ARCHES), now drives everything the repository
host schedules. check-versions compares PKGBUILDs against each
architecture's channel databases and writes one queue per channel and
architecture; auto-release works through the queues one architecture at
a time, each with its own backoff, so a failing build on one never
blocks the other; advance-channel --arch all re-runs an advance for every
published architecture and omarchy-release uses it for start and ship,
building the pinned pair once per architecture in its rc trigger; the
train observes channels through the reference (first) architecture
instead of a hard-coded x86_64. Queue and backoff files written under
the old per-channel names are treated as x86_64 until consumed.

Two things made an aarch64 builder image impossible to create: the
keyring bootstrap fetched omarchy-keyring from the target architecture's
own channel tree, which does not exist before that architecture has
published anything, and the QEMU probe only knew the x86_64-host,
aarch64-target case. The keyring (arch=any) now always comes from the
x86_64 tree, and the probe compares host and target architectures and
runs a container for the target platform.

clean-repo grouped versions with a regex that only knew any, x86_64 and
i686, so aarch64 packages would never have been pruned.
2026-09-04 23:40:49 -04:00

158 lines
5.7 KiB
Bash
Executable File

#!/bin/bash
# Run the release workflow for a channel when work is queued.
# Usage: auto-release <edge|rc|stable> [<arch>|all]
#
# With no architecture (what the timers pass), every published architecture
# is processed in turn, each against its own queue and its own backoff, so a
# failing aarch64 build never holds up x86_64 or the other way round.
#
# Safe to run on a tight schedule. Three guards make that true:
#
# 1. Nothing queued -> exit immediately (the common case).
# 2. A release already running -> skip this tick. The lock is taken
# non-blocking on purpose: waiting would pile up one stalled process per
# tick behind a long build, and they would all stampede when it finished.
# 3. The last attempt failed -> back off exponentially rather than rebuild
# the same broken tree every few minutes. A change to the repository
# (new commit) clears the backoff immediately, because that is the thing
# most likely to have fixed it.
set -e
BUILD_ROOT=$(realpath "${BASH_SOURCE[0]%/*}/..")
source "$BUILD_ROOT/helpers/message-helpers.sh"
source "$BUILD_ROOT/helpers/paths.sh"
source "$BUILD_ROOT/helpers/lock-helpers.sh"
MIRROR="${1:-}"
ARCH_ARG="${2:-all}"
# Backoff schedule: 10m, 20m, 40m, 80m, 160m, 320m, then hourly-ish forever
# (capped at 6h, the cadence this system ran at before frequent timers).
BACKOFF_BASE_SECONDS="${OMARCHY_BACKOFF_BASE:-600}"
BACKOFF_MAX_SECONDS="${OMARCHY_BACKOFF_MAX:-21600}"
if [[ -z "$MIRROR" ]]; then
print_error "Usage: $0 <mirror> [<arch>|all]"
echo " mirror: edge, rc, or stable"
echo " arch: one of $VALID_ARCHES, or all (default) for every published architecture"
exit 1
fi
if [[ "$MIRROR" != "edge" && "$MIRROR" != "rc" && "$MIRROR" != "stable" ]]; then
print_error "Invalid mirror: $MIRROR (must be 'edge', 'rc', or 'stable')"
exit 1
fi
if [[ "$ARCH_ARG" == "all" ]]; then
ARCHES=$(published_arches)
else
require_valid_arch "$ARCH_ARG"
ARCHES="$ARCH_ARG"
fi
# The build inputs are this checkout's contents; its HEAD identifies them.
current_fingerprint() {
git -C "$BUILD_ROOT" rev-parse HEAD 2>/dev/null || echo "unknown"
}
backoff_seconds() {
local count="$1" delay="$BACKOFF_BASE_SECONDS"
while ((count > 1)); do
delay=$((delay * 2))
((delay >= BACKOFF_MAX_SECONDS)) && { delay=$BACKOFF_MAX_SECONDS; break; }
count=$((count - 1))
done
echo "$delay"
}
# Returns 0 when this architecture's queue was processed (or was empty), 1 when
# the release failed. Backoff and "someone else holds the lock" are not
# failures: they are this tick deciding to do nothing.
release_arch() {
local arch="$1"
local state_file fail_file
state_file=$(sync_queue_file "$MIRROR" "$arch")
fail_file=$(sync_fail_file "$MIRROR" "$arch")
# A queue written under the pre-architecture name belongs to x86_64.
if [[ "$arch" == "x86_64" && ! -f "$state_file" && -f "$(legacy_sync_queue_file "$MIRROR")" ]]; then
state_file=$(legacy_sync_queue_file "$MIRROR")
fi
if [[ "$arch" == "x86_64" && ! -f "$fail_file" && -f "$(legacy_sync_fail_file "$MIRROR")" ]]; then
fail_file=$(legacy_sync_fail_file "$MIRROR")
fi
# Nothing queued: stay quiet. At a 5-minute cadence this is most
# invocations, and a header for each would bury the runs that matter in
# the journal.
[[ -f "$state_file" ]] || return 0
print_header "Processing Sync for $MIRROR ($arch)"
local fail_count=0 failed_at failed_fingerprint delay now retry_at
if [[ -f "$fail_file" ]]; then
FAILURE_COUNT=0 FAILURE_AT=0 FAILURE_FINGERPRINT=""
# shellcheck disable=SC1090
source "$fail_file" 2>/dev/null || true
fail_count="${FAILURE_COUNT:-0}"
failed_at="${FAILURE_AT:-0}"
failed_fingerprint="${FAILURE_FINGERPRINT:-}"
if [[ "$failed_fingerprint" != "$(current_fingerprint)" ]]; then
print_info "Repository changed since the last failure — clearing backoff and retrying"
rm -f "$fail_file"
fail_count=0
else
delay=$(backoff_seconds "$fail_count")
now=$(date +%s)
retry_at=$((failed_at + delay))
if ((now < retry_at)); then
print_warning "$MIRROR ($arch) has failed $fail_count time(s) on this tree — not retrying until $(date -d "@$retry_at" '+%H:%M:%S' 2>/dev/null || echo "+$((retry_at - now))s")"
echo " Push a fix (any new commit clears this), or: rm $fail_file"
return 0
fi
print_info "Backoff elapsed — retrying $MIRROR ($arch) (failure #$((fail_count + 1)) if this fails)"
fi
fi
# Non-blocking: a build in progress means this tick has nothing to do. The
# lock is reentrant, so once this run holds it the remaining architectures
# run under the same acquisition.
if ! try_release_lock; then
local holder
holder=$(release_lock_holder)
print_info "A release is already running (${holder:-holder unknown}) — skipping this tick"
return 0
fi
print_info "State file found: $state_file"
print_info "Starting release workflow for $MIRROR ($arch)..."
if "$BUILD_ROOT/bin/repo" release --mirror "$MIRROR" --arch "$arch" --skip-prod-check; then
print_success "Release completed successfully for $MIRROR ($arch)"
rm -f "$state_file" "$fail_file"
print_success "State file removed: $state_file"
return 0
fi
fail_count=$((fail_count + 1))
cat >"$fail_file" <<EOF
FAILURE_COUNT=$fail_count
FAILURE_AT=$(date +%s)
FAILURE_FINGERPRINT=$(current_fingerprint)
EOF
local next
next=$(backoff_seconds "$fail_count")
print_error "Release failed for $MIRROR ($arch) (attempt $fail_count)"
print_warning "State file retained for retry: $state_file"
print_warning "Backing off $((next / 60))m before the next attempt; a new commit retries sooner"
return 1
}
status=0
for arch in $ARCHES; do
release_arch "$arch" || status=1
done
exit "$status"