Archived
A push reaching the mirror should take minutes, not up to six hours. All four units now fire every 5 minutes, staggered a minute apart. Three guards make that cadence safe: - Scheduled runs take the release lock NON-BLOCKING (try_release_lock) and skip the tick when a build is running. Blocking would stack one stalled process per tick behind a long build and stampede when it finished. Manual commands still wait, as an operator expects. - check-versions takes the lock too, and now owns its git pull (--pull, passed by the unit) instead of an ExecStartPre: at this cadence an unlocked pull would swap PKGBUILDs out from under a running build. - A failed release records .build-failed-<channel> and backs off exponentially (10m, 20m, 40m … capped at 6h) rather than rebuilding the same broken tree every 5 minutes. Any new commit clears the backoff, since a push is the most likely fix. Idle ticks exit without output so the journal keeps showing the runs that matter, and bin/repo timers reports backoff state — a paused channel is otherwise indistinguishable from an idle one.
122 lines
4.2 KiB
Bash
Executable File
122 lines
4.2 KiB
Bash
Executable File
#!/bin/bash
|
|
# Run the release workflow for a channel when work is queued.
|
|
# Usage: auto-release <edge|rc|stable>
|
|
#
|
|
# Safe to run on a tight schedule. Three guards make that true:
|
|
#
|
|
# 1. Nothing queued -> exit immediately (the common case).
|
|
# 2. A release already running -> skip this tick. The lock is taken
|
|
# non-blocking on purpose: waiting would pile up one stalled process per
|
|
# tick behind a long build, and they would all stampede when it finished.
|
|
# 3. The last attempt failed -> back off exponentially rather than rebuild
|
|
# the same broken tree every few minutes. A change to the repository
|
|
# (new commit) clears the backoff immediately, because that is the thing
|
|
# most likely to have fixed it.
|
|
|
|
set -e
|
|
|
|
BUILD_ROOT=$(realpath "${BASH_SOURCE[0]%/*}/..")
|
|
source "$BUILD_ROOT/helpers/message-helpers.sh"
|
|
source "$BUILD_ROOT/helpers/paths.sh"
|
|
source "$BUILD_ROOT/helpers/lock-helpers.sh"
|
|
|
|
MIRROR="${1:-}"
|
|
STATE_DIR="${OMARCHY_STATE_DIR:-/root/.state}"
|
|
|
|
# Backoff schedule: 10m, 20m, 40m, 80m, 160m, 320m, then hourly-ish forever
|
|
# (capped at 6h, the cadence this system ran at before frequent timers).
|
|
BACKOFF_BASE_SECONDS="${OMARCHY_BACKOFF_BASE:-600}"
|
|
BACKOFF_MAX_SECONDS="${OMARCHY_BACKOFF_MAX:-21600}"
|
|
|
|
if [[ -z "$MIRROR" ]]; then
|
|
print_error "Usage: $0 <mirror>"
|
|
echo " mirror: edge, rc, or stable"
|
|
exit 1
|
|
fi
|
|
|
|
if [[ "$MIRROR" != "edge" && "$MIRROR" != "rc" && "$MIRROR" != "stable" ]]; then
|
|
print_error "Invalid mirror: $MIRROR (must be 'edge', 'rc', or 'stable')"
|
|
exit 1
|
|
fi
|
|
|
|
STATE_FILE="$STATE_DIR/.sync-needed-$MIRROR"
|
|
FAIL_FILE="$STATE_DIR/.build-failed-$MIRROR"
|
|
|
|
# Nothing queued: stay quiet. At a 5-minute cadence this is most invocations,
|
|
# and a header for each would bury the runs that matter in the journal.
|
|
if [[ ! -f "$STATE_FILE" ]]; then
|
|
exit 0
|
|
fi
|
|
|
|
print_header "Processing Sync for $MIRROR"
|
|
|
|
# The build inputs are this checkout's contents; its HEAD identifies them.
|
|
current_fingerprint() {
|
|
git -C "$BUILD_ROOT" rev-parse HEAD 2>/dev/null || echo "unknown"
|
|
}
|
|
|
|
backoff_seconds() {
|
|
local count="$1" delay="$BACKOFF_BASE_SECONDS"
|
|
while ((count > 1)); do
|
|
delay=$((delay * 2))
|
|
((delay >= BACKOFF_MAX_SECONDS)) && { delay=$BACKOFF_MAX_SECONDS; break; }
|
|
count=$((count - 1))
|
|
done
|
|
echo "$delay"
|
|
}
|
|
|
|
FAIL_COUNT=0
|
|
if [[ -f "$FAIL_FILE" ]]; then
|
|
# shellcheck disable=SC1090
|
|
source "$FAIL_FILE" 2>/dev/null || true
|
|
FAIL_COUNT="${FAILURE_COUNT:-0}"
|
|
failed_at="${FAILURE_AT:-0}"
|
|
failed_fingerprint="${FAILURE_FINGERPRINT:-}"
|
|
|
|
if [[ "$failed_fingerprint" != "$(current_fingerprint)" ]]; then
|
|
print_info "Repository changed since the last failure — clearing backoff and retrying"
|
|
rm -f "$FAIL_FILE"
|
|
FAIL_COUNT=0
|
|
else
|
|
delay=$(backoff_seconds "$FAIL_COUNT")
|
|
now=$(date +%s)
|
|
retry_at=$((failed_at + delay))
|
|
if ((now < retry_at)); then
|
|
print_warning "$MIRROR has failed $FAIL_COUNT time(s) on this tree — not retrying until $(date -d "@$retry_at" '+%H:%M:%S' 2>/dev/null || echo "+$((retry_at - now))s")"
|
|
echo " Push a fix (any new commit clears this), or: rm $FAIL_FILE"
|
|
exit 0
|
|
fi
|
|
print_info "Backoff elapsed — retrying $MIRROR (failure #$((FAIL_COUNT + 1)) if this fails)"
|
|
fi
|
|
fi
|
|
|
|
# Non-blocking: a build in progress means this tick has nothing to do.
|
|
if ! try_release_lock; then
|
|
holder=$(release_lock_holder)
|
|
print_info "A release is already running (${holder:-holder unknown}) — skipping this tick"
|
|
exit 0
|
|
fi
|
|
|
|
print_info "State file found: $STATE_FILE"
|
|
print_info "Starting release workflow for $MIRROR..."
|
|
|
|
if "$BUILD_ROOT/bin/repo" release --mirror "$MIRROR" --skip-prod-check; then
|
|
print_success "Release completed successfully for $MIRROR"
|
|
rm -f "$STATE_FILE"
|
|
rm -f "$FAIL_FILE"
|
|
print_success "State file removed: $STATE_FILE"
|
|
else
|
|
status=$?
|
|
FAIL_COUNT=$((FAIL_COUNT + 1))
|
|
cat >"$FAIL_FILE" <<EOF
|
|
FAILURE_COUNT=$FAIL_COUNT
|
|
FAILURE_AT=$(date +%s)
|
|
FAILURE_FINGERPRINT=$(current_fingerprint)
|
|
EOF
|
|
next=$(backoff_seconds "$FAIL_COUNT")
|
|
print_error "Release failed for $MIRROR (attempt $FAIL_COUNT)"
|
|
print_warning "State file retained for retry: $STATE_FILE"
|
|
print_warning "Backing off $((next / 60))m before the next attempt; a new commit retries sooner"
|
|
exit "$status"
|
|
fi
|