A push reaching the mirror should take minutes, not up to six hours. All four units now fire every 5 minutes, staggered a minute apart. Three guards make that cadence safe: - Scheduled runs take the release lock NON-BLOCKING (try_release_lock) and skip the tick when a build is running. Blocking would stack one stalled process per tick behind a long build and stampede when it finished. Manual commands still wait, as an operator expects. - check-versions takes the lock too, and now owns its git pull (--pull, passed by the unit) instead of an ExecStartPre: at this cadence an unlocked pull would swap PKGBUILDs out from under a running build. - A failed release records .build-failed-<channel> and backs off exponentially (10m, 20m, 40m … capped at 6h) rather than rebuilding the same broken tree every 5 minutes. Any new commit clears the backoff, since a push is the most likely fix. Idle ticks exit without output so the journal keeps showing the runs that matter, and bin/repo timers reports backoff state — a paused channel is otherwise indistinguishable from an idle one.
164 lines
5.4 KiB
Bash
Executable File
164 lines
5.4 KiB
Bash
Executable File
#!/bin/bash
|
|
# Show the release timers: schedule, last run and its result, what is queued,
|
|
# and what is running right now.
|
|
#
|
|
# Through bin/repo this forwards to the repository host, so the build box's
|
|
# state is one command away from any machine.
|
|
|
|
set -e
|
|
|
|
BUILD_ROOT=$(realpath "${BASH_SOURCE[0]%/*}/..")
|
|
source "$BUILD_ROOT/helpers/message-helpers.sh"
|
|
source "$BUILD_ROOT/helpers/paths.sh"
|
|
|
|
STATE_DIR="${OMARCHY_STATE_DIR:-/root/.state}"
|
|
|
|
if [[ "${1:-}" == "-h" || "${1:-}" == "--help" ]]; then
|
|
echo "Usage: $0"
|
|
echo ""
|
|
echo "Reports each omarchy release timer's schedule, last run and result,"
|
|
echo "any queued builds, and whether a release is running now."
|
|
exit 0
|
|
fi
|
|
|
|
print_header "Omarchy Release Timers"
|
|
|
|
if ! command -v systemctl >/dev/null 2>&1; then
|
|
print_warning "systemctl not found — timers live on the repository host"
|
|
echo "Run this against the host: bin/repo timers (with OMARCHY_REPO_HOST or .repo-host set)"
|
|
exit 0
|
|
fi
|
|
|
|
# Discover units from the repo so this never drifts from what setup installs.
|
|
UNITS=()
|
|
for unit in "$BUILD_ROOT"/systemd/*.timer; do
|
|
[[ -e "$unit" ]] || continue
|
|
UNITS+=("$(basename "$unit" .timer)")
|
|
done
|
|
|
|
if [[ ${#UNITS[@]} -eq 0 ]]; then
|
|
print_warning "No timer units found in $BUILD_ROOT/systemd"
|
|
exit 0
|
|
fi
|
|
|
|
# --- schedule ----------------------------------------------------------------
|
|
|
|
print_info "Schedule"
|
|
if ! systemctl list-timers --all 'omarchy-*' --no-pager 2>/dev/null | sed '/^$/d;s/^/ /'; then
|
|
echo " (could not read timers)"
|
|
fi
|
|
echo ""
|
|
|
|
# --- per-unit state ----------------------------------------------------------
|
|
|
|
print_info "Units"
|
|
for unit in "${UNITS[@]}"; do
|
|
# systemctl prints "not-found" AND exits nonzero, so a || fallback would
|
|
# append a second line and wreck the alignment. Take its output as-is.
|
|
enabled=$(systemctl is-enabled "$unit.timer" 2>/dev/null || true)
|
|
enabled=${enabled:-unknown}
|
|
service_state=$(systemctl is-active "$unit.service" 2>/dev/null || true)
|
|
result=$(systemctl show "$unit.service" -p Result --value 2>/dev/null || true)
|
|
exit_status=$(systemctl show "$unit.service" -p ExecMainStatus --value 2>/dev/null || true)
|
|
finished=$(systemctl show "$unit.service" -p ExecMainExitTimestamp --value 2>/dev/null || true)
|
|
|
|
case "$enabled" in
|
|
enabled) mark="✓" ;;
|
|
not-found | unknown) mark="✗" ;;
|
|
*) mark="⚠" ;;
|
|
esac
|
|
|
|
printf ' %s %-28s timer:%-14s' "$mark" "$unit" "$enabled"
|
|
|
|
if [[ "$service_state" == "activating" || "$service_state" == "active" ]]; then
|
|
printf 'RUNNING NOW\n'
|
|
elif [[ -z "$finished" ]]; then
|
|
printf 'never run\n'
|
|
elif [[ "$result" == "success" && "$exit_status" == "0" ]]; then
|
|
printf 'last: ok (%s)\n' "$finished"
|
|
else
|
|
printf 'last: FAILED [%s/exit %s] (%s)\n' "${result:-unknown}" "${exit_status:-?}" "$finished"
|
|
fi
|
|
done
|
|
echo ""
|
|
|
|
# --- queued work -------------------------------------------------------------
|
|
|
|
print_info "Queued builds (state files in $STATE_DIR)"
|
|
queued=false
|
|
for channel in edge rc stable; do
|
|
state_file="$STATE_DIR/.sync-needed-$channel"
|
|
if [[ -f "$state_file" ]]; then
|
|
queued=true
|
|
since=$(date -r "$state_file" '+%Y-%m-%d %H:%M:%S' 2>/dev/null || echo "unknown")
|
|
printf ' • %-7s queued since %s\n' "$channel" "$since"
|
|
fi
|
|
done
|
|
[[ "$queued" == false ]] && echo " (nothing queued — all channels up to date)"
|
|
echo ""
|
|
|
|
# A channel in backoff looks identical to an idle one from the outside, so
|
|
# say so plainly: it is queued but deliberately not being retried yet.
|
|
paused=false
|
|
for channel in edge rc stable; do
|
|
fail_file="$STATE_DIR/.build-failed-$channel"
|
|
[[ -f "$fail_file" ]] || continue
|
|
if [[ "$paused" == false ]]; then
|
|
print_error "Failing builds (backoff active)"
|
|
paused=true
|
|
fi
|
|
FAILURE_COUNT=0 FAILURE_AT=0 FAILURE_FINGERPRINT=""
|
|
# shellcheck disable=SC1090
|
|
source "$fail_file" 2>/dev/null || true
|
|
delay=600
|
|
for ((i = 1; i < FAILURE_COUNT; i++)); do
|
|
delay=$((delay * 2))
|
|
((delay >= 21600)) && { delay=21600; break; }
|
|
done
|
|
retry_at=$((FAILURE_AT + delay))
|
|
now=$(date +%s)
|
|
if ((now < retry_at)); then
|
|
when="retries at $(date -d "@$retry_at" '+%H:%M:%S' 2>/dev/null || echo "+$((retry_at - now))s")"
|
|
else
|
|
when="retries on the next tick"
|
|
fi
|
|
printf ' ✗ %-7s %s consecutive failure(s), %s\n' "$channel" "$FAILURE_COUNT" "$when"
|
|
printf ' last attempt %s on commit %s\n' \
|
|
"$(date -d "@$FAILURE_AT" '+%Y-%m-%d %H:%M:%S' 2>/dev/null || echo "$FAILURE_AT")" \
|
|
"${FAILURE_FINGERPRINT:0:12}"
|
|
done
|
|
if [[ "$paused" == true ]]; then
|
|
echo " Any new commit clears the backoff; or: rm $STATE_DIR/.build-failed-<channel>"
|
|
echo ""
|
|
fi
|
|
|
|
# --- release lock ------------------------------------------------------------
|
|
|
|
print_info "Release lock"
|
|
lock_file="$REPO_ROOT/.release.lock"
|
|
if [[ -s "$lock_file" ]]; then
|
|
holder=$(tail -1 "$lock_file" 2>/dev/null)
|
|
# A recorded holder whose process is gone is a leftover, not a live release.
|
|
pid=$(sed -n 's/^pid \([0-9]\+\).*/\1/p' <<<"$holder")
|
|
if [[ -n "$pid" ]] && kill -0 "$pid" 2>/dev/null; then
|
|
print_warning "HELD — $holder"
|
|
else
|
|
echo " free (last holder: $holder)"
|
|
fi
|
|
else
|
|
echo " free"
|
|
fi
|
|
echo ""
|
|
|
|
# --- recent failures ---------------------------------------------------------
|
|
|
|
failed=$(systemctl list-units --failed --no-legend 'omarchy-*' 2>/dev/null | sed 's/^/ /')
|
|
if [[ -n "$failed" ]]; then
|
|
print_error "Failed units"
|
|
echo "$failed"
|
|
echo ""
|
|
echo " Logs: journalctl -u <unit> -n 50"
|
|
fi
|
|
|
|
print_info "Full logs: journalctl -u omarchy-auto-release-<channel> -n 50"
|