From 2cc3510d2a42a1ee198df55bcc729c6abd0b0cfc Mon Sep 17 00:00:00 2001 From: David Heinemeier Hansson Date: Wed, 12 Aug 2026 18:37:40 +0200 Subject: [PATCH] Offer an AI diagnosis when a process crashes (#6746) * Offer an AI diagnosis when a process crashes systemd-coredump journals every core dump under a known MESSAGE_ID with the crashing program, pid, and signal as structured fields. omarchy-crash-watch follows that stream and raises a "Process crashed: " toast; clicking it opens omarchy-agent-crash, which briefs the default agent on the crash. The toast goes through omarchy-notification-send --exec rather than a libnotify action, because the shell runs clicks from its own omarchy-exec hint and never emits ActionInvoked. It keeps the default "omarchy-action" app name too, the only one shouldBypassDnd() lets through -- a crash being the last notification worth swallowing. It stays quiet until an agent is configured, since a diagnosis is all it offers. The method lives in a diagnose-crash skill rather than the prompt, so it is edited in one place and works with whichever agent is default. It covers investigating the core, and reporting a confirmed Omarchy bug upstream: scoped to bugs Omarchy controls, searched for duplicates first, only with the user's agreement, and signed with the model and harness that produced it. A migration reaches existing installs, whose skill symlinks and unit enablement would otherwise sit behind one-time setup paths. * Let the diagnosis clean up the core it extracted "Do not modify or delete anything" contradicted the symbolization step right above it, which writes a core to a temp file and deletes it on exit. Read literally, the core survives -- and the same section warns it holds passwords and tokens. The prohibition is about the system, not about your own scratch. * Do not spend a crash toast on a dead notification server The shell owns org.freedesktop.Notifications, so its own crash takes the notification server down with it -- and a shell crash is exactly what you want told about. The toast was sent once into that gap and the dedupe window was recorded regardless, so the rest of the crash loop went quiet for a minute and `journalctl -n 0` never replays what was missed. It now waits for the restarted shell to reclaim the bus name, as omarchy-migrate-notify already does, and only a delivered toast starts the dedupe window. --- bin/omarchy-agent-crash | 52 +++++++++ bin/omarchy-crash-watch | 86 +++++++++++++++ bin/omarchy-provision-user | 13 ++- default/agents/skills/diagnose-crash/SKILL.md | 97 ++++++++++++++++ .../agents/skills/diagnose-crash/reporting.md | 104 ++++++++++++++++++ .../systemd/user/omarchy-crash-watch.service | 16 +++ install/user/first-run/enable-user-units.sh | 3 +- migrations/1786539345.sh | 31 ++++++ test/shell.d/config-test.sh | 1 + 9 files changed, 398 insertions(+), 5 deletions(-) create mode 100755 bin/omarchy-agent-crash create mode 100755 bin/omarchy-crash-watch create mode 100644 default/agents/skills/diagnose-crash/SKILL.md create mode 100644 default/agents/skills/diagnose-crash/reporting.md create mode 100644 default/systemd/user/omarchy-crash-watch.service create mode 100644 migrations/1786539345.sh diff --git a/bin/omarchy-agent-crash b/bin/omarchy-agent-crash new file mode 100755 index 00000000..49a1147e --- /dev/null +++ b/bin/omarchy-agent-crash @@ -0,0 +1,52 @@ +#!/bin/bash + +# omarchy:summary=Diagnose a crashed process with the default coding agent +# omarchy:args= [comm] [exe] [signal] +# omarchy:examples=omarchy agent crash 1516893 + +# Clicked from a "Process crashed:" notification, or run by hand against any PID +# in `coredumpctl list`. The method lives in the diagnose-crash skill so it is +# edited in one place and works with whichever agent is default; this only +# gathers the facts and points at it. + +set -euo pipefail + +pid=${1:?usage: omarchy-agent-crash [comm] [exe] [signal]} + +if [[ ! $pid =~ ^[0-9]+$ ]]; then + echo "Not a PID: $pid" >&2 + echo "Usage: omarchy agent crash (see: coredumpctl list)" >&2 + exit 1 +fi + +comm=${2:-unknown} +exe=${3:-unknown} +signal=${4:-unknown} + +skill="$OMARCHY_PATH/default/agents/skills/diagnose-crash/SKILL.md" + +# Looked up live so a hand-run PID still gets a timestamp. A rotated-away core +# only costs the timestamp, so failure is tolerated. +when=$(coredumpctl list "$pid" --no-pager --no-legend 2>/dev/null | tail -1 | cut -d' ' -f1-4) || true +when=${when:-unknown} + +prompt=$( + cat </dev/null | + while IFS= read -r entry; do + IFS=$'\t' read -r uid comm pid exe signal < <( + jq -r '[(._UID // "-"), + (.COREDUMP_COMM // "-"), + (.COREDUMP_PID // "-"), + (.COREDUMP_EXE // "-"), + (.COREDUMP_SIGNAL_NAME // "-")] | @tsv' <<<"$entry" 2>/dev/null + ) + + [[ $pid =~ ^[0-9]+$ ]] || continue + + # The toast only offers a diagnosis, so it has nothing to offer until an + # agent is chosen. Checked per crash, not at startup, so picking one takes + # effect without restarting this service. + [[ -n $(omarchy-default-agent) ]] || continue + + # Only this user's crashes; a daemon dumping core is a sysadmin's problem. + [[ $uid =~ ^[0-9]+$ ]] || continue + ((uid == UID)) || continue + + # comm is truncated to 15 characters, so prefer the executable's basename. + name=$comm + [[ $exe == /* ]] && name=${exe##*/} + + [[ -n $ignore_pattern && $name =~ $ignore_pattern ]] && continue + + # Never announce our own machinery, or it notifies about itself. + [[ $name == omarchy-crash-* || $name == omarchy-agent-* ]] && continue + + now=$EPOCHSECONDS + (((now - ${last_notified[$name]:-0}) < dedupe_seconds)) && continue + + # Only a delivered toast starts the dedupe window. A failed send that + # counted would suppress the rest of a crash loop for a minute, and + # `journalctl -n 0` never replays what was missed. + announce "$name" "$pid" "$exe" "$signal" && last_notified[$name]=$now + done diff --git a/bin/omarchy-provision-user b/bin/omarchy-provision-user index b9094f72..0d590496 100755 --- a/bin/omarchy-provision-user +++ b/bin/omarchy-provision-user @@ -83,11 +83,16 @@ fi # Dev-aware skill symlinks. Cannot live in /etc/skel because OMARCHY_PATH may # point at a dev checkout (omarchy dev link) where the target differs. +# Loops every skill directory, so shipping a new one needs no edit here. mkdir -p ~/.agents/skills ~/.claude/skills ~/.codex/skills ~/.pi/agent/skills -ln -sfn "$OMARCHY_PATH/default/agents/skills/omarchy" ~/.agents/skills/omarchy -ln -sfn "$OMARCHY_PATH/default/agents/skills/omarchy" ~/.claude/skills/omarchy -ln -sfn "$OMARCHY_PATH/default/agents/skills/omarchy" ~/.codex/skills/omarchy -ln -sfn "$OMARCHY_PATH/default/agents/skills/omarchy" ~/.pi/agent/skills/omarchy +for skill in "$OMARCHY_PATH"/default/agents/skills/*/; do + skill=${skill%/} + name=${skill##*/} + ln -sfn "$skill" ~/.agents/skills/"$name" + ln -sfn "$skill" ~/.claude/skills/"$name" + ln -sfn "$skill" ~/.codex/skills/"$name" + ln -sfn "$skill" ~/.pi/agent/skills/"$name" +done mkdir -p ~/Downloads ~/Pictures ~/Videos ~/.config/gtk-3.0 xdg-user-dirs-update --set TEMPLATES "$HOME" diff --git a/default/agents/skills/diagnose-crash/SKILL.md b/default/agents/skills/diagnose-crash/SKILL.md new file mode 100644 index 00000000..7859c6ec --- /dev/null +++ b/default/agents/skills/diagnose-crash/SKILL.md @@ -0,0 +1,97 @@ +--- +name: diagnose-crash +description: > + Diagnose why a program crashed on this machine, from a systemd-coredump core dump. + Use when a process has segfaulted, aborted, or otherwise dumped core, when asked + why an application crashed or disappeared, or when a "Process crashed:" desktop + notification is acted on. Triggers: crash, segfault, SIGSEGV, SIGABRT, core dump, + coredumpctl, "why did X crash", "X keeps crashing", backtrace symbolization. + Covers reporting a confirmed Omarchy bug upstream — see reporting.md. +--- + +# Diagnosing a Crash + +Work from evidence. The goal is an honest account of what happened, not a +plausible-sounding story. + +## Establish the facts + +`coredumpctl info ` is the starting point. Beyond the backtrace, note the +**command line** the process was started with — it usually reveals what the +program was working on when it died, which is often the whole answer. + +`coredumpctl list` shows whether this crash is a one-off or a pattern. Repeated +crashes of the same program, or several programs dying together, point somewhere +different than a single failure does. + +## Rule out the boring causes first + +Check resource exhaustion before blaming the program: `free -h`, and the journal +for OOM kills. A process killed by the OOM killer is not a bug in that process. + +## Correlate against the timeline + +The crash timestamp is the most underused piece of evidence. Compare it against: + +- **Filesystem mtimes.** A directory or file whose mtime lands on the same second + as the crash strongly suggests what triggered it. +- **The journal** around that moment, for related warnings from the same or + neighbouring processes. +- **Recent package updates.** A crash that starts right after an update points at + the update. + +## Read the whole core, not just frame 0 + +Thread stacks other than the crashing one show what work was **in flight** — +thumbnailers, image loaders, IPC readers, GPU queues. That context often explains +the trigger even when the crashing frame itself cannot be symbolized. + +Note any third-party code in the address space: file-manager or browser +extensions, plugins, out-of-tree drivers. In-process third-party code is a common +crash source and worth flagging — but do not pin blame on it without evidence +that it is actually implicated. + +## Symbolize when you can + +This is Arch, which runs a public debuginfod server: + +```bash +core=$(mktemp -t crash-XXXXXX.core) +trap 'rm -f "$core"' EXIT +coredumpctl dump --output="$core" +DEBUGINFOD_URLS="https://debuginfod.archlinux.org" \ + gdb -q "$core" \ + -batch -ex 'set debuginfod enabled on' -ex 'bt' +``` + +A core is a verbatim copy of the process's memory and can hold passwords, tokens, +and private documents. Write it to a fresh `mktemp` path rather than a predictable +shared one, and delete it when you are done — never leave it lying in `/tmp`. + +Many packages publish no debug symbols. When frames stay unresolved, say so — +never invent function names to fill the gap. An unsymbolized stack still has +shape: which library each frame belongs to, and whether the crash came from a +signal handler, a main loop, or a worker thread. + +## Report + +1. What crashed, and what it was doing at the time. +2. The most likely mechanism — separating clearly what the evidence **proves** + from what you are **inferring**. +3. Whether any user data was lost, and where it can be recovered from. Check the + trash before concluding anything is gone. +4. Whether it is likely to recur, and what would avoid or fix it. + +Be straight about the limits of the evidence. If the cause is genuinely +ambiguous, say so rather than assembling confidence out of guesswork. + +**Leave the system as you found it.** Diagnosis reads; it does not fix, tidy, or +reconfigure. The one thing to clean up is your own: delete the core you extracted +above, which is a copy of the crashed process's memory. + +## If it is an Omarchy bug + +Most application crashes are upstream bugs in those applications, not Omarchy's +doing. In the minority of cases where the cause really does sit within Omarchy's +sphere of control, read [`reporting.md`](reporting.md) before offering to file +anything. diff --git a/default/agents/skills/diagnose-crash/reporting.md b/default/agents/skills/diagnose-crash/reporting.md new file mode 100644 index 00000000..154f1a25 --- /dev/null +++ b/default/agents/skills/diagnose-crash/reporting.md @@ -0,0 +1,104 @@ +# Reporting a Crash Upstream to Omarchy + +Read this only after concluding that a crash is genuinely Omarchy's to fix. + +## Is it even Omarchy's bug? + +Be strict here. Omarchy is a configuration layer over Arch Linux, so a crash +inside a third-party application — a file manager, a browser, a GNOME or Qt +library — is almost always an upstream bug in **that** project, not in Omarchy. + +Omarchy's sphere of control is roughly: + +- the `omarchy-*` commands +- the Quickshell shell and its plugins +- the Hyprland and terminal configuration it ships +- its themes +- its install and migration scripts +- how it packages and configures what it installs + +A crash in a program Omarchy merely installs is **not** an Omarchy bug unless +Omarchy's own packaging or configuration is implicated. + +If it is not Omarchy's, say so and stop. Suggesting the right upstream project is +useful; filing there yourself is not part of this. + +## Three conditions, all required + +1. **It is a verified bug in Omarchy's sphere**, established on evidence. Issues + are for verified bugs only. An "is this even a bug?" belongs on the Discord at + ; a feature idea belongs in GitHub Discussions + under Suggestions. +2. **The user has explicitly agreed.** Show them the exact title and body you + propose, and wait for a yes. Never file unprompted. +3. **The machine can file it** — `gh auth status` must succeed. If `gh` is missing + or unauthenticated, do not install or authenticate it. Say so, and hand the + user the finished text to submit themselves. + +## Search before filing + +A duplicate issue costs a maintainer more time than no report at all. + +```bash +gh search issues --repo basecamp/omarchy " crash" +gh issue list --repo basecamp/omarchy --state all --search " " +``` + +Search on the crashing program, the signal, and distinctive symbols from the +backtrace — not on the wording of the title you were about to write. + +`gh search issues` accepts only `open` or `closed` for `--state`, and errors on +anything else. Leaving it off searches both, which is what you want here. + +Include **closed** issues. A matching issue closed as fixed, when the crash still +reproduces on a current system, is a regression — and reporting that is worth far +more than another duplicate. + +## Adding to an existing report + +If a plausible match comes back, read it properly first: + +```bash +gh issue view --repo basecamp/omarchy --comments +``` + +Confirm it is genuinely the same failure. The same program crashing is not the +same bug if the trigger or the stack differs. + +If it is the same, add to that issue rather than opening a new one — but only +when you have something the thread does not already contain: a different +reproduction, a symbolized stack where it has none, a narrower trigger, a version +where it regressed. + +A comment that only says the bug happens to you too is noise. If that is all you +have, tell the user so and file nothing. + +```bash +gh issue comment --repo basecamp/omarchy --body "..." +``` + +## Filing a new issue + +Only when the search turns up nothing that matches: + +```bash +gh issue create --repo basecamp/omarchy --title "..." --body "..." +``` + +Include what happened, what was expected, steps to reproduce, system details from +`omarchy version`, and diagnostics from `omarchy debug --no-sudo --print` (which +also writes `/tmp/omarchy-debug.log`; the interactive `omarchy debug` can upload +it and print a shareable URL worth including). + +`gh` cannot attach media. If a screenshot would help, save one and give the user +the path to drag into the web form. + +## Signing + +End the issue or comment with a line naming the model and agent harness that +produced it, so a human reader knows it was machine-authored: + +> Filed by \ via \. + +Use your actual model and harness names. If you are not certain of them, say so +plainly rather than inventing a version string. diff --git a/default/systemd/user/omarchy-crash-watch.service b/default/systemd/user/omarchy-crash-watch.service new file mode 100644 index 00000000..56cee15c --- /dev/null +++ b/default/systemd/user/omarchy-crash-watch.service @@ -0,0 +1,16 @@ +[Unit] +Description=Announce process crashes and offer an AI diagnosis +# Needs the session bus to notify, and uwsm-app to open the diagnosis terminal. +# Both are up only after graphical-session.target. +After=graphical-session.target +PartOf=graphical-session.target +ConditionEnvironment=WAYLAND_DISPLAY + +[Service] +Type=simple +ExecStart=/usr/bin/omarchy-crash-watch +Restart=always +RestartSec=5 + +[Install] +WantedBy=graphical-session.target diff --git a/install/user/first-run/enable-user-units.sh b/install/user/first-run/enable-user-units.sh index 2af44e88..9865fc20 100755 --- a/install/user/first-run/enable-user-units.sh +++ b/install/user/first-run/enable-user-units.sh @@ -17,4 +17,5 @@ systemctl --user enable --now \ omarchy-recover-internal-monitor.service \ omarchy-sleep-lock.service \ omarchy-migrate-notify.service \ - omarchy-fcitx5.service + omarchy-fcitx5.service \ + omarchy-crash-watch.service diff --git a/migrations/1786539345.sh b/migrations/1786539345.sh new file mode 100644 index 00000000..09e34b1d --- /dev/null +++ b/migrations/1786539345.sh @@ -0,0 +1,31 @@ +echo "Announce process crashes and offer an AI diagnosis" + +# Both halves are set up in paths that only run once -- omarchy-provision-user +# exits after finalize-user is marked, and install/user/first-run is skipped +# after the first login -- so existing installs need them done here. + +skills_source="$OMARCHY_PATH/default/agents/skills" + +if [[ -d $skills_source/diagnose-crash ]]; then + for skills_dir in ~/.agents/skills ~/.claude/skills ~/.codex/skills ~/.pi/agent/skills; do + mkdir -p "$skills_dir" + ln -sfn "$skills_source/diagnose-crash" "$skills_dir/diagnose-crash" + done +fi + +systemctl --user daemon-reload >/dev/null 2>&1 || true + +# `systemctl enable` needs a live user manager, which an update from a TTY does +# not have, so fall back to writing the symlink it would have written. +if ! systemctl --user enable omarchy-crash-watch.service >/dev/null 2>&1; then + wants_dir="$HOME/.config/systemd/user/graphical-session.target.wants" + mkdir -p "$wants_dir" + ln -sfn /usr/lib/systemd/user/omarchy-crash-watch.service \ + "$wants_dir/omarchy-crash-watch.service" +fi + +# Nothing to start into over SSH; the next graphical login handles it. A failed +# start only delays crash toasts, so it stays quiet. +if systemctl --user is-active --quiet graphical-session.target; then + systemctl --user start omarchy-crash-watch.service >/dev/null 2>&1 || true +fi diff --git a/test/shell.d/config-test.sh b/test/shell.d/config-test.sh index 09aaa32f..3add6e30 100755 --- a/test/shell.d/config-test.sh +++ b/test/shell.d/config-test.sh @@ -140,6 +140,7 @@ package_defaults = [ ("default/systemd/user/omarchy-migrate-notify.service", "/usr/lib/systemd/user/omarchy-migrate-notify.service", "systemd/user/omarchy-migrate-notify.service"), ("default/systemd/user/omarchy-tailscale-receive.service", "/usr/lib/systemd/user/omarchy-tailscale-receive.service", "systemd/user/omarchy-tailscale-receive.service"), ("default/systemd/user/omarchy-fcitx5.service", "/usr/lib/systemd/user/omarchy-fcitx5.service", "systemd/user/omarchy-fcitx5.service"), + ("default/systemd/user/omarchy-crash-watch.service", "/usr/lib/systemd/user/omarchy-crash-watch.service", "systemd/user/omarchy-crash-watch.service"), ("default/systemd/zram-generator.conf.d/90-omarchy.conf", "/usr/lib/systemd/zram-generator.conf.d/90-omarchy.conf", "systemd/zram-generator.conf.d/90-omarchy.conf"), ("default/fonts/omarchy/omarchy.ttf", "/usr/share/fonts/omarchy/omarchy.ttf", "omarchy.ttf"), ("default/snapper/root", "/etc/snapper/config-templates/omarchy", "snapper/root"),