diff --git a/changelog.d/133.md b/changelog.d/133.md new file mode 100644 index 0000000..fc4f074 --- /dev/null +++ b/changelog.d/133.md @@ -0,0 +1,3 @@ +### Fixed + +- `rig forgejo-runner status` no longer lets a service's `(active)` stand as proof the runner is fetching jobs (#133) diff --git a/commands/forgejo-runner-status.sh b/commands/forgejo-runner-status.sh index f201e9f..6119d47 100755 --- a/commands/forgejo-runner-status.sh +++ b/commands/forgejo-runner-status.sh @@ -11,6 +11,31 @@ log() { printf 'rig-forgejo-runner: %s\n' "$*"; } warn() { printf 'rig-forgejo-runner: WARNING: %s\n' "$*" >&2; } die() { printf 'rig-forgejo-runner: ERROR: %s\n' "$1" >&2; exit "${2:-1}"; } +# forgejo_runner_liveness_note — what `active` does not cover. +# +# `active` is the strongest health signal this command has, and it proves only +# that a process exists — not that the runner is still asking Forgejo for work. +# A poller can go quiet while the daemon stays up: measured 2026-07-30 (#129, +# #133), a daemon logged "[poller] launched" and never fetched a job dispatched +# four minutes later, while a daemon started fresh claimed that same queued task +# in one second. Both times it read as a label-mapping bug on the forge, which +# is the wrong place to look. +# +# A FUNCTION rather than an inline `if`, because the state boundary is the part +# worth pinning: an absent or inactive unit must say nothing, and a grep over +# the source cannot tell the difference (codex/kimi, !134). +# +# log, not warn: nothing has been DETECTED here. An idle runner with no queued +# jobs is silent in exactly the same way a stalled one is, so there is no signal +# separating them — a warning on every status run would be crying wolf, and warn +# in this file means a drift actually measured (the .runner mode below). +forgejo_runner_liveness_note() { + [ "${1:-}" = active ] || return 0 + log " note: 'active' is not proof the runner is fetching jobs — only that the process is up." + log " If a job stays queued and its run page says it never started, run" + log " 'systemctl restart forgejo-runner' and re-read before suspecting the labels." +} + usage() { cat <<'EOF' usage: rig forgejo-runner status [--user ] @@ -80,6 +105,8 @@ log "name: ${RUNNER_NAME:-unknown}" log "labels: ${LABELS}" log "dir: ${RUNNER_DIR}" log "service: ${SERVICE}" +forgejo_runner_liveness_note "${STATE:-}" + # status is the only command an operator runs when nothing is obviously wrong, # which makes it the right place to notice a mode that drifted. It reports and diff --git a/test/cli.sh b/test/cli.sh index fd79ee2..5e010bf 100644 --- a/test/cli.sh +++ b/test/cli.sh @@ -3428,6 +3428,47 @@ check "forgejo-runner: install converges the mode on EVERY run, not only at regi fr_secure_every_run check "forgejo-runner: status warns on a drifted mode" 0 "FORGEJO_RUNNER_FILE_MODE" \ grep -o "FORGEJO_RUNNER_FILE_MODE" "$ROOT/commands/forgejo-runner-status.sh" +# #133: `active` is the strongest signal this command has, and it proves only +# that a process exists. A poller can go quiet while the process stays up — +# measured for #129: a daemon logged "[poller] launched" and never fetched a +# job dispatched four minutes later, while a fresh daemon claimed the same +# queued task in one second. status is where an operator looks when nothing +# is obviously wrong, so it says so there. +FJS="$ROOT/commands/forgejo-runner-status.sh" +check "forgejo-runner: status says 'active' is not proof the runner is fetching" 0 "not proof" \ + grep -o "not proof" "$FJS" +check "forgejo-runner: …and names the remedy, so the line is actionable" 0 "restart" \ + grep -oi "systemctl restart forgejo-runner" "$FJS" +# It must NOT be a warning: nothing has been detected. An idle-but-healthy +# runner logs nothing either, so there is no signal that separates it from a +# stalled one — a WARNING on every status run would be crying wolf, and this +# file reserves warn for drift it has actually measured (the .runner mode). +check "forgejo-runner: the liveness note is informational, never a WARNING" 1 "" \ + grep -nE 'warn ".*not proof' "$FJS" + +# codex/kimi on !134: the three greps above prove the LINES EXIST; nothing +# proved they fire only when the unit is active. Deleting the state guard left +# the suite 790/790 green, so the acceptance boundary #133 cares about most — +# no misleading liveness note on an absent or inactive unit — was unprotected. +# Drive the decision instead, extracted the way test/drill.sh extracts its own. +# Its own scratch dir: $WORK is rm -rf'd at :3206, well before this block. +FJS_DIR="$(mktemp -d)" +FJSW="$FJS_DIR/fjs-note.sh" +{ printf '%s\n' 'log() { printf "rig-forgejo-runner: %s\\n" "$*"; }' + awk '/^forgejo_runner_liveness_note\(\) \{/,/^\}/' "$FJS" +} > "$FJSW" +check "forgejo-runner: liveness note extracted (guards the awk)" 0 "forgejo_runner_liveness_note() {" \ + cat "$FJSW" +note_for() { bash -c '. "$1"; forgejo_runner_liveness_note "$2"' _ "$FJSW" "$1"; } +check "liveness note: an ACTIVE unit is told what active does not prove" 0 "not proof" note_for active +check "liveness note: …and is given the remedy" 0 "systemctl restart forgejo-runner" note_for active +note_is_empty() { [ -z "$(note_for "$1")" ]; } +check "liveness note: an INACTIVE unit gets nothing" 0 "" note_is_empty inactive +check "liveness note: an ABSENT unit (empty state) gets nothing" 0 "" note_is_empty "" +rm -rf "$FJS_DIR" +# The header contract at :25 — no token, no network call — survives this. +check "forgejo-runner: status still makes no network call" 1 "" \ + grep -nE '^[^#]*(curl|wget) ' "$FJS" rm -rf "$FRW" # --- the checksum gate, DRIVEN not grepped (review !110) --------------------