diff --git a/changelog.d/133.md b/changelog.d/133.md new file mode 100644 index 0000000..fc4f074 --- /dev/null +++ b/changelog.d/133.md @@ -0,0 +1,3 @@ +### Fixed + +- `rig forgejo-runner status` no longer lets a service's `(active)` stand as proof the runner is fetching jobs (#133) diff --git a/commands/forgejo-runner-status.sh b/commands/forgejo-runner-status.sh index f201e9f..146d1bd 100755 --- a/commands/forgejo-runner-status.sh +++ b/commands/forgejo-runner-status.sh @@ -80,6 +80,23 @@ log "name: ${RUNNER_NAME:-unknown}" log "labels: ${LABELS}" log "dir: ${RUNNER_DIR}" log "service: ${SERVICE}" +# `active` is the strongest health signal this command has, and it proves only +# that a process exists — not that the runner is still asking Forgejo for work. +# A poller can go quiet while the daemon stays up: measured 2026-07-30 (#129, +# #133), a daemon logged "[poller] launched" and never fetched a job dispatched +# four minutes later, while a daemon started fresh claimed that same queued +# task in one second. Both times it read as a label-mapping bug on the forge. +# +# log, not warn: nothing has been DETECTED here. An idle runner with no queued +# jobs is silent in exactly the same way a stalled one is, so there is no +# signal separating them — a warning on every status run would be crying wolf, +# and warn in this file means a drift actually measured (the .runner mode +# below). Saying what the signal does not cover is the honest middle. +if [ "${STATE:-}" = active ]; then + log " note: 'active' is not proof the runner is fetching jobs — only that the process is up." + log " If a job stays queued and its run page says it never started, run" + log " 'systemctl restart forgejo-runner' and re-read before suspecting the labels." +fi # status is the only command an operator runs when nothing is obviously wrong, # which makes it the right place to notice a mode that drifted. It reports and diff --git a/test/cli.sh b/test/cli.sh index 4846778..4cf5bcd 100644 --- a/test/cli.sh +++ b/test/cli.sh @@ -3407,6 +3407,26 @@ check "forgejo-runner: install converges the mode on EVERY run, not only at regi fr_secure_every_run check "forgejo-runner: status warns on a drifted mode" 0 "FORGEJO_RUNNER_FILE_MODE" \ grep -o "FORGEJO_RUNNER_FILE_MODE" "$ROOT/commands/forgejo-runner-status.sh" +# #133: `active` is the strongest signal this command has, and it proves only +# that a process exists. A poller can go quiet while the process stays up — +# measured for #129: a daemon logged "[poller] launched" and never fetched a +# job dispatched four minutes later, while a fresh daemon claimed the same +# queued task in one second. status is where an operator looks when nothing +# is obviously wrong, so it says so there. +FJS="$ROOT/commands/forgejo-runner-status.sh" +check "forgejo-runner: status says 'active' is not proof the runner is fetching" 0 "not proof" \ + grep -o "not proof" "$FJS" +check "forgejo-runner: …and names the remedy, so the line is actionable" 0 "restart" \ + grep -oi "systemctl restart forgejo-runner" "$FJS" +# It must NOT be a warning: nothing has been detected. An idle-but-healthy +# runner logs nothing either, so there is no signal that separates it from a +# stalled one — a WARNING on every status run would be crying wolf, and this +# file reserves warn for drift it has actually measured (the .runner mode). +check "forgejo-runner: the liveness note is informational, never a WARNING" 1 "" \ + grep -nE 'warn ".*not proof' "$FJS" +# The header contract at :25 — no token, no network call — survives this. +check "forgejo-runner: status still makes no network call" 1 "" \ + grep -nE '^[^#]*(curl|wget) ' "$FJS" rm -rf "$FRW" # --- the checksum gate, DRIVEN not grepped (review !110) --------------------