Merge pull request 'fix(forgejo-runner): 'active' is not proof the runner is fetching' (#134) from build/133-status-active-is-not-health into main
Some checks are pending
ci / check (push) Waiting to run
ci / install (push) Waiting to run
ci / db-integration (push) Waiting to run
release / release (push) Waiting to run

Reviewed-on: #134
Reviewed-by: kimi-reviewer-andresmgsl <andres+4@heavyduty.builders>
Reviewed-by: grok-reviewer-andresmgsl <andres+3@heavyduty.builders>
This commit is contained in:
andres 2026-07-31 20:55:18 +00:00
commit 306844daa8
3 changed files with 71 additions and 0 deletions

3
changelog.d/133.md Normal file
View file

@ -0,0 +1,3 @@
### Fixed
- `rig forgejo-runner status` no longer lets a service's `(active)` stand as proof the runner is fetching jobs (#133)

View file

@ -11,6 +11,31 @@ log() { printf 'rig-forgejo-runner: %s\n' "$*"; }
warn() { printf 'rig-forgejo-runner: WARNING: %s\n' "$*" >&2; }
die() { printf 'rig-forgejo-runner: ERROR: %s\n' "$1" >&2; exit "${2:-1}"; }
# forgejo_runner_liveness_note <systemctl-state> — what `active` does not cover.
#
# `active` is the strongest health signal this command has, and it proves only
# that a process exists — not that the runner is still asking Forgejo for work.
# A poller can go quiet while the daemon stays up: measured 2026-07-30 (#129,
# #133), a daemon logged "[poller] launched" and never fetched a job dispatched
# four minutes later, while a daemon started fresh claimed that same queued task
# in one second. Both times it read as a label-mapping bug on the forge, which
# is the wrong place to look.
#
# A FUNCTION rather than an inline `if`, because the state boundary is the part
# worth pinning: an absent or inactive unit must say nothing, and a grep over
# the source cannot tell the difference (codex/kimi, !134).
#
# log, not warn: nothing has been DETECTED here. An idle runner with no queued
# jobs is silent in exactly the same way a stalled one is, so there is no signal
# separating them — a warning on every status run would be crying wolf, and warn
# in this file means a drift actually measured (the .runner mode below).
forgejo_runner_liveness_note() {
[ "${1:-}" = active ] || return 0
log " note: 'active' is not proof the runner is fetching jobs — only that the process is up."
log " If a job stays queued and its run page says it never started, run"
log " 'systemctl restart forgejo-runner' and re-read before suspecting the labels."
}
usage() {
cat <<'EOF'
usage: rig forgejo-runner status [--user <name>]
@ -80,6 +105,8 @@ log "name: ${RUNNER_NAME:-unknown}"
log "labels: ${LABELS}"
log "dir: ${RUNNER_DIR}"
log "service: ${SERVICE}"
forgejo_runner_liveness_note "${STATE:-}"
# status is the only command an operator runs when nothing is obviously wrong,
# which makes it the right place to notice a mode that drifted. It reports and

View file

@ -3428,6 +3428,47 @@ check "forgejo-runner: install converges the mode on EVERY run, not only at regi
fr_secure_every_run
check "forgejo-runner: status warns on a drifted mode" 0 "FORGEJO_RUNNER_FILE_MODE" \
grep -o "FORGEJO_RUNNER_FILE_MODE" "$ROOT/commands/forgejo-runner-status.sh"
# #133: `active` is the strongest signal this command has, and it proves only
# that a process exists. A poller can go quiet while the process stays up —
# measured for #129: a daemon logged "[poller] launched" and never fetched a
# job dispatched four minutes later, while a fresh daemon claimed the same
# queued task in one second. status is where an operator looks when nothing
# is obviously wrong, so it says so there.
FJS="$ROOT/commands/forgejo-runner-status.sh"
check "forgejo-runner: status says 'active' is not proof the runner is fetching" 0 "not proof" \
grep -o "not proof" "$FJS"
check "forgejo-runner: …and names the remedy, so the line is actionable" 0 "restart" \
grep -oi "systemctl restart forgejo-runner" "$FJS"
# It must NOT be a warning: nothing has been detected. An idle-but-healthy
# runner logs nothing either, so there is no signal that separates it from a
# stalled one — a WARNING on every status run would be crying wolf, and this
# file reserves warn for drift it has actually measured (the .runner mode).
check "forgejo-runner: the liveness note is informational, never a WARNING" 1 "" \
grep -nE 'warn ".*not proof' "$FJS"
# codex/kimi on !134: the three greps above prove the LINES EXIST; nothing
# proved they fire only when the unit is active. Deleting the state guard left
# the suite 790/790 green, so the acceptance boundary #133 cares about most —
# no misleading liveness note on an absent or inactive unit — was unprotected.
# Drive the decision instead, extracted the way test/drill.sh extracts its own.
# Its own scratch dir: $WORK is rm -rf'd at :3206, well before this block.
FJS_DIR="$(mktemp -d)"
FJSW="$FJS_DIR/fjs-note.sh"
{ printf '%s\n' 'log() { printf "rig-forgejo-runner: %s\\n" "$*"; }'
awk '/^forgejo_runner_liveness_note\(\) \{/,/^\}/' "$FJS"
} > "$FJSW"
check "forgejo-runner: liveness note extracted (guards the awk)" 0 "forgejo_runner_liveness_note() {" \
cat "$FJSW"
note_for() { bash -c '. "$1"; forgejo_runner_liveness_note "$2"' _ "$FJSW" "$1"; }
check "liveness note: an ACTIVE unit is told what active does not prove" 0 "not proof" note_for active
check "liveness note: …and is given the remedy" 0 "systemctl restart forgejo-runner" note_for active
note_is_empty() { [ -z "$(note_for "$1")" ]; }
check "liveness note: an INACTIVE unit gets nothing" 0 "" note_is_empty inactive
check "liveness note: an ABSENT unit (empty state) gets nothing" 0 "" note_is_empty ""
rm -rf "$FJS_DIR"
# The header contract at :25 — no token, no network call — survives this.
check "forgejo-runner: status still makes no network call" 1 "" \
grep -nE '^[^#]*(curl|wget) ' "$FJS"
rm -rf "$FRW"
# --- the checksum gate, DRIVEN not grepped (review !110) --------------------