diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aa2ef1b..40454c5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,3 +20,16 @@ jobs: shellcheck -x "${files[@]}" - name: cli tests run: bash test/cli.sh + + # Kept SEPARATE from `check` on purpose: this job pulls a Postgres image and + # stands up throwaway containers, and a slow image pull must never delay the + # fast shellcheck + cli.sh feedback above. ubuntu-latest ships Docker running + # and passwordless sudo, so test/db-integration.sh EXECUTES here (it only + # skips where Docker is absent). It is the automated proof that dump/restore + # actually round-trips, not just that the args parse. + db-integration: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: db dump/restore round-trip + run: bash test/db-integration.sh diff --git a/README.md b/README.md index f16497c..34da5ce 100644 --- a/README.md +++ b/README.md @@ -187,6 +187,44 @@ gunzip -c | docker exec -i sh -c \ non-emptiness *before* the prompt and before anything touches the database, so a fat-fingered path fails cheaply. +#### Verifying a dump/restore actually works + +A `.gz` that opens without error is not proof of a good backup: a `pg_dump` +truncated mid-stream still compresses into a perfectly valid gzip file that +*looks* exactly like a complete one. The same ethos as the nightly dump applies +here — **a backup you have never read back is not yet a backup.** The only +fully-trustworthy proof is to restore the artifact and read the rows back out. + +On a real Coolify box you can do that **without touching prod data** by +restoring into a fresh *scratch* database rather than over the live one: + +```sh +# 1. dump the live database (read-only; harmless) +rig db dump coolify-db /srv/snapshots/verify.sql.gz + +# 2. create a throwaway database as the container's OWN superuser +docker exec coolify-db sh -c 'createdb -U "$POSTGRES_USER" rig_verify' + +# 3. restore the artifact INTO the scratch db (not the live one) +rig db restore /srv/snapshots/verify.sql.gz coolify-db rig_verify --yes + +# 4. spot-check a table you expect to see +docker exec coolify-db sh -c \ + 'psql -U "$POSTGRES_USER" -d rig_verify -c "\dt"' +docker exec coolify-db sh -c \ + 'psql -U "$POSTGRES_USER" -d rig_verify -c "SELECT count(*) FROM "' + +# 5. drop the scratch db — live data was never touched +docker exec coolify-db sh -c 'dropdb -U "$POSTGRES_USER" rig_verify' +``` + +If the counts and tables are there, the artifact is real. This is exactly the +round-trip `test/db-integration.sh` automates in CI (the `db-integration` job): +it seeds a known table in a source container whose superuser is *not* the +default, dumps it, restores into a second container whose superuser differs, and +asserts the rows and an ordered checksum survived — the same proof, done against +throwaway containers on every push. + ### `rig runner install --repo ` Runner box only, run after `rig bootstrap runner` (the same two-step rhythm diff --git a/test/db-integration.sh b/test/db-integration.sh new file mode 100755 index 0000000..996b218 --- /dev/null +++ b/test/db-integration.sh @@ -0,0 +1,178 @@ +#!/usr/bin/env bash +# test/db-integration.sh — a REAL dump/restore round-trip for `rig db`. +# +# test/cli.sh proves the arg parsing; this proves the actual thing works. It +# stands up throwaway PostgreSQL containers, seeds a known table, runs the real +# `rig db dump` / `rig db restore`, and reads the rows back out the far side — +# because a dump piped through gzip can look perfectly valid while being +# truncated, and only restoring it and reading it back proves otherwise. +# +# It exercises the two invariants db.sh is built around: +# * the source superuser is a NON-default name (src_super), so a green run +# proves the code reads the CONTAINER's own $POSTGRES_USER/$POSTGRES_DB and +# never a hardcoded `postgres`; +# * the destination superuser is a DIFFERENT name (dst_super), so the restore +# only succeeds because --no-owner --no-acl stripped the source role graph — +# a plain dump would abort under ON_ERROR_STOP=1 on the first missing role. +# +# Skips cleanly (exit 0) when it cannot run — no Docker, no reachable daemon, or +# no way to become root (rig db requires root) — so it never reddens a dev box +# that simply has no Docker. On ubuntu-latest CI, Docker is preinstalled and +# running and passwordless sudo works, so it EXECUTES for real there. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +RIG="$ROOT/bin/rig" +PG_IMAGE="${RIG_DBIT_PG_IMAGE:-postgres:16-alpine}" + +PASS=0 FAIL=0 +ok() { echo "ok: $*"; PASS=$((PASS + 1)); } +bad() { echo "FAIL: $*"; FAIL=$((FAIL + 1)); } +skip() { echo "skip: $*"; exit 0; } +die() { echo "FAIL: $*" >&2; exit 1; } # trap still runs cleanup + +# --- privilege / docker preflight ------------------------------------------- +# rig db requires root (require_common_guards). CI's runner user is not root but +# has passwordless sudo and docker-group access; mirror that exactly. EVERYTHING +# that touches Docker or rig goes through as_root so containers and the artifact +# share one owner and cleanup is uniform (root can always reach the socket). +if [ "$(id -u)" -eq 0 ]; then + as_root() { "$@"; } +elif sudo -n true >/dev/null 2>&1; then + as_root() { sudo "$@"; } +else + skip "not root and no passwordless sudo — rig db requires root; cannot run the round-trip here" +fi + +command -v docker >/dev/null 2>&1 || skip "docker not installed — nothing to exercise" +as_root docker info >/dev/null 2>&1 || skip "docker daemon not reachable — skipping the live round-trip" +# Pull up front so an offline box SKIPS (not FAILS): a missing network is not a +# regression in rig. On CI the image is fetched here and the run below is fast. +as_root docker pull "$PG_IMAGE" >/dev/null 2>&1 || skip "could not pull ${PG_IMAGE} (offline?) — skipping" + +# --- unique names + guaranteed cleanup -------------------------------------- +PREFIX="rig_dbit_$$" +SRC="${PREFIX}_src" +DST="${PREFIX}_dst" +WORKDIR="$(mktemp -d)" +ARTIFACT="$WORKDIR/roundtrip.sql.gz" + +cleanup() { + as_root docker rm -f "$SRC" "$DST" >/dev/null 2>&1 || true + as_root rm -rf "$WORKDIR" >/dev/null 2>&1 || true +} +trap cleanup EXIT + +# A stale run must never collide with this one. +as_root docker rm -f "$SRC" "$DST" >/dev/null 2>&1 || true + +# --- helpers ---------------------------------------------------------------- +# Poll until answers a real query as /. pg_isready alone +# reports "ready" during the entrypoint's temp-server phase, so gate on an +# actual SELECT succeeding instead. Bounded (~60s) so a wedged box fails, loudly. +wait_ready() { # + local c="$1" u="$2" d="$3" i=0 + while [ "$i" -lt 60 ]; do + if as_root docker exec "$c" psql -U "$u" -d "$d" -tAc 'SELECT 1' >/dev/null 2>&1; then + return 0 + fi + i=$((i + 1)) + sleep 1 + done + die "container ${c} never started accepting connections as ${u}/${d}" +} + +# An ordered, checksummable fingerprint of the fixture table — the thing that +# must survive the round-trip byte for byte. +fingerprint() { # + as_root docker exec "$1" psql -U "$2" -d "$3" -tAc \ + "SELECT id||'|'||name||'|'||qty FROM widgets ORDER BY id" | md5sum | cut -d' ' -f1 +} +rowcount() { # + as_root docker exec "$1" psql -U "$2" -d "$3" -tAc 'SELECT count(*) FROM widgets' | tr -d '[:space:]' +} + +# --- source: NON-default superuser, seeded fixture -------------------------- +as_root docker run -d --name "$SRC" \ + -e POSTGRES_USER=src_super -e POSTGRES_DB=src_appdb -e POSTGRES_PASSWORD=srcpw \ + "$PG_IMAGE" >/dev/null +wait_ready "$SRC" src_super src_appdb + +as_root docker exec "$SRC" psql -U src_super -d src_appdb -v ON_ERROR_STOP=1 -c " + CREATE TABLE widgets (id int PRIMARY KEY, name text NOT NULL, qty int NOT NULL); + INSERT INTO widgets VALUES (1,'alpha',10),(2,'beta',20),(3,'gamma',30); +" >/dev/null + +# assert_eq +assert_eq() { + if [ "$2" = "$3" ]; then ok "$1"; else bad "$1 — wanted [$2] got [$3]"; fi +} + +SRC_FP="$(fingerprint "$SRC" src_super src_appdb)" +assert_eq "source seeded with 3 known rows" 3 "$(rowcount "$SRC" src_super src_appdb)" + +# --- dump (explicit outfile) ------------------------------------------------ +if as_root "$RIG" db dump "$SRC" "$ARTIFACT" >/dev/null 2>&1; then + ok "rig db dump exited 0" +else + bad "rig db dump failed" +fi +if as_root test -s "$ARTIFACT"; then + ok "dump wrote a non-empty artifact" +else + die "dump produced no usable artifact — nothing to restore" +fi + +# --- dump (default outfile naming: -.sql.gz) ------- +( cd "$WORKDIR" && as_root "$RIG" db dump "$SRC" >/dev/null 2>&1 ) +if compgen -G "$WORKDIR/${SRC}-*.sql.gz" >/dev/null; then + ok "dump with no outfile wrote -.sql.gz in cwd" +else + bad "default dump name not produced" +fi + +# --- destination: DIFFERENT superuser (the portability proof) --------------- +as_root docker run -d --name "$DST" \ + -e POSTGRES_USER=dst_super -e POSTGRES_DB=dst_appdb -e POSTGRES_PASSWORD=dstpw \ + "$PG_IMAGE" >/dev/null +wait_ready "$DST" dst_super dst_appdb + +# --- restore into the default database -------------------------------------- +if as_root "$RIG" db restore "$ARTIFACT" "$DST" --yes >/dev/null 2>&1; then + ok "rig db restore exited 0 across a differing superuser" +else + bad "rig db restore failed (a role-graph leak would abort here)" +fi + +DST_FP="$(fingerprint "$DST" dst_super dst_appdb)" +assert_eq "destination has exactly 3 rows after restore" 3 "$(rowcount "$DST" dst_super dst_appdb)" +if [ "$DST_FP" = "$SRC_FP" ]; then + ok "restored fingerprint matches source ($SRC_FP)" +else + bad "restored data differs — src=$SRC_FP dst=$DST_FP" +fi + +# --- idempotency: --clean --if-exists means a second restore is a no-op-ish -- +if as_root "$RIG" db restore "$ARTIFACT" "$DST" --yes >/dev/null 2>&1; then + ok "second restore exited 0 (dump is --clean --if-exists)" +else + bad "second restore failed" +fi +assert_eq "still exactly 3 rows after re-restore (no duplication)" 3 "$(rowcount "$DST" dst_super dst_appdb)" +assert_eq "fingerprint stable after re-restore" "$SRC_FP" "$(fingerprint "$DST" dst_super dst_appdb)" + +# --- restore into a NAMED scratch db (the [db] arg / shared-Postgres path) --- +# This is exactly the safe manual proof the README documents: restore into a +# fresh scratch database rather than over live data. +as_root docker exec "$DST" createdb -U dst_super rig_verify >/dev/null +if as_root "$RIG" db restore "$ARTIFACT" "$DST" rig_verify --yes >/dev/null 2>&1; then + ok "rig db restore into a named [db] exited 0" +else + bad "restore into named database failed" +fi +assert_eq "named-database restore reproduced the source fingerprint" \ + "$SRC_FP" "$(fingerprint "$DST" dst_super rig_verify)" + +echo "---" +echo "$PASS passed, $FAIL failed" +[ "$FAIL" -eq 0 ]