#!/usr/bin/env python3 # -*- coding: utf-8 -*- """retest-floating.py — the MONTHLY re-test of same-tag security fixes (`09` §3 decision 52, R-740). ONE command. python3 scripts/retest-floating.py --dry-run # list only: which services' registry digest moved python3 scripts/retest-floating.py [--only app,app] # re-test them: bench, box, writer — one commit per app python3 scripts/retest-floating.py --push # ... and push each commit (the pre-push gates run) python3 scripts/retest-floating.py --engines-only ... # database/redis lines only (decision 52's start) WHAT IT DOES. For every template whose ladder's NEWEST entry is its compose, it asks the registry for each service's digest NOW and compares it with the digest that entry TESTED. A service whose digest moved is a same-name upstream fix (typically a database or redis line). Candidates are ordered database/redis first. For each app, in order: 1. BENCH — `upgrade-test.py --soak 600 --retest ` on the bench LXC (the full method: FROM at the tested digest, seed, read back, TO at the new digest, read back, the 10-minute memory watch, the abort), evidence copied off the bench before the next app; 2. BOX — scratch guest 9202 on the DRILL catalog: a fresh install renders the OLD tested digest (the live ladder says so), the fixture seeds; a drill-only commit adds the re-test entry; the product's guarded Update runs the step; the seed reads back; the running digest is checked; 3. WRITER — `upgrade-test.py --write-ladder` from both verdicts into THIS checkout (a re-test entry: `from` == `to`, `digest_from`, `box_evidence`), `catalog_gates.py --fast `, one commit (and `--push`). A failure stops THAT app (its reason in the summary) and never the list. Nothing is written without both venues. WHAT IT NEEDS (checked first; a missing one stops the run and says which): the bench LXC 9401 on demo-hp with /opt/upg (runbook `felhom.eu/documentation/runbooks/monthly-floating-retest.md` §2 creates it), guest 9202 pointed at the drill catalog (§3), SC= holding `.ctlpw` (9202's dashboard password — never committed), ssh to demo-hp, and a push credential for the drill repo. It is therefore NOT a cron job: it is run by a person or a CC session once a month, from DooPlex (the runbook says why). Exit: 0 every candidate done (or none) · 1 at least one app stopped · 2 could not start (a prerequisite). """ import argparse import base64 import datetime import io import json import os import re import subprocess import sys import tarfile import time HERE = os.path.dirname(os.path.abspath(__file__)) CAT = os.path.dirname(HERE) sys.path.insert(0, HERE) import ladder # noqa: E402 import image_digest # noqa: E402 ENGINE_RE = re.compile(r"(^|/)(postgres|postgis|mariadb|mysql|redis|valkey|mongo)", re.I) DRILL = os.environ.get("DRILL", "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill") BENCH_HOST, BENCH_CT = os.environ.get("BENCH_HOST", "demo-hp"), os.environ.get("BENCH_CT", "9401") def candidates(only=None): """[(app, {svc: (ref, tested, now)}, engine_first)] — every service whose registry digest moved.""" out = [] for app in sorted(os.listdir(os.path.join(CAT, "templates"))): if only and app not in only: continue d = os.path.join(CAT, "templates", app) try: entries, _, errs = ladder.parse(open(os.path.join(d, ".felhom.yml"), encoding="utf-8").read()) comp = ladder.images_in(open(os.path.join(d, "docker-compose.yml"), encoding="utf-8").read()) except OSError: continue if errs or not entries or entries[-1].get("to") != comp: continue tested = entries[-1].get("digest") or {} moved = {} for svc, ref in comp.items(): now, why = image_digest.resolve(ref) if not now: moved[svc] = (ref, tested.get(svc), "UNKNOWN: " + str(why)) elif tested.get(svc) and now != tested[svc]: moved[svc] = (ref, tested[svc], now) if moved: out.append((app, moved, any(ENGINE_RE.search(r) for r, _, _ in moved.values()))) return sorted(out, key=lambda x: (not x[2], x[0])) def sh(args, timeout=3600, inp=None): try: return subprocess.run(args, capture_output=True, text=True, timeout=timeout, input=inp) except (subprocess.TimeoutExpired, OSError) as e: return subprocess.CompletedProcess(args, 124, "", str(e)) def bench(script, timeout=3600): t = "/tmp/rt-%d-%d.sh" % (os.getpid(), int(time.time() * 1000) % 100000) return sh(["ssh", "-o", "BatchMode=yes", BENCH_HOST, "export LC_ALL=C; cat > %s; pct push %s %s %s >/dev/null 2>&1; pct exec %s -- bash %s; rc=$?; " "pct exec %s -- rm -f %s; rm -f %s; exit $rc" % (t, BENCH_CT, t, t, BENCH_CT, t, BENCH_CT, t, t)], timeout=timeout, inp=script) def prerequisites(): missing = [] # R-749: ask for what the bench must BRING (docker, python3) — /opt/upg is what sync_bench() puts there AFTER this # check; asking for it here refused every freshly created bench. r = bench("command -v docker >/dev/null && command -v python3 >/dev/null && docker info >/dev/null 2>&1 && echo bench-ok", timeout=120) if "bench-ok" not in (r.stdout or ""): missing.append("the bench LXC %s on %s with docker and python3 (runbook §2)" % (BENCH_CT, BENCH_HOST)) if not os.path.isdir(os.path.join(DRILL, ".git")): missing.append("the drill checkout at %s (runbook §3)" % DRILL) if not os.path.isfile(os.path.join(os.environ.get("SC", os.path.expanduser("~/.felhom-retest")), ".ctlpw")): missing.append("SC/.ctlpw — 9202's dashboard password (runbook §3)") return missing def sync_bench(): """This checkout's scripts/*.py and templates/ → /opt/upg on the bench (the harness runs them as they stand).""" tgz = io.BytesIO() with tarfile.open(fileobj=tgz, mode="w:gz") as t: for f in os.listdir(HERE): if f.endswith(".py"): t.add(os.path.join(HERE, f), arcname=f) t.add(os.path.join(CAT, "templates"), arcname="templates") return sh(["ssh", "-o", "BatchMode=yes", BENCH_HOST, "cat > /tmp/rt.b64; pct push %s /tmp/rt.b64 /tmp/rt.b64; rm -f /tmp/rt.b64; pct exec %s -- bash -c " "'mkdir -p /opt/upg && cd /opt/upg && rm -rf templates && base64 -d /tmp/rt.b64 | tar xzf - && rm -f /tmp/rt.b64 && echo synced'" % (BENCH_CT, BENCH_CT)], timeout=600, inp=base64.b64encode(tgz.getvalue()).decode()) def run_bench(app, evdir): r = bench("cd /opt/upg && rm -rf evidence/RT-%s && timeout 3000 python3 upgrade-test.py --soak 600 --retest %s " "> /opt/upg/RT-%s.log 2>&1; echo rc=$?" % (app, app, app), timeout=3400) pull = bench("cd /opt/upg && tar czf - RT-%s.log evidence/RT-%s 2>/dev/null | base64 -w0" % (app, app), timeout=600) os.makedirs(evdir, exist_ok=True) try: tarfile.open(fileobj=io.BytesIO(base64.b64decode(pull.stdout.strip()))).extractall(evdir) except Exception as e: # noqa: BLE001 return None, "the bench evidence could not be copied off (%s); bench said %s" % (e, (r.stdout or "").strip()[-80:]) vp = os.path.join(evdir, "evidence", "RT-%s" % app, "verdict.json") if not os.path.isfile(vp): return None, "no bench verdict (%s)" % (r.stdout or "").strip()[-120:] v = json.load(open(vp)) return (vp if v.get("verdict") == "proven" else None), "bench verdict %s (%s)" % (v.get("verdict"), str(v.get("abort_detail"))[:120]) def run_box(app, moved, evdir): """The box venue: retest_box.py does it (kept apart: it holds the 9202 session).""" r = sh([sys.executable, os.path.join(HERE, "retest_box.py"), app, evdir], timeout=3600, inp=json.dumps({s: [v[0], v[1], v[2]] for s, v in moved.items()})) vp = os.path.join(evdir, "box-verdict-%s.json" % app) if not os.path.isfile(vp): return None, "no box verdict: %s" % ((r.stdout or "") + (r.stderr or "")).strip()[-200:] v = json.load(open(vp)) return (vp if v.get("verdict") == "proven" else None), "box verdict %s (%s)" % (v.get("verdict"), v.get("why", "")) def write(app, bench_v, box_v, ev_rel, push): r = sh([sys.executable, os.path.join(HERE, "upgrade-test.py"), "--write-ladder", bench_v, "--box", box_v, "--catalog", CAT, "--evidence", ev_rel + "/bench", "--box-evidence", ev_rel + "/box"], timeout=600) if r.returncode != 0: return False, "writer: " + (r.stdout or r.stderr).strip()[-200:] g = sh([sys.executable, os.path.join(HERE, "catalog_gates.py"), "--fast", app], timeout=900) if g.returncode != 0: sh(["git", "-C", CAT, "checkout", "--", "templates/%s" % app]) sh(["git", "-C", CAT, "clean", "-fdq", "templates/%s" % app]) return False, "gates refused the entry — reverted: " + g.stdout.strip()[-200:] sh(["git", "-C", CAT, "add", "templates/%s" % app]) c = sh(["git", "-C", CAT, "commit", "-q", "-m", "%s: re-test of the same tag at a new digest (decision 52, R-740)\n\n%s\n\nEvidence: %s" % (app, (r.stdout or "").strip(), ev_rel)]) if c.returncode != 0: return False, "commit failed: " + c.stderr.strip()[-160:] if push: p = sh(["git", "-C", CAT, "push", "-q", "origin", "main"], timeout=900) if p.returncode != 0: return False, "push refused (the commit stays local): " + (p.stdout + p.stderr).strip()[-200:] return True, (r.stdout or "").strip().splitlines()[-1] def main(argv): ap = argparse.ArgumentParser() ap.add_argument("--dry-run", action="store_true") ap.add_argument("--only", default="") ap.add_argument("--push", action="store_true") ap.add_argument("--engines-only", action="store_true", help="only database/redis lines (decision 52: start there). Measured 2026-09-30: EXACT tags are re-pushed " "under the same name too (nextcloud 34.0.4-apache, linuxserver sonarr 4.0.20) — R-743") ap.add_argument("--evidence", default=os.path.join(os.environ.get("SC", os.path.expanduser("~/.felhom-retest")), "evidence"), help="evidence root; copy it into felhom.eu/documentation/audits/ for the record") ap.add_argument("--evidence-rel", default="", help="the path the ladder entry cites (relative to the workspace root)") a = ap.parse_args(argv) only = set(x for x in a.only.split(",") if x) allc = candidates(only) cands = [c for c in allc if c[2] or not a.engines_only] skipped = [c[0] for c in allc if c not in cands] stamp = datetime.date.today().isoformat() print("# retest-floating %s — catalog %s" % (stamp, sh(["git", "-C", CAT, "rev-parse", "--short", "HEAD"]).stdout.strip())) if not cands: print("nothing to re-test today: every ladder head's tested digest is what the registry serves" + (" (engines only — not re-tested, not engines: %s)" % ", ".join(skipped) if skipped else "")) return 0 for app, moved, eng in cands: print("%s%s:" % (app, " (engine)" if eng else "")) for svc, (ref, old, new) in sorted(moved.items()): print(" %s %s tested %s registry %s" % (svc, ref, (old or "-")[:19], new[:19] if not new.startswith("UNKNOWN") else new)) if a.dry_run: print("(dry run — nothing re-tested)") return 0 missing = prerequisites() if missing: print("CANNOT START — missing: " + "; ".join(missing)) return 2 s = sync_bench() if "synced" not in (s.stdout or "") or "bench-ok" not in (bench("test -f /opt/upg/upgrade-test.py && echo bench-ok", timeout=120).stdout or ""): print("CANNOT START — the bench could not be synced: " + (s.stdout + s.stderr).strip()[-200:]) return 2 results = [] for app, moved, _ in cands: if any(v[2].startswith("UNKNOWN") for v in moved.values()): results.append((app, False, "a registry could not be asked — not re-tested")) continue evdir = os.path.join(a.evidence, stamp, app) rel = (a.evidence_rel.rstrip("/") + "/" + app) if a.evidence_rel else evdir bv, why = run_bench(app, os.path.join(evdir, "bench")) if not bv: results.append((app, False, why)); continue xv, why = run_box(app, moved, os.path.join(evdir, "box")) if not xv: results.append((app, False, why)); continue ok, why = write(app, bv, xv, rel, a.push) results.append((app, ok, why)) print("\n# summary") for app, ok, why in results: print(" %-18s %s %s" % (app, "DONE " if ok else "STOPPED", why)) return 0 if all(ok for _, ok, _ in results) else 1 if __name__ == "__main__": sys.exit(main(sys.argv[1:]))