Files
app-catalog-felhom.eu/scripts/retest-floating.py
T
admin 9e53205938
gates / gates (push) Successful in 1s
retest-floating: check what the bench brings (docker, python3), not what the sync puts there (R-749)
The check asked for /opt/upg/upgrade-test.py before sync_bench() copies it, so a bench freshly created by the
runbook was refused in one minute (2026-10-01). After the sync, the file is now required.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-10-01 07:17:49 +02:00

233 lines
13 KiB
Python

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""retest-floating.py — the MONTHLY re-test of same-tag security fixes (`09` §3 decision 52, R-740). ONE command.
python3 scripts/retest-floating.py --dry-run # list only: which services' registry digest moved
python3 scripts/retest-floating.py [--only app,app] # re-test them: bench, box, writer — one commit per app
python3 scripts/retest-floating.py --push # ... and push each commit (the pre-push gates run)
python3 scripts/retest-floating.py --engines-only ... # database/redis lines only (decision 52's start)
WHAT IT DOES. For every template whose ladder's NEWEST entry is its compose, it asks the registry for each service's
digest NOW and compares it with the digest that entry TESTED. A service whose digest moved is a same-name upstream
fix (typically a database or redis line). Candidates are ordered database/redis first. For each app, in order:
1. BENCH — `upgrade-test.py --soak 600 --retest <app>` on the bench LXC (the full method: FROM at the tested
digest, seed, read back, TO at the new digest, read back, the 10-minute memory watch, the abort), evidence
copied off the bench before the next app;
2. BOX — scratch guest 9202 on the DRILL catalog: a fresh install renders the OLD tested digest (the live ladder
says so), the fixture seeds; a drill-only commit adds the re-test entry; the product's guarded Update runs the
step; the seed reads back; the running digest is checked;
3. WRITER — `upgrade-test.py --write-ladder` from both verdicts into THIS checkout (a re-test entry: `from` ==
`to`, `digest_from`, `box_evidence`), `catalog_gates.py --fast <app>`, one commit (and `--push`).
A failure stops THAT app (its reason in the summary) and never the list. Nothing is written without both venues.
WHAT IT NEEDS (checked first; a missing one stops the run and says which): the bench LXC 9401 on demo-hp with
/opt/upg (runbook `felhom.eu/documentation/runbooks/monthly-floating-retest.md` §2 creates it), guest 9202 pointed
at the drill catalog (§3), SC=<a 0600 dir> holding `.ctlpw` (9202's dashboard password — never committed), ssh to
demo-hp, and a push credential for the drill repo. It is therefore NOT a cron job: it is run by a person or a CC
session once a month, from DooPlex (the runbook says why).
Exit: 0 every candidate done (or none) · 1 at least one app stopped · 2 could not start (a prerequisite).
"""
import argparse
import base64
import datetime
import io
import json
import os
import re
import subprocess
import sys
import tarfile
import time
HERE = os.path.dirname(os.path.abspath(__file__))
CAT = os.path.dirname(HERE)
sys.path.insert(0, HERE)
import ladder # noqa: E402
import image_digest # noqa: E402
ENGINE_RE = re.compile(r"(^|/)(postgres|postgis|mariadb|mysql|redis|valkey|mongo)", re.I)
DRILL = os.environ.get("DRILL", "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill")
BENCH_HOST, BENCH_CT = os.environ.get("BENCH_HOST", "demo-hp"), os.environ.get("BENCH_CT", "9401")
def candidates(only=None):
"""[(app, {svc: (ref, tested, now)}, engine_first)] — every service whose registry digest moved."""
out = []
for app in sorted(os.listdir(os.path.join(CAT, "templates"))):
if only and app not in only:
continue
d = os.path.join(CAT, "templates", app)
try:
entries, _, errs = ladder.parse(open(os.path.join(d, ".felhom.yml"), encoding="utf-8").read())
comp = ladder.images_in(open(os.path.join(d, "docker-compose.yml"), encoding="utf-8").read())
except OSError:
continue
if errs or not entries or entries[-1].get("to") != comp:
continue
tested = entries[-1].get("digest") or {}
moved = {}
for svc, ref in comp.items():
now, why = image_digest.resolve(ref)
if not now:
moved[svc] = (ref, tested.get(svc), "UNKNOWN: " + str(why))
elif tested.get(svc) and now != tested[svc]:
moved[svc] = (ref, tested[svc], now)
if moved:
out.append((app, moved, any(ENGINE_RE.search(r) for r, _, _ in moved.values())))
return sorted(out, key=lambda x: (not x[2], x[0]))
def sh(args, timeout=3600, inp=None):
try:
return subprocess.run(args, capture_output=True, text=True, timeout=timeout, input=inp)
except (subprocess.TimeoutExpired, OSError) as e:
return subprocess.CompletedProcess(args, 124, "", str(e))
def bench(script, timeout=3600):
t = "/tmp/rt-%d-%d.sh" % (os.getpid(), int(time.time() * 1000) % 100000)
return sh(["ssh", "-o", "BatchMode=yes", BENCH_HOST,
"export LC_ALL=C; cat > %s; pct push %s %s %s >/dev/null 2>&1; pct exec %s -- bash %s; rc=$?; "
"pct exec %s -- rm -f %s; rm -f %s; exit $rc" % (t, BENCH_CT, t, t, BENCH_CT, t, BENCH_CT, t, t)],
timeout=timeout, inp=script)
def prerequisites():
missing = []
# R-749: ask for what the bench must BRING (docker, python3) — /opt/upg is what sync_bench() puts there AFTER this
# check; asking for it here refused every freshly created bench.
r = bench("command -v docker >/dev/null && command -v python3 >/dev/null && docker info >/dev/null 2>&1 && echo bench-ok", timeout=120)
if "bench-ok" not in (r.stdout or ""):
missing.append("the bench LXC %s on %s with docker and python3 (runbook §2)" % (BENCH_CT, BENCH_HOST))
if not os.path.isdir(os.path.join(DRILL, ".git")):
missing.append("the drill checkout at %s (runbook §3)" % DRILL)
if not os.path.isfile(os.path.join(os.environ.get("SC", os.path.expanduser("~/.felhom-retest")), ".ctlpw")):
missing.append("SC/.ctlpw — 9202's dashboard password (runbook §3)")
return missing
def sync_bench():
"""This checkout's scripts/*.py and templates/ → /opt/upg on the bench (the harness runs them as they stand)."""
tgz = io.BytesIO()
with tarfile.open(fileobj=tgz, mode="w:gz") as t:
for f in os.listdir(HERE):
if f.endswith(".py"):
t.add(os.path.join(HERE, f), arcname=f)
t.add(os.path.join(CAT, "templates"), arcname="templates")
return sh(["ssh", "-o", "BatchMode=yes", BENCH_HOST,
"cat > /tmp/rt.b64; pct push %s /tmp/rt.b64 /tmp/rt.b64; rm -f /tmp/rt.b64; pct exec %s -- bash -c "
"'mkdir -p /opt/upg && cd /opt/upg && rm -rf templates && base64 -d /tmp/rt.b64 | tar xzf - && rm -f /tmp/rt.b64 && echo synced'"
% (BENCH_CT, BENCH_CT)], timeout=600, inp=base64.b64encode(tgz.getvalue()).decode())
def run_bench(app, evdir):
r = bench("cd /opt/upg && rm -rf evidence/RT-%s && timeout 3000 python3 upgrade-test.py --soak 600 --retest %s "
"> /opt/upg/RT-%s.log 2>&1; echo rc=$?" % (app, app, app), timeout=3400)
pull = bench("cd /opt/upg && tar czf - RT-%s.log evidence/RT-%s 2>/dev/null | base64 -w0" % (app, app), timeout=600)
os.makedirs(evdir, exist_ok=True)
try:
tarfile.open(fileobj=io.BytesIO(base64.b64decode(pull.stdout.strip()))).extractall(evdir)
except Exception as e: # noqa: BLE001
return None, "the bench evidence could not be copied off (%s); bench said %s" % (e, (r.stdout or "").strip()[-80:])
vp = os.path.join(evdir, "evidence", "RT-%s" % app, "verdict.json")
if not os.path.isfile(vp):
return None, "no bench verdict (%s)" % (r.stdout or "").strip()[-120:]
v = json.load(open(vp))
return (vp if v.get("verdict") == "proven" else None), "bench verdict %s (%s)" % (v.get("verdict"), str(v.get("abort_detail"))[:120])
def run_box(app, moved, evdir):
"""The box venue: retest_box.py does it (kept apart: it holds the 9202 session)."""
r = sh([sys.executable, os.path.join(HERE, "retest_box.py"), app, evdir], timeout=3600,
inp=json.dumps({s: [v[0], v[1], v[2]] for s, v in moved.items()}))
vp = os.path.join(evdir, "box-verdict-%s.json" % app)
if not os.path.isfile(vp):
return None, "no box verdict: %s" % ((r.stdout or "") + (r.stderr or "")).strip()[-200:]
v = json.load(open(vp))
return (vp if v.get("verdict") == "proven" else None), "box verdict %s (%s)" % (v.get("verdict"), v.get("why", ""))
def write(app, bench_v, box_v, ev_rel, push):
r = sh([sys.executable, os.path.join(HERE, "upgrade-test.py"), "--write-ladder", bench_v, "--box", box_v,
"--catalog", CAT, "--evidence", ev_rel + "/bench", "--box-evidence", ev_rel + "/box"], timeout=600)
if r.returncode != 0:
return False, "writer: " + (r.stdout or r.stderr).strip()[-200:]
g = sh([sys.executable, os.path.join(HERE, "catalog_gates.py"), "--fast", app], timeout=900)
if g.returncode != 0:
sh(["git", "-C", CAT, "checkout", "--", "templates/%s" % app])
sh(["git", "-C", CAT, "clean", "-fdq", "templates/%s" % app])
return False, "gates refused the entry — reverted: " + g.stdout.strip()[-200:]
sh(["git", "-C", CAT, "add", "templates/%s" % app])
c = sh(["git", "-C", CAT, "commit", "-q", "-m", "%s: re-test of the same tag at a new digest (decision 52, R-740)\n\n%s\n\nEvidence: %s"
% (app, (r.stdout or "").strip(), ev_rel)])
if c.returncode != 0:
return False, "commit failed: " + c.stderr.strip()[-160:]
if push:
p = sh(["git", "-C", CAT, "push", "-q", "origin", "main"], timeout=900)
if p.returncode != 0:
return False, "push refused (the commit stays local): " + (p.stdout + p.stderr).strip()[-200:]
return True, (r.stdout or "").strip().splitlines()[-1]
def main(argv):
ap = argparse.ArgumentParser()
ap.add_argument("--dry-run", action="store_true")
ap.add_argument("--only", default="")
ap.add_argument("--push", action="store_true")
ap.add_argument("--engines-only", action="store_true",
help="only database/redis lines (decision 52: start there). Measured 2026-09-30: EXACT tags are re-pushed "
"under the same name too (nextcloud 34.0.4-apache, linuxserver sonarr 4.0.20) — R-743")
ap.add_argument("--evidence", default=os.path.join(os.environ.get("SC", os.path.expanduser("~/.felhom-retest")), "evidence"),
help="evidence root; copy it into felhom.eu/documentation/audits/ for the record")
ap.add_argument("--evidence-rel", default="", help="the path the ladder entry cites (relative to the workspace root)")
a = ap.parse_args(argv)
only = set(x for x in a.only.split(",") if x)
allc = candidates(only)
cands = [c for c in allc if c[2] or not a.engines_only]
skipped = [c[0] for c in allc if c not in cands]
stamp = datetime.date.today().isoformat()
print("# retest-floating %s — catalog %s" % (stamp, sh(["git", "-C", CAT, "rev-parse", "--short", "HEAD"]).stdout.strip()))
if not cands:
print("nothing to re-test today: every ladder head's tested digest is what the registry serves"
+ (" (engines only — not re-tested, not engines: %s)" % ", ".join(skipped) if skipped else ""))
return 0
for app, moved, eng in cands:
print("%s%s:" % (app, " (engine)" if eng else ""))
for svc, (ref, old, new) in sorted(moved.items()):
print(" %s %s tested %s registry %s" % (svc, ref, (old or "-")[:19], new[:19] if not new.startswith("UNKNOWN") else new))
if a.dry_run:
print("(dry run — nothing re-tested)")
return 0
missing = prerequisites()
if missing:
print("CANNOT START — missing: " + "; ".join(missing))
return 2
s = sync_bench()
if "synced" not in (s.stdout or "") or "bench-ok" not in (bench("test -f /opt/upg/upgrade-test.py && echo bench-ok", timeout=120).stdout or ""):
print("CANNOT START — the bench could not be synced: " + (s.stdout + s.stderr).strip()[-200:])
return 2
results = []
for app, moved, _ in cands:
if any(v[2].startswith("UNKNOWN") for v in moved.values()):
results.append((app, False, "a registry could not be asked — not re-tested"))
continue
evdir = os.path.join(a.evidence, stamp, app)
rel = (a.evidence_rel.rstrip("/") + "/" + app) if a.evidence_rel else evdir
bv, why = run_bench(app, os.path.join(evdir, "bench"))
if not bv:
results.append((app, False, why)); continue
xv, why = run_box(app, moved, os.path.join(evdir, "box"))
if not xv:
results.append((app, False, why)); continue
ok, why = write(app, bv, xv, rel, a.push)
results.append((app, ok, why))
print("\n# summary")
for app, ok, why in results:
print(" %-18s %s %s" % (app, "DONE " if ok else "STOPPED", why))
return 0 if all(ok for _, ok, _ in results) else 1
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))