From 6a3ead9ebe102989d93e6a779008b8cbd6507d40 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Wed, 30 Sep 2026 23:06:09 +0200 Subject: [PATCH] Same-tag security fixes as tested steps (09 decision 52, R-740): re-test entries, their gates, the monthly command MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - A RE-TEST entry: from == to, digest = the registry's new digest, digest_from = the tested one, box_evidence. ladder.check_entry refuses one with no new digest, no digest_from or no box proof; check-test-record rule 2b ties digest_from to the previous entry's digest; check-test-record-move now judges re-tests too (they change .felhom.yml only — the gate looked at compose moves alone) and refuses a digest the registry no longer serves. Decoys: 8 cases in test_gate_decoys.py, seen red with the rules switched off. - upgrade-test.py --retest [svc]: FROM the ladder head's tested digest TO the registry's current one, the full method; --write-ladder writes a re-test entry (plain refs + digest_from), refusing without the box venue or when the registry moved again. Writer tests, red-proofed. - scripts/retest-floating.py — ONE command: --dry-run lists, --engines-only is the ruled start; bench, box (retest_box.py on 9202 via the drill catalog), writer, gates, one commit per app. box_walk.py moves the box client into the catalog. Run today: nothing to re-test on the database/redis lines. - End to end on 9202 (drill): docmost at the OLD redis digest, the re-test, "run tonight's chain now" -> the leg pressed it, the new digest runs, read back, badge current. - Also: upgrade-test.py BENCH_ENV_OVERRIDES (R-739, wanderer's DB address on the bench, recorded per verdict); test_gate_decoys.py read kimai's tag and date from the clone (red on main since kimai moved). Evidence: felhom.eu/documentation/audits/night-rulings-2026-09-30/A/ Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- REUSE.md | 2 + scripts/box_walk.py | 563 ++++++++++++++++++++++++++++++ scripts/check-test-record-move.py | 47 +++ scripts/check-test-record.py | 16 + scripts/ladder.py | 17 + scripts/retest-floating.py | 230 ++++++++++++ scripts/retest_box.py | 158 +++++++++ scripts/test_gate_decoys.py | 56 ++- scripts/test_ladder_writer.py | 43 +++ scripts/upgrade-test.py | 115 +++++- 10 files changed, 1226 insertions(+), 21 deletions(-) create mode 100644 scripts/box_walk.py create mode 100644 scripts/retest-floating.py create mode 100644 scripts/retest_box.py diff --git a/REUSE.md b/REUSE.md index 60c32b1..25cce94 100644 --- a/REUSE.md +++ b/REUSE.md @@ -25,6 +25,8 @@ Templates are config; the few script helpers other scripts must REUSE, never re- | **Open first-run screen → `setup_gate:`** (decision 46, controller ≥ 0.280.0) | `templates/n8n/.felhom.yml` (probe), `templates/uptime-kuma/.felhom.yml` (no probe → the household's button); `FIRST-ADMIN.md` | `setup_gate: true` + optional `setup_done_probe: {url: http://:/, field: , done: ""}` | TRAPS: the probe must FLIP on the setup — measure it before and after on 9202; an app with open sign-up after setup (R-711) is not closed by the gate; `url` is read on the docker network, so it names the container, not the subdomain. | | **Open sign-up after the setup → `signup_block:`** (decision 47, controller ≥ 0.281.0) | `templates/opengist/.felhom.yml`, `templates/calcom/.felhom.yml` | `signup_block: ""` + `app_info.add_people` (hu) / `i18n.en.app_info.add_people` | TRAPS: block the app's API sign-up call, not only the page; an app's own invite link often uses the same address (the household's 15-minute window covers it); `add_people` is copy — `--capture-freeze`. | | **The app's own sign-up switch → `after_setup:`** (decisions 47/49, controller ≥ 0.282.0) | `templates/homebox/` (compose `HBOX_OPTIONS_ALLOW_REGISTRATION=${SIGNUP_OPEN:-true}` + `.felhom.yml` `after_setup.env`) | compose default OPEN; `after_setup: {env: {SIGNUP_CLOSED: "true"}}` | TRAPS: the default must be OPEN (an installed app is unchanged by the catalog); several apps' switch also refuses the household's FIRST account, so never set it at install; an app with no env switch (opengist, wishlist) keeps the block alone — make that block case-insensitive. | +| **A same-tag security fix → a RE-TEST step** (`09` decision 52, 2026-09-30) | `scripts/retest-floating.py` (the ONE monthly command), `scripts/retest_box.py` (the box venue), `scripts/box_walk.py` (the 9202 client), `upgrade-test.py --retest` + `--write-ladder` | An entry whose `from` == `to`, with `digest_from` (the tested digest it started from) and `box_evidence`. `--dry-run` lists; `--engines-only` is the ruled start. Runbook `felhom.eu/documentation/runbooks/monthly-floating-retest.md`. TRAPS: never write one by hand (the gates refuse no new digest, a digest the registry stopped serving, a missing box proof, a `digest_from` that is not the previous entry's digest); `image_digest.resolve` IGNORES a `@digest` in its argument — ask the registry by manifest for a digest. | +| **A bench-only environment override** (R-739) | `upgrade-test.py` `BENCH_ENV_OVERRIDES` | Only for an app that cannot run on the bench at all (wanderer: its web server calls the DB at the public https name). Every verdict carries `bench_overrides`. Never a template change. | | Controller-side health probe | `templates/vaultwarden/.felhom.yml` (`healthcheck:` block) | `healthcheck.checks[]` with `type: http` (port only), `type: api` (port + `path` + `expect.status: 200`), or `type: tcp` (port only — mealie, crafty-controller). Prefer `api` with a real health path when the app has one. | | App lifecycle (`available`/`hidden`/`abandoned`) | `templates/plant-it/.felhom.yml` (`lifecycle:` block) | Optional top-level `lifecycle:` in `.felhom.yml`. Absent/empty ≡ `available`. `hidden` = not offered for new installs; `abandoned` = same, PLUS a permanent "Nem karbantartott" badge + notice on every box already running it. **Deployed instances keep full function in both states** — lifecycle governs what is OFFERED, never what runs; the controller refuses a deploy of a non-available template server-side (fail-closed, so a stale link or direct POST cannot install one). Unknown value → treated as `available` + one WARN, never a broken template. **Do NOT take an app out of circulation by deleting or moving its directory** — that orphans every customer already running it, which is what the 2026-07-21 `retired/` experiment got wrong. The resolvability gate skips non-available apps, so an abandoned app's dead image is not a standing red. | | **Catalog gates — THE entry point** | `scripts/catalog_gates.py` | **Run `python3 scripts/catalog_gates.py ` after ANY template change** (mandated in `CLAUDE.md`). Runs all four gates below in order — image-pins, image-resolvable, volume-persistence, engine-major (2026-09-13; git-history diff, hook-only until CI fetches deeper, R-452) — and exits **non-zero if any fails**; **2 (UNDETERMINED) is reported distinctly and is never a pass**, 1 (convicted) outranks 2 in the summary. Naming app(s) scopes the two gates that accept scoping, which is the normal after-a-change run; with no names the RUNTIME gate deploys every template, so that form is **scratch host only**. **Why a runner** (operator ruling 2026-08-02, R-161): the only gates in this project that ever get run are the ones with a single entry point named in a CLAUDE.md — `felhom.eu/scripts/site_gates.py` is run, R-29's three orphans are named nowhere and have stopped nothing. Controller-side enforcement was rejected because a load-time check reads only the file and a static audit reports the catalog clean **including papra** — it would pass on the very defect it exists to catch; CI was rejected for now (neither repo has any, no users yet). Adding a fourth gate here means adding it to `GATES` in this file — nothing else. | diff --git a/scripts/box_walk.py b/scripts/box_walk.py new file mode 100644 index 0000000..556b175 --- /dev/null +++ b/scripts/box_walk.py @@ -0,0 +1,563 @@ +#!/usr/bin/env python3 +"""box_walk.py — the BOX venue's client (moved into the catalog 2026-09-30 from the audits' walk.py, so the monthly +re-test `retest-floating.py` has a stable home). ONE app's walk on scratch guest 9202, through the product's own +endpoints. Configuration by environment: SC (a 0600 scratch dir holding `.ctlpw`, the 9202 dashboard password — +never committed), EV (where evidence goes), GUEST/BASE/DOMAIN as before. + +EVIDENCE, NOT PRODUCT. It presses exactly the buttons a person presses: + POST /api/stacks//deploy · POST /api/sync · POST /api/stacks/rescan + POST /api/stacks//update · POST /api/stacks//remove +and reads GET /api/stacks/. No controller code exists for it. + +The walk, per `09` §6.4 and the update-night brief §4: + 1 deploy from the DRILL catalog at the LIVE pin + 2 seed through the app's OWN front door (R-156: never a volume, never SQL) + 3 read the seed back <- control C1; a fixture that cannot prove itself proves nothing + 4 „Mentés most" + 5 commit the real one-step bump to the DRILL repo, sync, rescan, read the badge in BOTH languages + 6 press the guarded Update, record every phase with timestamps + 7 read the seed back through the front door + 8 the four version observables side by side + 9 write the verdict record in `09`'s JSON shape + +`inconclusive` is a first-class verdict and is NEVER collapsed into `failed`. +""" +import argparse, json, os, re, subprocess, sys, time +from datetime import datetime, timezone + +SC = os.environ.get('SC', os.path.expanduser('~/.felhom-retest')) +EV = os.environ.get('EV', os.path.join(SC, 'evidence')) +DRILL = "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill" +# GUEST=9201 selects demo-hp's hub-enabled guest (the mail proof); default 9202, the scratch guest. +GUEST = os.environ.get("GUEST", "9202") +BASE = os.environ.get("BASE") or {"9202": "https://192.168.0.114", "9201": "https://192.168.0.155"}[GUEST] +DOMAIN = os.environ.get("DOMAIN", "enkisfelhom.hu") +HOSTHDR = f"Host: felhom.{DOMAIN}" +HP = "demo-hp" + +LOG = [] + + +def say(*a): + line = " ".join(str(x) for x in a) + ts = datetime.now().strftime("%H:%M:%S") + print(f"{ts} {line}", flush=True) + LOG.append(f"{ts} {line}") + + +def sh(args, timeout=300, inp=None): + try: + return subprocess.run(args, capture_output=True, text=True, timeout=timeout, input=inp) + except (subprocess.TimeoutExpired, OSError) as e: + return subprocess.CompletedProcess(args, 124, "", f"{e}") + + +def guest(script, timeout=600): + """Run a bash script inside guest 9202. Piped as a file — never as an argument (quoting).""" + # ONE TEMP FILE PER CALL (night 2026-09-23): the shared /tmp/w.sh swapped scripts under + # two concurrent walks (memory: guest-helper-shares-one-tmp-file). + import secrets as _s + t = f"/tmp/w{GUEST}-{os.getpid()}-{_s.token_hex(4)}.sh" + r = sh(["ssh", "-o", "ConnectTimeout=20", "-o", "StrictHostKeyChecking=accept-new", HP, + f"export LC_ALL=C; cat > {t}; pct push {GUEST} {t} {t} >/dev/null 2>&1; " + f"pct exec {GUEST} -- bash {t}; pct exec {GUEST} -- rm -f {t}; rm -f {t}"], + timeout=timeout, inp=script) + return r.stdout or "" + + +def login(): + pw = open(f"{SC}/.ctlpw").read().strip() + sh(["curl", "-sk", "-D", f"{SC}/hdr{os.getpid()}.txt", "-o", "/dev/null", "-H", HOSTHDR, + "-X", "POST", "--data-urlencode", f"password={pw}", f"{BASE}/login"]) + h = open(f"{SC}/hdr{os.getpid()}.txt").read() + m = re.search(r"felhom_session=[A-Za-z0-9._-]+", h, re.I) + if not m: + sys.exit("login failed: no session cookie") + open(f"{SC}/sess{os.getpid()}.txt", "w").write(m.group(0)) + r = sh(["curl", "-sk", "-L", "-H", HOSTHDR, "-H", f"Cookie: {m.group(0)}", f"{BASE}/"]) + c = re.search(r' the felhom_gate cookie the household's browser would hold (setup gate, v0.280.0) + + +def _loc(out): + m = re.search(r"(?im)^location:\s*(\S+)", out or "") + return m.group(1) if m else "" + + +def gate_cookie(sub): + """Pass the setup gate (`09` decision 46) the way the HOUSEHOLD does: the app host redirects to the + dashboard's /__gate/start, which (with the dashboard session) redirects back to the app host's + /__felhom_gate/cb, which sets `felhom_gate`. Never printed.""" + import urllib.parse + sess = open(f"{SC}/sess{os.getpid()}.txt").read().strip() + r = sh(["curl", "-sk", "-D", "-", "-o", "/dev/null", "-H", "Accept: text/html", "-H", f"Host: {sub}.{DOMAIN}", f"{BASE}/"]) + loc = _loc(r.stdout) + if "/__gate/start" not in loc: + GATE[sub] = "" + return "" # not gated (open, or no gate for this app) + u = urllib.parse.urlsplit(loc) + r = sh(["curl", "-sk", "-D", "-", "-o", "/dev/null", "-H", "Accept: text/html", "-H", f"Host: {u.hostname}", "-H", f"Cookie: {sess}", + f"{BASE}{u.path}?{u.query}"]) + loc = _loc(r.stdout) + u = urllib.parse.urlsplit(loc) + if "/__felhom_gate/cb" not in u.path: + say(f" gate: the dashboard did not hand back a callback for {sub} ({loc[:80]})") + return "" + r = sh(["curl", "-sk", "-D", "-", "-o", "/dev/null", "-H", "Accept: text/html", "-H", f"Host: {u.hostname}", f"{BASE}{u.path}?{u.query}"]) + m = re.search(r"(?im)^set-cookie:\s*(felhom_gate=[^;\r\n]+)", r.stdout or "") + GATE[sub] = m.group(1) if m else "" + say(f" gate: {sub} is gated — passed as the household (cookie {'set' if GATE[sub] else 'NOT set'})") + return GATE[sub] + + +def app_curl(sub, path, *extra, method=None, data=None, timeout=45, _retry=True): + """A call to the APP's own front door on 9202 — the household's route, not ours. Carries the setup + gate's cookie when the app is gated, merged into a fixture's own Cookie header (never a second one).""" + raw = list(extra) + gc = GATE[sub] if sub in GATE else gate_cookie(sub) + ext = list(raw) + if gc: + merged = False + for i, a in enumerate(ext): + if isinstance(a, str) and a.lower().startswith("cookie:") and i > 0 and ext[i - 1] == "-H": + ext[i] = a + "; " + gc + merged = True + if not merged: + ext = ["-H", f"Cookie: {gc}"] + ext + args = ["curl", "-sSk", "--max-time", str(timeout), "-H", f"Host: {sub}.{DOMAIN}", + "-w", "\n%{http_code} %{redirect_url}"] + if method: + args += ["-X", method] + if data is not None: + args += ["--data-binary", "@-"] + args += ext + [f"{BASE}{path}"] + r = sh(args, timeout=timeout + 30, inp=data) + body, _, tail = (r.stdout or "").rpartition("\n") + code, _, redir = tail.strip().partition(" ") + if _retry and "/__gate/start" in redir: + GATE.pop(sub, None) # the gate cookie expired or was never taken — log in as the household again + return app_curl(sub, path, *raw, method=method, data=data, timeout=timeout, _retry=False) + return r.returncode, code.strip(), body + + +def stack(name): + _, d = ctl("GET", f"/api/stacks/{name}") + return (d.get("data") or {}) if isinstance(d, dict) else {} + + +def wait_app(sub, path="/", want=("200", "302", "303", "401", "403"), tries=60, delay=5): + """Settling says the container runs; this says the APP answers. Not the same thing.""" + last = None + for _ in range(tries): + rc, code, _ = app_curl(sub, path, timeout=15) + last = (rc, code) + if rc == 0 and code in want: + return True + time.sleep(delay) + say(f" app never answered on {sub}{path} (last rc={last[0]} code={last[1]})") + return False + + +# ------------------------------------------------------------------ the walk + + +DRIVE = "/mnt/felhom-drives/scratch_hdd/userdata" + +# What THIS run generated for a deploy, per app. Deploy secrets are ENCRYPTED AT REST in +# `app.yaml` (`ENC:…`), which is right and which means a fixture cannot read an app's admin +# password back off the box — the household sees it once. So the value the harness itself +# generated is kept here for the life of the run, and nowhere else. +GENERATED = {} + + +def deploy_values(name, sub): + """Fill EVERY required deploy field the way the wizard would, by asking the box what this app + asks for — `GET /api/stacks//deploy-fields` — instead of assuming DOMAIN+SUBDOMAIN. + + Measured 2026-09-21: three apps in one batch refused at the deploy with a correct 400 because + a required field was absent — `HDD_PATH` (navidrome, audiobookshelf) and an admin password + (grafana). The refusals happen BEFORE anything is created (`deploy.go:324`), which is the only + reason this was safe to discover by running it (live-probes rule). + + A `path` field must name a directory that ALREADY EXISTS (`deploy.go:330`), so one is made on + the scratch drive first — the same act the drive browser performs for a household. + """ + code, d = ctl("GET", f"/api/stacks/{name}/deploy-fields") + fields = (((d.get("data") or {}).get("metadata") or {}).get("deploy_fields")) or [] + values = {"DOMAIN": DOMAIN, "SUBDOMAIN": sub} + made = [] + for f in fields: + ev, ty = f.get("env_var"), f.get("type") + if ev in values: + continue + # `type: password` is MANDATORY whatever `required` says — `deploy.go:305-312` refuses + # when the caller sends none, deliberately ("the user needs to know their password"), + # while `.felhom.yml` declares `required: false` and the API serves that verbatim. A + # caller that trusts the contract gets a 400. Measured tonight on grafana; filed. + if not f.get("required") and ty != "password": + continue # the controller generates the optional secrets itself + if ty == "path": + p = f"{DRIVE}/{name}" + values[ev] = p + made.append(p) + elif ty in ("secret", "password"): + import secrets as _s + values[ev] = "Drill-" + _s.token_hex(12) + GENERATED.setdefault(name, {})[ev] = values[ev] + elif f.get("default"): + values[ev] = f["default"] + else: + values[ev] = f"drill-{name}" + if made: + guest("mkdir -p " + " ".join(made) + "; ls -ld " + " ".join(made)) + say(f" [1] made the drive paths this app requires: {made}") + extra = [k for k in values if k not in ("DOMAIN", "SUBDOMAIN")] + if extra: + say(f" [1] required fields filled beyond DOMAIN/SUBDOMAIN: {extra}") + return values + + +def deploy(name, sub, extra_values=None): + st = stack(name) + if st.get("deployed"): + say(f" [1] {name} already deployed — reusing") + return True + values = deploy_values(name, sub) + if extra_values: + values.update(extra_values) + body = {"values": values} + if os.environ.get("KEPT"): # decision 36: the household's answer when the drive holds old data ("fresh" moves it aside, deletes nothing) + body["kept_data"] = os.environ["KEPT"] + code, d = ctl("POST", f"/api/stacks/{name}/deploy", body) + say(f" [1] deploy -> {code} {str(d)[:120]}") + if code != "202": + return False + # WAIT FOR `deployed`, NOT FOR `running`. Measured 2026-09-21 on tandoor: docker reported the + # container `healthy` while the controller's own state read `unhealthy` — a gate on `running` + # alone therefore times out on an app that is up. The state is RECORDED rather than required; + # the real gate is the fixture's own `wait_app`, which asks whether the APP answers. + seen = None + for _ in range(90): + time.sleep(5) + st = stack(name) + seen = st.get("state") + # `deployed` alone is NOT enough and `state` alone is NOT right. Measured 2026-09-21: + # tandoor reads `unhealthy` while serving (R-618), so gating on "running" hangs; and romm + # read `deployed=True, state=degraded, pinned_images=None` twenty seconds in, i.e. the + # deploy had not finished writing app.yaml. The PIN is the deploy's own completion mark + # (`runComposeDeploy` writes it), so that is what to wait for. + pins = (st.get("app_config") or {}).get("pinned_images") + if st.get("deployed") and pins and seen in ("running", "unhealthy", "degraded"): + say(f" [1] deployed, controller state={seen}, " + f"pinned={(st.get('app_config') or {}).get('pinned_images')}") + if seen != "running": + say(f" [1] NOTE: the controller's own state is {seen!r}, not 'running' — recorded, " + f"not treated as a failure; the fixture's front-door wait is the real gate") + return True + say(f" [1] never became deployed (last controller state={seen!r})") + return False + + +def backup_now(name): + """R-648 (2026-09-23): NO whole-box „Mentés most" from a drill, ever. + + `POST /api/backup/run` is the only backup endpoint and it is WHOLE-BOX: on 9201 it stopped and + restarted 9 of 10 standing apps twice, and on 9202 it broke a deploy in flight (R-634). The product + has NO per-app backup endpoint (router.go: /backup/run, /backup/tier2 only); the per-app backup + exists only inside the guarded update, whose `backing-up` phase calls RunAppBackupNow for the one + app. So this presses nothing: the update takes the throwaway app's own backup, and says so in its + phase list. A seed written "after the backup" is therefore written before the update's own backup + — the undo's last-second copy is still the one that must bring it back.""" + say(f" [4] backup press SKIPPED for {name} (R-648: whole-box only; the update's backing-up phase backs up {name} alone)") + return None + +def drill_bump(app, frm, to, service_hint=None): + """Serialised across concurrent walks: one git working tree, one lock.""" + import fcntl + with open(f"{SC}/drill.lock", "w") as lk: + fcntl.flock(lk, fcntl.LOCK_EX) + sh(["git", "-C", DRILL, "pull", "-q", "--rebase", "origin", "main"], timeout=120) + return _drill_bump(app, frm, to, service_hint) + + +def _drill_bump(app, frm, to, service_hint=None): + """Commit the edge to the DRILL repo. catalog_since set by hand (the drill repo has no gates). + + `frm`/`to` may be comma-separated lists of the SAME length: an app whose own version lives in + two images (adventurelog's backend and frontend) moves both in one edge, while its engine + sidecar stays where it is — `09` §3b Q3's rule is per SERVICE, and an app-half edge must move + every service that carries the app's own version and no others. + """ + comp = f"{DRILL}/templates/{app}/docker-compose.yml" + fy = f"{DRILL}/templates/{app}/.felhom.yml" + s = open(comp).read() + froms = [x.strip() for x in frm.split(",") if x.strip()] + tos = [x.strip() for x in to.split(",") if x.strip()] + if len(froms) != len(tos): + say(f" [5] from/to lists differ in length: {froms} vs {tos}") + return None + for f1, t1 in zip(froms, tos): + if f"image: {f1}" not in s: + say(f" [5] FROM ref not found in compose: {f1}") + return None + s = s.replace(f"image: {f1}", f"image: {t1}") + open(comp, "w").write(s) + f = open(fy).read() + today = datetime.now().strftime("%Y-%m-%d") + f = re.sub(r'^catalog_since:.*$', f'catalog_since: "{today}"', f, count=1, flags=re.M) + open(fy, "w").write(f) + sh(["git", "-C", DRILL, "add", "-A"]) + sh(["git", "-C", DRILL, "commit", "-q", "-m", f"DRILL {app}: {frm} -> {to}"]) + r = sh(["git", "-C", DRILL, "push", "-q", "origin", "main"], timeout=120) + h = sh(["git", "-C", DRILL, "rev-parse", "--short=12", "HEAD"]).stdout.strip() + say(f" [5] drill commit {h}: {app} {frm} -> {to} (push rc={r.returncode})") + return h + + +def sync_rescan(expect_app=None, expect_ref=None, tries=12, delay=5): + """Sync, rescan, and — when told what to expect — WAIT FOR THE BADGE TO CATCH UP. + + R-607: `POST /api/sync` answers "nincs valtozas" while the catalog HAS moved, and + `catalog_images` stays stale until a separate rescan. Tonight showed the rescan alone is not + enough either: mealie's badge read "Naprakesz" seconds after its bump was pushed, and the + Update that followed moved nothing and still reported "Frissitve". So when the caller knows + which reference should appear, this polls for it and SAYS HOW LONG IT TOOK — which is the + NUMBER R-607 asks for and has never had. + """ + t0 = time.time() + ctl("POST", "/api/sync") + time.sleep(2) + ctl("POST", "/api/stacks/rescan") + time.sleep(2) + if not expect_app or not expect_ref: + return None + for i in range(tries): + cat = stack(expect_app).get("catalog_images") or {} + if expect_ref in cat.values(): + waited = round(time.time() - t0, 1) + if i: + say(f" [sync] the badge needed {waited}s and {i+1} sync+rescan rounds to catch up " + f"to {expect_ref} — R-607's window, measured") + return waited + time.sleep(delay) + ctl("POST", "/api/sync") + time.sleep(1) + ctl("POST", "/api/stacks/rescan") + say(f" [sync] the badge NEVER caught up to {expect_ref} in {round(time.time()-t0,1)}s — " + f"catalog_images = {stack(expect_app).get('catalog_images')}") + return None + + +def badges(name): + out = {} + for lang, suffix in (("hu", ""), ("en", "?lang=en")): + h = page(f"/apps/{name}{suffix}") + m = re.findall(r']*title="([^"]*)"[^>]*>([^<]*)<', h) + out[lang] = [{"title": a.strip(), "text": b.strip()} for a, b in m][:3] + return out + + +def press_update(name, poll=1.0, cap_s=1800): + code, d = ctl("POST", f"/api/stacks/{name}/update") + say(f" [6] Update -> {code} {str(d)[:220]}") + if code not in ("202", "200"): + return {"accepted": False, "http": code, "refusal": d, "phases": [], "duration_s": 0} + phases, seen, t0 = [], None, time.time() + while time.time() - t0 < cap_s: + st = stack(name) + ph = st.get("update_phase") + if ph != seen: + seen = ph + rec = {"t": round(time.time() - t0, 1), "phase": ph, + "label": st.get("update_phase_label"), "updating": st.get("updating"), + "error": st.get("update_error"), "hold": st.get("hold_reason")} + phases.append(rec) + say(f" +{rec['t']:>6.1f}s phase={ph} label={rec['label']} " + f"err={rec['error']} hold={rec['hold']}") + if not st.get("updating") and ph in ("done", "failed", "undone", None) and time.time() - t0 > 3: + break + time.sleep(poll) + st = stack(name) + return {"accepted": True, "http": code, "phases": phases, + "duration_s": round(time.time() - t0, 1), + "final_phase": st.get("update_phase"), "update_error": st.get("update_error"), + "hold_reason": st.get("hold_reason"), "state": st.get("state")} + + +def observables(name): + st = stack(name) + ac = st.get("app_config") or {} + live = guest(f""" +grep -E '^\\s+image:' /opt/docker/stacks/{name}/docker-compose.yml 2>/dev/null | sed 's/^ *//' +echo '---inspect---' +for c in $(docker ps -a --filter label=com.docker.compose.project={name} --format '{{{{.Names}}}}'); do + echo -n "$c "; docker inspect "$c" --format '{{{{.Config.Image}}}} running={{{{.State.Running}}}} restarts={{{{.RestartCount}}}}' +done +""") + a, _, b = live.partition("---inspect---") + return { + "pinned_images": ac.get("pinned_images"), + "installed_images": {k: (v.get("ref") if isinstance(v, dict) else v) + for k, v in (ac.get("installed_images") or {}).items()}, + "catalog_images": st.get("catalog_images"), + "live_compose_image_lines": [x for x in a.strip().splitlines() if x.strip()], + "docker_inspect": [x for x in b.strip().splitlines() if x.strip()], + } + + +def app_logs(name, lines=400): + """The app's own container log, DECODED. The endpoint answers a JSON envelope whose `logs` is + one string with escaped newlines — a scan over the envelope sees a single enormous line and + finds nothing, which reads exactly like "the app printed no migration line" and is not. R-96 + rule 3 in a new place: an absent line is not evidence when the instrument cannot see lines.""" + code, d = ctl("GET", f"/api/stacks/{name}/logs?lines={lines}") + if isinstance(d, dict): + data = d.get("data") + if isinstance(data, dict) and isinstance(data.get("logs"), str): + return data["logs"] + if isinstance(d.get("_raw"), str): + return d["_raw"] + return str(d) + + +def write_verdict(rec, appdir): + os.makedirs(appdir, exist_ok=True) + p = os.path.join(appdir, "verdict.json") + json.dump(rec, open(p, "w"), indent=2, ensure_ascii=False) + say(f" [9] verdict {rec['verdict']} -> {p}") + + +def remove(name): + """Remove through the PRODUCT, never `docker rm` (live-probes rule). The remove endpoint + refuses a running stack — `409 still running` — so the stop is part of the act, not a tidy-up.""" + c1, d1 = ctl("POST", f"/api/stacks/{name}/stop") + say(f" [X] stop -> {c1} {str(d1)[:100]}") + for _ in range(24): + time.sleep(5) + if stack(name).get("state") != "running": + break + code, d = ctl("POST", f"/api/stacks/{name}/remove", + {"remove_hdd_data": True, "remove_backups": True}) + say(f" [X] remove (with drive data) -> {code} {str(d)[:160]}") + if code == "409": + # R-442's fail-closed guard: when the storage subsystem cannot RESOLVE the app's drive + # path, the removal is REFUSED and the app is kept rather than half-deleted. On guest 9202 + # `/api/disks` answers `agent not configured`, so every app deployed with an HDD_PATH hits + # this. The household's other choice — remove the app, KEEP the data — is accepted, and the + # harness takes it, then tidies its own directory by name at teardown. + say(" [X] refused because the drive path cannot be resolved (R-442, fail-closed and right)" + " — removing the app and KEEPING the drive data instead") + code, d = ctl("POST", f"/api/stacks/{name}/remove", + {"remove_hdd_data": False, "remove_backups": True}) + say(f" [X] remove (keeping drive data) -> {code} {str(d)[:160]}") + time.sleep(5) + st = stack(name) + left = guest(f"ls -d /opt/docker/stacks/{name} 2>/dev/null; " + f"docker ps -a --filter label=com.docker.compose.project={name} --format '{{{{.Names}}}}'") + say(f" [X] after remove: deployed={st.get('deployed')} leftovers={left.strip()!r}") + return code + + +def app_env(name, key): + """Read one deploy value the CUSTOMER was given (e.g. the generated admin password) from the + app's own `app.yaml`. This is not seeding — it is how the household logs in; the controller + shows them the same value. Data still goes in through the app's own front door.""" + out = guest(f"grep -E '^\\s*{key}:' /opt/docker/stacks/{name}/app.yaml 2>/dev/null | head -1") + if ":" in out: + return out.split(":", 1)[1].strip().strip('"').strip("'") + return "" + + +def snapshots(name): + """The restorable copies the backups page offers for this app.""" + code, d = ctl("GET", f"/api/backup/snapshots?stack={name}") + data = d.get("data") if isinstance(d, dict) else None + if isinstance(data, dict): + for k in ("snapshots", "items", "restore_points"): + if isinstance(data.get(k), list): + return data[k] + return data if isinstance(data, list) else [] + + +def restore(name, snapshot_id=None, wait_s=1200): + """The household's own way out: the „Visszaállítás a mentésből" button on the backups page. + + A FORM post, not an API call — `POST /backup/restore` with `_csrf`, `stack_name`, + `snapshot_id` — because that is the button the sentence tells them to press. + """ + snaps = snapshots(name) + if snapshot_id is None: + if not snaps: + say(f" [R] no restorable copy offered for {name}") + return {"ok": False, "why": "no snapshot offered", "snapshots": snaps} + first = snaps[0] + snapshot_id = first.get("id") or first.get("snapshot_id") or first.get("short_id") + say(f" [R] restoring {name} from snapshot {snapshot_id!r} (of {len(snaps)} offered)") + sess = open(f"{SC}/sess{os.getpid()}.txt").read().strip() + csrf = open(f"{SC}/csrf{os.getpid()}.txt").read().strip() + r = sh(["curl", "-sk", "-D", "-", "-o", "/dev/null", "-H", HOSTHDR, "-H", f"Cookie: {sess}", + "-X", "POST", + "--data-urlencode", f"_csrf={csrf}", + "--data-urlencode", f"stack_name={name}", + "--data-urlencode", f"snapshot_id={snapshot_id}", + f"{BASE}/backup/restore"], timeout=180) + head = (r.stdout or "").split("\n")[0].strip() + loc = [l for l in (r.stdout or "").split("\n") if l.lower().startswith("location:")] + say(f" [R] POST /backup/restore -> {head} {loc[:1]}") + t0 = time.time() + last = None + while time.time() - t0 < wait_s: + code, d = ctl("GET", "/api/backup/restore-status") + dd = d.get("data") or {} + cur = (dd.get("running"), dd.get("phase") or dd.get("state"), dd.get("message")) + if cur != last: + say(f" +{round(time.time()-t0,1):>6.1f}s restore {cur}") + last = cur + if not dd.get("running", False) and time.time() - t0 > 5: + break + time.sleep(2) + st = stack(name) + say(f" [R] after restore: state={st.get('state')} hold={st.get('hold_reason')!r} " + f"phase={st.get('update_phase')}") + return {"ok": True, "snapshot_id": snapshot_id, "snapshots": snaps, + "http": head, "location": loc[:1], "seconds": round(time.time() - t0, 1), + "state_after": st.get("state"), "hold_after": st.get("hold_reason"), + "observables_after": observables(name)} diff --git a/scripts/check-test-record-move.py b/scripts/check-test-record-move.py index 04d5e4d..a77e20b 100644 --- a/scripts/check-test-record-move.py +++ b/scripts/check-test-record-move.py @@ -42,6 +42,7 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import ladder # noqa: E402 TEMPLATE_RE = re.compile(r"^templates/([^/]+)/docker-compose\.ya?ml$") +FELHOM_RE = re.compile(r"^templates/([^/]+)/\.felhom\.ya?ml$") ZERO_SHA_RE = re.compile(r"^0{40}$") MEM_RE = re.compile(r"^\s+(memory|mem_limit):", re.M) @@ -140,6 +141,45 @@ def judge_app(app, a, b, resolver, network=True): return problems, inconclusive +def judge_retests(app, a, b, resolver, network=True): + """(problems, inconclusive) for the RE-TEST entries (`from` == `to`, decision 52) this range adds to one + template. A re-test moves no image line, so judge_app never looks at it; it changes `.felhom.yml` only. Each new + re-test must be well-formed, proven, not backfilled, the ladder's NEWEST entry naming the compose's refs, and its + digest must be what the registry serves for those refs NOW (the image a box would pull).""" + cpath, fpath = "templates/%s/docker-compose.yml" % app, "templates/%s/.felhom.yml" % app + after_c = show(b, cpath) + if after_c is None: + return [], [] + after = ladder.images_in(after_c) + _e, raws_a, _ = ladder.parse(show(a, fpath) or "") + ents_b, raws_b, errs_b = ladder.parse(show(b, fpath) or "") + old = set(raws_a) + new = [(i, e) for i, (e, raw) in enumerate(zip(ents_b, raws_b)) if raw not in old and e.get("from") == e.get("to")] + problems, inconclusive = [], [] + for i, e in new: + for q in ladder.check_entry(e): + problems.append("new re-test: " + q) + if e.get("backfilled") is not None or e.get("verdict") != "proven": + problems.append("new re-test is not a proven new test (verdict %r, backfilled %r)" % (e.get("verdict"), e.get("backfilled"))) + if i != len(ents_b) - 1 or e.get("to") != after: + problems.append("new re-test is not the ladder's newest entry naming the compose's refs %s" % after) + if problems or not new: + return problems, inconclusive + if not network: + print(" %s: re-test digest comparison SKIPPED (--no-network) — not a pass for a real push" % app) + return problems, inconclusive + e = new[-1][1] + for svc, ref in sorted(after.items()): + want = (e.get("digest") or {}).get(svc) + got, why = resolver(ref) + if got is None: + inconclusive.append("%s %s: the registry could not be asked (%s)" % (svc, ref, why)) + elif got != want: + problems.append("re-test %s %s: the registry serves %s, the re-test says %s — the re-tested image is " + "not the image a box would pull" % (svc, ref, got, want)) + return problems, inconclusive + + def main(argv, resolver=None): spec = "origin/main..HEAD" network = "--no-network" not in argv @@ -177,6 +217,13 @@ def main(argv, resolver=None): convicted[app] = p if inc: undecided[app] = inc + # decision 52: re-tests change .felhom.yml only — judged separately, for every template whose .felhom.yml changed + for app in sorted({FELHOM_RE.match(n).group(1) for n in names.split("\n") if FELHOM_RE.match(n)}): + p, inc = judge_retests(app, a, b, resolver, network) + if p: + convicted.setdefault(app, []).extend(p) + if inc: + undecided.setdefault(app, []).extend(inc) print("test-record-move gate — range %s..%s: %d compose file(s) changed" % (a, b, len(apps))) for app, ps in convicted.items(): for p in ps: diff --git a/scripts/check-test-record.py b/scripts/check-test-record.py index 76a6835..55955e6 100644 --- a/scripts/check-test-record.py +++ b/scripts/check-test-record.py @@ -16,6 +16,8 @@ WHAT IT CHECKS, per template that carries `update_ladder:` (format and field rul (verdict proven|unrecorded only; a sha256 digest per `to` service; the memory watch's peak on any entry not backfilled; marks.memory_tight agrees with the peak); 2. the ladder is CONTINUOUS — each entry's `from` is the previous entry's `to`; + 2b. a RE-TEST (`from` == `to`, `09` §3 decision 52) starts FROM the previous entry's digest + (`digest_from`), and is never the first entry; 3. the NEWEST entry's `to` is EXACTLY the compose's current image per service. This is the fact that makes the rule hold without history: a compose moved without a new entry no longer matches its ladder's head, whoever pushed it and however. @@ -55,6 +57,20 @@ def check_app(app_dir): for i in range(1, len(entries)): if entries[i].get("from") != entries[i - 1].get("to"): problems.append("entry %d's `from` is not entry %d's `to` — the ladder has a gap" % (i + 1, i)) + # rule 2b (decision 52): a RE-TEST (from == to) was tested FROM the previous entry's digest — never from some + # other build of the tag — and is never the first entry (there is nothing to re-test). + for i, e in enumerate(entries): + if e.get("from") != e.get("to") or not isinstance(e.get("digest_from"), dict): + continue + if i == 0: + problems.append("entry 1 is a re-test (from == to) — a re-test needs an earlier entry to re-test") + continue + prev = entries[i - 1].get("digest") or {} + bad = sorted(s for s, d in e["digest_from"].items() if prev.get(s) != d) + if bad: + problems.append("entry %d is a re-test FROM %s, but entry %d tested %s — a re-test starts from the " + "previous entry's digest" % (i + 1, {s: e["digest_from"][s][:19] for s in bad}, i, + {s: str(prev.get(s))[:19] for s in bad})) for i, e in enumerate(entries[:-1]): to = e.get("to") if not isinstance(to, dict) or not to: diff --git a/scripts/ladder.py b/scripts/ladder.py index 454c175..fe9fb5f 100644 --- a/scripts/ladder.py +++ b/scripts/ladder.py @@ -32,6 +32,13 @@ AN ENTRY (all keys required unless marked): `upgrade-test.py --write-ladder` only when the bench AND the box both converted it backfilled (optional) "YYYY-MM-DD" — written by the backfill from an EXISTING record, never by a new test; a new move may not carry it + digest_from (a RE-TEST only, `09` §3 decision 52) {service: "sha256:…"} — the digest the re-test + was run FROM. A re-test is an entry whose `from` equals its `to` (the same tags): the + catalog proved the same tag at a NEW digest (an upstream same-name fix). It needs + `digest_from` for every service, at least one service whose `digest` differs from it, + and `box_evidence` (both venues); `check-test-record.py` rule 2b ties `digest_from` to + the previous entry's `digest`. Written by `upgrade-test.py --write-ladder` from a + `--retest` verdict, never by hand. STEP DEFINITIONS (`09` §6.4 part 5, controller v0.268.0): every entry but the NEWEST carries its own complete compose file at `templates//steps/.yml` — the box climbs one step at a time @@ -138,6 +145,16 @@ def check_entry(e): for svc in e["digest"]: if svc not in e["to"]: p.append("digest names a service %r that `to` does not" % svc) + if e["from"] == e["to"] and v == "proven" and backfilled is None: # a RE-TEST (decision 52) + df = e.get("digest_from") + if not isinstance(df, dict) or set(df) != set(e["to"]) or not all(isinstance(x, str) and DIGEST_RE.match(x) for x in df.values()): + p.append("a re-test (from == to) needs digest_from: {service: sha256} for every service — the digest it was tested FROM") + elif all(df[s] == e["digest"].get(s) for s in e["to"]): + p.append("a re-test (from == to) whose digest is the same as its digest_from tests nothing new — no new digest") + if not (isinstance(e.get("box_evidence"), str) and e["box_evidence"].strip()): + p.append("a re-test (from == to) must cite BOTH venues: box_evidence is missing") + elif e.get("digest_from") is not None: + p.append("digest_from belongs only to a re-test (from == to)") conv = e.get("engine_conversion") if conv is not None: # `09` §6.4 part 10 — the box converts ONLY on this mark if not (isinstance(conv, dict) and set(conv) == {"service", "engine", "from", "to"} diff --git a/scripts/retest-floating.py b/scripts/retest-floating.py new file mode 100644 index 0000000..5a6be85 --- /dev/null +++ b/scripts/retest-floating.py @@ -0,0 +1,230 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +"""retest-floating.py — the MONTHLY re-test of same-tag security fixes (`09` §3 decision 52, R-740). ONE command. + + python3 scripts/retest-floating.py --dry-run # list only: which services' registry digest moved + python3 scripts/retest-floating.py [--only app,app] # re-test them: bench, box, writer — one commit per app + python3 scripts/retest-floating.py --push # ... and push each commit (the pre-push gates run) + python3 scripts/retest-floating.py --engines-only ... # database/redis lines only (decision 52's start) + +WHAT IT DOES. For every template whose ladder's NEWEST entry is its compose, it asks the registry for each service's +digest NOW and compares it with the digest that entry TESTED. A service whose digest moved is a same-name upstream +fix (typically a database or redis line). Candidates are ordered database/redis first. For each app, in order: + 1. BENCH — `upgrade-test.py --soak 600 --retest ` on the bench LXC (the full method: FROM at the tested + digest, seed, read back, TO at the new digest, read back, the 10-minute memory watch, the abort), evidence + copied off the bench before the next app; + 2. BOX — scratch guest 9202 on the DRILL catalog: a fresh install renders the OLD tested digest (the live ladder + says so), the fixture seeds; a drill-only commit adds the re-test entry; the product's guarded Update runs the + step; the seed reads back; the running digest is checked; + 3. WRITER — `upgrade-test.py --write-ladder` from both verdicts into THIS checkout (a re-test entry: `from` == + `to`, `digest_from`, `box_evidence`), `catalog_gates.py --fast `, one commit (and `--push`). +A failure stops THAT app (its reason in the summary) and never the list. Nothing is written without both venues. + +WHAT IT NEEDS (checked first; a missing one stops the run and says which): the bench LXC 9401 on demo-hp with +/opt/upg (runbook `felhom.eu/documentation/runbooks/monthly-floating-retest.md` §2 creates it), guest 9202 pointed +at the drill catalog (§3), SC= holding `.ctlpw` (9202's dashboard password — never committed), ssh to +demo-hp, and a push credential for the drill repo. It is therefore NOT a cron job: it is run by a person or a CC +session once a month, from DooPlex (the runbook says why). + +Exit: 0 every candidate done (or none) · 1 at least one app stopped · 2 could not start (a prerequisite). +""" +import argparse +import base64 +import datetime +import io +import json +import os +import re +import subprocess +import sys +import tarfile +import time + +HERE = os.path.dirname(os.path.abspath(__file__)) +CAT = os.path.dirname(HERE) +sys.path.insert(0, HERE) +import ladder # noqa: E402 +import image_digest # noqa: E402 + +ENGINE_RE = re.compile(r"(^|/)(postgres|postgis|mariadb|mysql|redis|valkey|mongo)", re.I) +DRILL = os.environ.get("DRILL", "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill") +BENCH_HOST, BENCH_CT = os.environ.get("BENCH_HOST", "demo-hp"), os.environ.get("BENCH_CT", "9401") + + +def candidates(only=None): + """[(app, {svc: (ref, tested, now)}, engine_first)] — every service whose registry digest moved.""" + out = [] + for app in sorted(os.listdir(os.path.join(CAT, "templates"))): + if only and app not in only: + continue + d = os.path.join(CAT, "templates", app) + try: + entries, _, errs = ladder.parse(open(os.path.join(d, ".felhom.yml"), encoding="utf-8").read()) + comp = ladder.images_in(open(os.path.join(d, "docker-compose.yml"), encoding="utf-8").read()) + except OSError: + continue + if errs or not entries or entries[-1].get("to") != comp: + continue + tested = entries[-1].get("digest") or {} + moved = {} + for svc, ref in comp.items(): + now, why = image_digest.resolve(ref) + if not now: + moved[svc] = (ref, tested.get(svc), "UNKNOWN: " + str(why)) + elif tested.get(svc) and now != tested[svc]: + moved[svc] = (ref, tested[svc], now) + if moved: + out.append((app, moved, any(ENGINE_RE.search(r) for r, _, _ in moved.values()))) + return sorted(out, key=lambda x: (not x[2], x[0])) + + +def sh(args, timeout=3600, inp=None): + try: + return subprocess.run(args, capture_output=True, text=True, timeout=timeout, input=inp) + except (subprocess.TimeoutExpired, OSError) as e: + return subprocess.CompletedProcess(args, 124, "", str(e)) + + +def bench(script, timeout=3600): + t = "/tmp/rt-%d-%d.sh" % (os.getpid(), int(time.time() * 1000) % 100000) + return sh(["ssh", "-o", "BatchMode=yes", BENCH_HOST, + "export LC_ALL=C; cat > %s; pct push %s %s %s >/dev/null 2>&1; pct exec %s -- bash %s; rc=$?; " + "pct exec %s -- rm -f %s; rm -f %s; exit $rc" % (t, BENCH_CT, t, t, BENCH_CT, t, BENCH_CT, t, t)], + timeout=timeout, inp=script) + + +def prerequisites(): + missing = [] + r = bench("test -f /opt/upg/upgrade-test.py && echo bench-ok", timeout=120) + if "bench-ok" not in (r.stdout or ""): + missing.append("the bench LXC %s on %s with /opt/upg (runbook §2)" % (BENCH_CT, BENCH_HOST)) + if not os.path.isdir(os.path.join(DRILL, ".git")): + missing.append("the drill checkout at %s (runbook §3)" % DRILL) + if not os.path.isfile(os.path.join(os.environ.get("SC", os.path.expanduser("~/.felhom-retest")), ".ctlpw")): + missing.append("SC/.ctlpw — 9202's dashboard password (runbook §3)") + return missing + + +def sync_bench(): + """This checkout's scripts/*.py and templates/ → /opt/upg on the bench (the harness runs them as they stand).""" + tgz = io.BytesIO() + with tarfile.open(fileobj=tgz, mode="w:gz") as t: + for f in os.listdir(HERE): + if f.endswith(".py"): + t.add(os.path.join(HERE, f), arcname=f) + t.add(os.path.join(CAT, "templates"), arcname="templates") + return sh(["ssh", "-o", "BatchMode=yes", BENCH_HOST, + "cat > /tmp/rt.b64; pct push %s /tmp/rt.b64 /tmp/rt.b64; rm -f /tmp/rt.b64; pct exec %s -- bash -c " + "'mkdir -p /opt/upg && cd /opt/upg && rm -rf templates && base64 -d /tmp/rt.b64 | tar xzf - && rm -f /tmp/rt.b64 && echo synced'" + % (BENCH_CT, BENCH_CT)], timeout=600, inp=base64.b64encode(tgz.getvalue()).decode()) + + +def run_bench(app, evdir): + r = bench("cd /opt/upg && rm -rf evidence/RT-%s && timeout 3000 python3 upgrade-test.py --soak 600 --retest %s " + "> /opt/upg/RT-%s.log 2>&1; echo rc=$?" % (app, app, app), timeout=3400) + pull = bench("cd /opt/upg && tar czf - RT-%s.log evidence/RT-%s 2>/dev/null | base64 -w0" % (app, app), timeout=600) + os.makedirs(evdir, exist_ok=True) + try: + tarfile.open(fileobj=io.BytesIO(base64.b64decode(pull.stdout.strip()))).extractall(evdir) + except Exception as e: # noqa: BLE001 + return None, "the bench evidence could not be copied off (%s); bench said %s" % (e, (r.stdout or "").strip()[-80:]) + vp = os.path.join(evdir, "evidence", "RT-%s" % app, "verdict.json") + if not os.path.isfile(vp): + return None, "no bench verdict (%s)" % (r.stdout or "").strip()[-120:] + v = json.load(open(vp)) + return (vp if v.get("verdict") == "proven" else None), "bench verdict %s (%s)" % (v.get("verdict"), str(v.get("abort_detail"))[:120]) + + +def run_box(app, moved, evdir): + """The box venue: retest_box.py does it (kept apart: it holds the 9202 session).""" + r = sh([sys.executable, os.path.join(HERE, "retest_box.py"), app, evdir], timeout=3600, + inp=json.dumps({s: [v[0], v[1], v[2]] for s, v in moved.items()})) + vp = os.path.join(evdir, "box-verdict-%s.json" % app) + if not os.path.isfile(vp): + return None, "no box verdict: %s" % ((r.stdout or "") + (r.stderr or "")).strip()[-200:] + v = json.load(open(vp)) + return (vp if v.get("verdict") == "proven" else None), "box verdict %s (%s)" % (v.get("verdict"), v.get("why", "")) + + +def write(app, bench_v, box_v, ev_rel, push): + r = sh([sys.executable, os.path.join(HERE, "upgrade-test.py"), "--write-ladder", bench_v, "--box", box_v, + "--catalog", CAT, "--evidence", ev_rel + "/bench", "--box-evidence", ev_rel + "/box"], timeout=600) + if r.returncode != 0: + return False, "writer: " + (r.stdout or r.stderr).strip()[-200:] + g = sh([sys.executable, os.path.join(HERE, "catalog_gates.py"), "--fast", app], timeout=900) + if g.returncode != 0: + sh(["git", "-C", CAT, "checkout", "--", "templates/%s" % app]) + sh(["git", "-C", CAT, "clean", "-fdq", "templates/%s" % app]) + return False, "gates refused the entry — reverted: " + g.stdout.strip()[-200:] + sh(["git", "-C", CAT, "add", "templates/%s" % app]) + c = sh(["git", "-C", CAT, "commit", "-q", "-m", "%s: re-test of the same tag at a new digest (decision 52, R-740)\n\n%s\n\nEvidence: %s" + % (app, (r.stdout or "").strip(), ev_rel)]) + if c.returncode != 0: + return False, "commit failed: " + c.stderr.strip()[-160:] + if push: + p = sh(["git", "-C", CAT, "push", "-q", "origin", "main"], timeout=900) + if p.returncode != 0: + return False, "push refused (the commit stays local): " + (p.stdout + p.stderr).strip()[-200:] + return True, (r.stdout or "").strip().splitlines()[-1] + + +def main(argv): + ap = argparse.ArgumentParser() + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--only", default="") + ap.add_argument("--push", action="store_true") + ap.add_argument("--engines-only", action="store_true", + help="only database/redis lines (decision 52: start there). Measured 2026-09-30: EXACT tags are re-pushed " + "under the same name too (nextcloud 34.0.4-apache, linuxserver sonarr 4.0.20) — R-743") + ap.add_argument("--evidence", default=os.path.join(os.environ.get("SC", os.path.expanduser("~/.felhom-retest")), "evidence"), + help="evidence root; copy it into felhom.eu/documentation/audits/ for the record") + ap.add_argument("--evidence-rel", default="", help="the path the ladder entry cites (relative to the workspace root)") + a = ap.parse_args(argv) + only = set(x for x in a.only.split(",") if x) + allc = candidates(only) + cands = [c for c in allc if c[2] or not a.engines_only] + skipped = [c[0] for c in allc if c not in cands] + stamp = datetime.date.today().isoformat() + print("# retest-floating %s — catalog %s" % (stamp, sh(["git", "-C", CAT, "rev-parse", "--short", "HEAD"]).stdout.strip())) + if not cands: + print("nothing to re-test today: every ladder head's tested digest is what the registry serves" + + (" (engines only — not re-tested, not engines: %s)" % ", ".join(skipped) if skipped else "")) + return 0 + for app, moved, eng in cands: + print("%s%s:" % (app, " (engine)" if eng else "")) + for svc, (ref, old, new) in sorted(moved.items()): + print(" %s %s tested %s registry %s" % (svc, ref, (old or "-")[:19], new[:19] if not new.startswith("UNKNOWN") else new)) + if a.dry_run: + print("(dry run — nothing re-tested)") + return 0 + missing = prerequisites() + if missing: + print("CANNOT START — missing: " + "; ".join(missing)) + return 2 + s = sync_bench() + if "synced" not in (s.stdout or ""): + print("CANNOT START — the bench could not be synced: " + (s.stdout + s.stderr).strip()[-200:]) + return 2 + results = [] + for app, moved, _ in cands: + if any(v[2].startswith("UNKNOWN") for v in moved.values()): + results.append((app, False, "a registry could not be asked — not re-tested")) + continue + evdir = os.path.join(a.evidence, stamp, app) + rel = (a.evidence_rel.rstrip("/") + "/" + app) if a.evidence_rel else evdir + bv, why = run_bench(app, os.path.join(evdir, "bench")) + if not bv: + results.append((app, False, why)); continue + xv, why = run_box(app, moved, os.path.join(evdir, "box")) + if not xv: + results.append((app, False, why)); continue + ok, why = write(app, bv, xv, rel, a.push) + results.append((app, ok, why)) + print("\n# summary") + for app, ok, why in results: + print(" %-18s %s %s" % (app, "DONE " if ok else "STOPPED", why)) + return 0 if all(ok for _, ok, _ in results) else 1 + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/scripts/retest_box.py b/scripts/retest_box.py new file mode 100644 index 0000000..72588dd --- /dev/null +++ b/scripts/retest_box.py @@ -0,0 +1,158 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +"""retest_box.py (stdin: {"svc": [ref, tested_digest, new_digest], ...}) + +The BOX venue of a same-tag re-test (`09` §3 decision 52), on scratch guest 9202 pointed at the DRILL catalog. Called +by retest-floating.py; usable alone. Through the product's own endpoints only (box_walk.py): + 1. the drill checkout is pulled; its template for must be the live one (the runbook resets the drill first); + 2. is (re)installed fresh — the box renders the ladder head's TESTED (old) digest — and the running digest of + every re-tested service is read back: it must BE the old one, or the proof would start from the wrong image; + 3. the fixture seeds and reads back (C1); + 4. a DRILL-only commit appends the re-test entry (from == to, digest = new, digest_from = old); sync + rescan until + the box's catalog_digests carry the new digest; the badge is read (both languages); + 5. PRESS=update (default): the product's guarded Update; PRESS=chain: "run tonight's chain now" + (POST /api/debug/backup/night-chain) and the leg's own log line is quoted; + 6. the seed reads back, the running digest must be the NEW one, the badge is read again; + 7. /box-verdict-.json: proven only if 5 ended done, 6 read back and 6's digest is the new one. +The app is removed through the product at the end (KEEP=1 keeps it). +""" +import json +import os +import re +import subprocess +import sys +import time + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +import box_walk as w # noqa: E402 +import upgrade_fixtures_box as fixtures # noqa: E402 +import upgrade_fixtures_box28 as _f28 # noqa: E402 + +FX = dict(_f28.FIXTURES28) +FX.update(fixtures.FIXTURES) +DRILL = os.environ.get("DRILL", "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill") +# RETEST_SAME_AS: the checkout the drill must equal (default: this catalog). The end-to-end proof of 2026-09-30 simulated +# the month in the drill itself and set it to the drill. +CAT = os.environ.get("RETEST_SAME_AS") or os.path.dirname(HERE) + + +def main(): + app, evdir = sys.argv[1], sys.argv[2] + moved = json.loads(sys.stdin.read() or "{}") + os.makedirs(evdir, exist_ok=True) + log = open(os.path.join(evdir, "box.txt"), "a", buffering=1) + verdict = {"app": app, "verdict": "failed", "venue": "box 9202 (drill catalog), " + os.environ.get("PRESS", "update"), + "retested": {s: {"ref": v[0], "from_digest": v[1], "to_digest": v[2]} for s, v in moved.items()}} + + def say(*a): + w.say(*a) + log.write(" ".join(str(x) for x in a) + "\n") + + def finish(why): + verdict["why"] = why + json.dump(verdict, open(os.path.join(evdir, "box-verdict-%s.json" % app), "w"), indent=2) + say("RESULT %s — %s" % (verdict["verdict"], why)) + return 0 if verdict["verdict"] == "proven" else 1 + + def running_digests(): + st = w.stack(app) + ii = (st.get("app_config") or {}).get("installed_images") or {} + return {s: (ii.get(s) or {}).get("digest") for s in moved} + + fx = FX.get(app) + if fx is None: + return finish("no box fixture for %s" % app) + sub = getattr(fx, "sub", app) + w.login() + conf = w.guest("grep -A3 '^git:' /var/lib/docker/volumes/felhom-controller-data/_data/controller.yaml") + if "app-catalog-drill" not in conf: + return finish("9202 is not on the drill catalog (runbook §3) — nothing done") + subprocess.run(["git", "-C", DRILL, "pull", "-q", "--rebase", "origin", "main"], check=True) + for f in ("docker-compose.yml", ".felhom.yml"): + if open(os.path.join(DRILL, "templates", app, f)).read() != open(os.path.join(CAT, "templates", app, f)).read(): + return finish("the drill's %s/%s differs from this checkout's — reset the drill first (runbook §3)" % (app, f)) + if w.stack(app).get("deployed"): + say("removing the existing %s first (scratch box)" % app) + w.remove(app) + w.sync_rescan() + if not w.deploy(app, sub): + return finish("the install did not complete") + before = running_digests() + say("running digests after install: %s" % before) + wrong = {s: d for s, d in before.items() if d != moved[s][1]} + if wrong: + return finish("the box did not start at the tested digest: %s" % wrong) + tok = fx.seed(w, sub, say) + if tok is None or not fx.verify(w, sub, tok, say): + return finish("C1: the fixture could not seed and read back before the re-test") + verdict["seed_read_before"] = True + # the drill-only re-test entry + fy = os.path.join(DRILL, "templates", app, ".felhom.yml") + comp = os.path.join(DRILL, "templates", app, "docker-compose.yml") + sys.path.insert(0, HERE) + import ladder + entries, _, _ = ladder.parse(open(fy).read()) + head = entries[-1] + e = dict(head) + e["from"] = dict(head["to"]) + e["digest"] = dict(head["digest"]) + e["digest_from"] = dict(head["digest"]) + for s, v in moved.items(): + e["digest"][s] = v[2] + e["digest_from"][s] = v[1] + e["tested_at"], e["evidence"], e["box_evidence"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "DRILL", "DRILL (box proof in progress)" + for k in ("engine_conversion", "backfilled"): + e.pop(k, None) + open(fy, "w").write(ladder.append_entry(open(fy).read(), e)) + subprocess.run(["git", "-C", DRILL, "commit", "-q", "-am", "DRILL %s: re-test of the same tag %s (box proof)" % (app, {s: v[2][:19] for s, v in moved.items()})], check=True) + subprocess.run(["git", "-C", DRILL, "push", "-q", "origin", "main"], check=True, capture_output=True) + say("drill:", subprocess.run(["git", "-C", DRILL, "log", "--oneline", "-1"], capture_output=True, text=True).stdout.strip()) + for i in range(24): + w.sync_rescan() + cd = w.stack(app).get("catalog_digests") or {} + if all(cd.get(s) == v[2] for s, v in moved.items()): + say("the box's catalog_digests carry the new digest after %d sync round(s)" % (i + 1)) + break + time.sleep(5) + else: + return finish("the box never read the re-test's digest from the drill catalog") + verdict["badge_before"] = w.badges(app) + say("badge before: %s" % verdict["badge_before"]) + since = w.guest("date -u +%Y-%m-%dT%H:%M:%SZ").strip() + if os.environ.get("PRESS") == "chain": + code, d = w.ctl("POST", "/api/debug/backup/night-chain") + say("night chain -> %s %s" % (code, str(d)[:200])) + leg = "" + for _ in range(240): + time.sleep(10) + leg = w.guest("docker logs --since %s felhom-controller 2>&1 | grep -E 'update-leg' | grep -v DEBUG | cut -c1-300" % since) + if re.search(r"update leg .*: done=", leg): + break + say("the leg's own lines:\n" + leg) + verdict["leg_log"] = leg.strip().splitlines() + pressed = "%s: step pressed" % app in leg + ended = re.search(r"%s: step ended (\w+)" % re.escape(app), leg) + final = ended.group(1) if ended else None + if not pressed: + return finish("the leg did not press %s" % app) + else: + res = w.press_update(app, poll=1, cap_s=1800) + final = res.get("final_phase") + time.sleep(10) + after = running_digests() + read = fx.verify(w, sub, tok, say) + verdict.update({"final_phase": final, "digests_before": before, "digests_after": after, "seed_read_after": read, + "badge_after": w.badges(app), "from": w.stack(app).get("app_config", {}).get("pinned_images"), + "to": w.stack(app).get("app_config", {}).get("pinned_images"), "measured_at": since}) + say("after: phase=%s digests=%s read_back=%s badge=%s" % (final, after, read, verdict["badge_after"])) + if final == "done" and read and all(after.get(s) == v[2] for s, v in moved.items()): + verdict["verdict"] = "proven" + code = finish("the re-test step ran on the box" if verdict["verdict"] == "proven" else "the step did not end done / did not read back / runs another digest") + if not os.environ.get("KEEP"): + w.remove(app) + return code + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/test_gate_decoys.py b/scripts/test_gate_decoys.py index 82a9695..a47b328 100644 --- a/scripts/test_gate_decoys.py +++ b/scripts/test_gate_decoys.py @@ -457,6 +457,34 @@ def test_record_cases(clone): case_tr_static("DECOY: the step's .felhom.yml exists by NAME and holds no healthcheck", clone, [(NF, two), (step_rel, step_body), (meta_rel, lambda t: "display_name: Navidrome\n")], 1, ("not a real step file",)) + # ── `09` §3 decision 52: a RE-TEST — the same tag proved at a NEW digest (from == to). ───────────── + print("-- test-record: a same-tag re-test (decision 52)") + TR_D3 = "sha256:" + "c" * 64 + frm_step_rel = "templates/navidrome/" + _l.step_file(frm) + frm_meta_rel = "templates/navidrome/" + _l.step_meta_file(frm) + same_step = lambda t: io.open(os.path.join(clone, NC), encoding="utf-8").read() + rt = lambda dig, dfrom, **kw: tr_entry(frm, frm, {"navidrome": dig}, digest_from={"navidrome": dfrom}, **kw) + good_rt = rt(TR_D2, TR_D1, box_evidence="felhom.eu/documentation/audits/x/box/") + with_steps = [(frm_step_rel, same_step), (frm_meta_rel, meta_body)] + case_tr_static("GENUINE: head + a re-test from its digest, both venues", clone, + [(NF, ladder_of(head, good_rt))] + with_steps, 0) + case_tr_static("FACT: a re-test with NO new digest", clone, + [(NF, ladder_of(head, rt(TR_D1, TR_D1, box_evidence="x")))] + with_steps, 1, ("no new digest",)) + case_tr_static("FACT: a re-test without the box venue", clone, + [(NF, ladder_of(head, rt(TR_D2, TR_D1)))] + with_steps, 1, ("BOTH venues",)) + case_tr_static("FACT: a re-test FROM a digest the previous entry never tested", clone, + [(NF, ladder_of(head, rt(TR_D2, TR_D3, box_evidence="x")))] + with_steps, 1, ("starts from the previous entry",)) + case_tr_static("DECOY: a re-test that names digest_from only in its evidence text", clone, + [(NF, ladder_of(head, tr_entry(frm, frm, {"navidrome": TR_D2}, box_evidence="x", evidence="digest_from sha256:aaaa")))] + with_steps, + 1, ("needs digest_from",)) + rt_live = lambda t: tr_append(rt(TR_D2, TR_D1, box_evidence="felhom.eu/documentation/audits/x/box/"))(t) + case_tr_move("GENUINE: a re-test whose digest the registry serves now", clone, + [(NF, rt_live)] + with_steps, 0, ("test-record-move gate",), {old: TR_D2}) + case_tr_move("FACT: a re-test whose digest the registry no longer serves", clone, + [(NF, rt_live)] + with_steps, 1, ("re-test navidrome",), {old: TR_D3}) + case_tr_move("FACT: a re-test with no new digest reaches the move gate too", clone, + [(NF, tr_append(rt(TR_D1, TR_D1, box_evidence="x")))] + with_steps, 1, ("no new digest",), {old: TR_D1}) + def main(): gate = os.path.join(ROOT, "scripts", "check-engine-major.py") if not os.path.isfile(gate): @@ -475,6 +503,14 @@ def main(): DMDB = cur_image(clone, DOCMOST, "docmost-postgres") DM_MAJ = int(DMDB.split(":")[1].split("-")[0].split(".")[0]) DM_NEXT = DMDB.replace(":%d" % DM_MAJ, ":%d" % (DM_MAJ + 1), 1) + # kimai's own image too (2026-09-30): it moved apache-2.57.0 -> 2.67.0 through its ladder, and the typed tag here + # failed every run "fixture drifted" before a single case ran (R-663's shape a third time). Read it. + KAPP = cur_image(clone, KIMAI, "kimai") + _krepo, _ktag = KAPP.rsplit(":", 1) + _kpre = _ktag[:len(_ktag) - len(_ktag.lstrip("abcdefghijklmnopqrstuvwxyz-"))] + _kmaj, _kmin = (int(x) for x in _ktag[len(_kpre):].split(".")[:2]) + KAPP_MINOR = "%s:%s%d.%d.0" % (_krepo, _kpre, _kmaj, _kmin + 1) + KAPP_MAJOR = "%s:%s%d.0.0" % (_krepo, _kpre, _kmaj + 1) if not KDB.startswith("mariadb:11."): raise SystemExit("kimai-db is %s — the cases below assume a MariaDB 11 line; fixture drifted" % KDB) @@ -491,7 +527,7 @@ def main(): # R-469 + R-450 (2026-09-21): a MariaDB major bundled with the app's own bump is the # bookstack 0b73e5e shape — two migrations behind one edge — and stays refused. case("FACT: kimai-db 11.6 -> 12.3 BUNDLED with the kimai app bump", clone, - [(KIMAI, lambda t: swap_image("kimai", "kimai/kimai2:apache-2.57.0", "kimai/kimai2:apache-2.58.0")( + [(KIMAI, lambda t: swap_image("kimai", KAPP, KAPP_MINOR)( swap_image("kimai-db", KDB, "mariadb:12.3")(t)))], expect_rc=1, must_contain=("IN THE SAME COMMIT as kimai", "OWN EDGE", "R-450")) case("FACT: kimai-db mariadb:11.6 -> mariadb:lts (major unreadable)", clone, @@ -557,7 +593,7 @@ def main(): case("DECOY: major moves only in a comment + serverVersion env", clone, [(KIMAI, comment_and_env)], expect_rc=0, must_contain=("engine-major gate OK",)) case("DECOY: the APP image crosses a major (kimai 2.57 -> 3.0)", clone, - [(KIMAI, swap_image("kimai", "kimai/kimai2:apache-2.57.0", "kimai/kimai2:apache-3.0.0"))], + [(KIMAI, swap_image("kimai", KAPP, KAPP_MAJOR))], expect_rc=0, must_contain=("engine-major gate OK",)) case("DECOY: 'mariadb:12.3' lands in README.md, not a template", clone, [("README.md", lambda t: t + "\nDecoy: mariadb:11.6 -> mariadb:12.3 pending.\n")], @@ -570,24 +606,28 @@ def main(): CS = "check-catalog-since.py" def set_since(date): def _fn(text): - new = re.sub(r'^catalog_since:\s*"?\d{4}-\d{2}-\d{2}"?', 'catalog_since: "%s"' % date, text, count=1, flags=re.M) - if new == text: + new, n = re.subn(r'^catalog_since:\s*"?\d{4}-\d{2}-\d{2}"?', 'catalog_since: "%s"' % date, text, count=1, flags=re.M) + if n == 0: # the FIELD is missing (2026-09-30: "unchanged" is not drift — kimai's date was today) raise SystemExit("kimai's .felhom.yml carries no catalog_since — fixture drifted") return new return _fn + # The cases below need kimai's catalog_since to be a PAST date at HEAD (2026-09-30: kimai moved that day, so + # "untouched" and "set to today" were the same text). The throwaway clone gets one fixture commit. + edit(clone, KIMAI_FY, lambda t: re.sub(r'^catalog_since:.*$', 'catalog_since: "2026-01-01"', t, count=1, flags=re.M)) + commit(clone, "fixture: kimai catalog_since in the past") case("FACT: kimai image moves, catalog_since untouched", clone, - [(KIMAI, swap_image("kimai", "kimai/kimai2:apache-2.57.0", "kimai/kimai2:apache-2.58.0"))], + [(KIMAI, swap_image("kimai", KAPP, KAPP_MINOR))], expect_rc=1, must_contain=("CATALOG-SINCE GATE FAILED", "kimai", "catalog_since is still"), gate=CS) case("FACT: kimai image moves, catalog_since set to a FUTURE year", clone, - [(KIMAI, swap_image("kimai", "kimai/kimai2:apache-2.57.0", "kimai/kimai2:apache-2.58.0")), + [(KIMAI, swap_image("kimai", KAPP, KAPP_MINOR)), (KIMAI_FY, set_since("2036-09-13"))], expect_rc=1, must_contain=("in the future",), gate=CS) case("GENUINE: kimai image moves AND catalog_since = today", clone, - [(KIMAI, swap_image("kimai", "kimai/kimai2:apache-2.57.0", "kimai/kimai2:apache-2.58.0")), + [(KIMAI, swap_image("kimai", KAPP, KAPP_MINOR)), (KIMAI_FY, set_since(today))], expect_rc=0, must_contain=("catalog-since gate OK", "1 image move(s) dated"), gate=CS) case("DECOY: image moves; today's date lands in a COMMENT and README, the field stays", clone, - [(KIMAI, lambda t: swap_image("kimai", "kimai/kimai2:apache-2.57.0", "kimai/kimai2:apache-2.58.0")(t).replace("services:", "# catalog_since: %s\nservices:" % today, 1)), + [(KIMAI, lambda t: swap_image("kimai", KAPP, KAPP_MINOR)(t).replace("services:", "# catalog_since: %s\nservices:" % today, 1)), ("README.md", lambda t: t + "\ncatalog_since: %s (kimai)\n" % today)], expect_rc=1, must_contain=("CATALOG-SINCE GATE FAILED",), gate=CS) case("DECOY: only a comment + env line change, images untouched, date untouched", clone, diff --git a/scripts/test_ladder_writer.py b/scripts/test_ladder_writer.py index 2e89fbf..32627f1 100644 --- a/scripts/test_ladder_writer.py +++ b/scripts/test_ladder_writer.py @@ -184,5 +184,48 @@ class WriterTest(unittest.TestCase): self.assertEqual(open(sp).read(), before, "a refused restep must not write") + # --- a RE-TEST (decision 52): the same tag at a new digest ------------------------------------------------- + def run_retest(self, new_digest, registry, box_evidence="ev/box/"): + if "update_ladder:" not in open(self.fy).read(): + rc, out = self.run_writer(bench()); self.assertEqual(rc, 0, out) # the head: LIVE -> NEXT at digest D + ref = "deluan/navidrome:" + NEXT + b = bench(frm=NEXT, to=NEXT) + b["from"] = {"navidrome": ref + "@" + D} + b["to"] = {"navidrome": ref + "@" + new_digest} + image_digest.resolve = lambda r: (registry, None) + bp, xp = os.path.join(self.tmp, "rb.json"), os.path.join(self.tmp, "rx.json") + json.dump(b, open(bp, "w")); json.dump({"verdict": "proven", "to": {"navidrome": ref}}, open(xp, "w")) + args = [bp, "--box", xp, "--catalog", self.tmp, "--evidence", "ev/bench/"] + if box_evidence: + args += ["--box-evidence", box_evidence] + buf = io.StringIO() + with redirect_stdout(buf): + rc = ut.write_ladder(args) + return rc, buf.getvalue() + + def test_retest_writes_a_same_tag_entry_the_gate_accepts(self): + """COMPANION RED-PROOF: drop the `retest` digest_from block in write_ladder → the gate refuses 'needs digest_from'.""" + E = "sha256:" + "e" * 64 + rc, out = self.run_retest(E, E) + self.assertEqual(rc, 0, out) + entries, _, errs = ladder.parse(open(self.fy).read()) + self.assertEqual(errs, []) + e = entries[-1] + self.assertEqual(e["from"], e["to"]) + self.assertEqual(e["digest"], {"navidrome": E}) + self.assertEqual(e["digest_from"], {"navidrome": D}) + import subprocess + r = subprocess.run([sys.executable, os.path.join(HERE, "check-test-record.py"), "--root", self.tmp, "navidrome"], + capture_output=True, text=True) + self.assertEqual(r.returncode, 0, r.stdout) + + def test_retest_refuses_without_the_box_and_on_a_moved_digest(self): + E, F = "sha256:" + "e" * 64, "sha256:" + "f" * 64 + rc, out = self.run_retest(E, E, box_evidence=None) + self.assertEqual(rc, 1); self.assertIn("box venue", out) + rc, out = self.run_retest(E, F) # the registry moved again after the bench tested E + self.assertEqual(rc, 1); self.assertIn("re-run", out) + + if __name__ == "__main__": unittest.main(verbosity=2) diff --git a/scripts/upgrade-test.py b/scripts/upgrade-test.py index 5e83eba..b930e99 100755 --- a/scripts/upgrade-test.py +++ b/scripts/upgrade-test.py @@ -41,6 +41,8 @@ Usage: python3 upgrade-test.py [--soak SECONDS] [ …] (se --catalog --evidence [--box-evidence ] (the ONLY ladder writer) python3 upgrade-test.py --restep --definition --catalog --evidence (rewrites ONE superseded step's own definition from a bench re-proof of it) + python3 upgrade-test.py [--soak SECONDS] --retest [ ...] (the same tags at the registry's new + digest — `09` §3 decision 52; the writer then records a re-test entry) python3 upgrade-test.py --list --soak: how long the memory watch runs after a successful readback (default 600; 0 = off) Layout: templates under /opt/upg/templates, evidence under /opt/upg/evidence @@ -153,6 +155,29 @@ LOAD_PATHS = { "/api/stats", "/api/users/me", "/api/config", "/"], } LOAD_CONCURRENCY = 4 + +# BENCH-ONLY environment overrides (R-739, 2026-09-30): what the bench must change so an app can run at all without a +# public name and TLS. wanderer's web server calls its database at the PUBLIC `https://.` (the +# only address it reads — measured in the v0.20.0 and v0.21.0 images), which the bench has no name or certificate for; +# on a box traefik answers it. The bench points it at the database's own container instead. Every verdict that used an +# override carries it (`bench_overrides`), so the difference from a box is on the record, never silent. +BENCH_ENV_OVERRIDES = { + "wanderer": {"wanderer": {"PUBLIC_POCKETBASE_URL": "http://wanderer-db:8090"}}, +} + + +def apply_bench_overrides(app: str, compose_text: str) -> str: + over = BENCH_ENV_OVERRIDES.get(app) or {} + out, cur = [], None + for line in compose_text.splitlines(): + m = re.match(r"^ ([A-Za-z0-9_-]+):\s*$", line) + if m: + cur = m.group(1) + me = re.match(r"^(\s+-\s*)([A-Z0-9_]+)=(.*)$", line) + if me and cur in over and me.group(2) in over[cur]: + line = "%s%s=%s" % (me.group(1), me.group(2), over[cur][me.group(2)]) + out.append(line) + return "\n".join(out) + ("\n" if compose_text.endswith("\n") else "") MEMORY_TIGHT = 0.80 @@ -178,7 +203,7 @@ def render(app: str, images: dict, workdir: Path, env: dict, template: str = Non line = mi.group(1) + images[cur] out.append(line) workdir.mkdir(parents=True, exist_ok=True) - (workdir / "docker-compose.yml").write_text(pg_mounts_for("\n".join(out) + "\n")) + (workdir / "docker-compose.yml").write_text(apply_bench_overrides(app, pg_mounts_for("\n".join(out) + "\n"))) (workdir / ".env").write_text("".join(f"{k}={v}\n" for k, v in env.items())) return workdir / "docker-compose.yml" @@ -791,6 +816,7 @@ def run_edge(edge_id: str) -> dict: # memory is a MEASUREMENT beside the verdict; a kill or a restart during it DOES decide # the verdict, a tight peak only marks it (R-635, 2026-09-23). "memory": None, "marks": [], + "bench_overrides": BENCH_ENV_OVERRIDES.get(app) or None, "duration_s": 0, "measured_at": None, "evidence": f"evidence/{edge_id}"} t0 = time.time() log = [] @@ -979,6 +1005,51 @@ def add_move_edge(app: str, moves: list) -> str: return eid +def plain_refs(m: dict) -> dict: + """{svc: ref} with any `@sha256:…` dropped — the ladder's `from`/`to` name TAGS; digests ride in `digest`.""" + return {k: v.split("@", 1)[0] for k, v in (m or {}).items()} + + +def ref_digest(ref: str): + return ref.split("@", 1)[1] if "@" in ref else None + + +def add_retest_edge(app: str, services: list) -> str: + """`--retest [svc …]` (`09` §3 decision 52) → an edge FROM the template's refs at the digest the ladder's + newest entry TESTED, TO the same refs at the digest the registry serves NOW — the same-tag upstream fix. With no + services named, every service whose registry digest differs is re-tested; none differing is "nothing to re-test". + The method is the full one (seed, read-back, memory watch, abort) — no shortcut because only a digest changed.""" + sys.path.insert(0, str(Path(__file__).resolve().parent)) + import ladder, image_digest + frm = template_images(app, TEMPLATES) + entries, _, errs = ladder.parse((TEMPLATES / app / ".felhom.yml").read_text()) + if errs or not entries or entries[-1].get("to") != frm: + raise SystemExit(f"--retest {app}: the template has no ladder whose newest entry is its compose ({errs or 'no entry'})") + tested = entries[-1].get("digest") or {} + now = {} + for svc, ref in frm.items(): + d, why = image_digest.resolve(ref) + if not d: + raise SystemExit(f"--retest {app}: {svc} {ref}: the registry could not be asked ({why}) — nothing decided") + now[svc] = d + pick = services or [s for s in frm if now[s] != tested.get(s)] + for s in pick: + if s not in frm: + raise SystemExit(f"--retest {app}: no service {s!r} (one of {sorted(frm)})") + pick = [s for s in pick if now[s] != tested.get(s)] + if not pick: + raise SystemExit(f"--retest {app}: nothing to re-test — every service's registry digest equals the tested one") + f = dict(frm) + t = dict(frm) + for s in pick: + f[s] = f"{frm[s]}@{tested[s]}" + t[s] = f"{frm[s]}@{now[s]}" + eid = f"RT-{app}" + EDGES[eid] = dict(app=app, note="re-test of the same tag at a new digest (decision 52): " + ", ".join( + f"{s} {tested.get(s, '?')[:19]} -> {now[s][:19]}" for s in pick), frm=f, to=t) + return eid + + def write_ladder(argv) -> int: """`--write-ladder --box --catalog --evidence [--box-evidence ]` — THE ONLY WRITER of a ladder entry (`09` §6.4 part 4; never by hand). @@ -1007,19 +1078,28 @@ def write_ladder(argv) -> int: comp_p, fy_p = tdir / "docker-compose.yml", tdir / ".felhom.yml" comp = comp_p.read_text() cur = ladder.images_in(comp) - if cur != bench["from"]: - print(f"REFUSED {app}: the template is at {cur}, the bench tested FROM {bench['from']}") + bfrom, bto = plain_refs(bench["from"]), plain_refs(bench["to"]) + retest = bfrom == bto # `09` §3 decision 52: the same tags, tested at a new digest + if cur != bfrom: + print(f"REFUSED {app}: the template is at {cur}, the bench tested FROM {bfrom}") return 1 - if box.get("to") and {k: v for k, v in box["to"].items() if k in bench["to"]} != \ - {k: v for k, v in bench["to"].items() if k in box["to"]}: - print(f"REFUSED {app}: the box walked TO {box.get('to')}, the bench tested TO {bench['to']}") + btox = plain_refs(box.get("to") or {}) + if btox and {k: v for k, v in btox.items() if k in bto} != {k: v for k, v in bto.items() if k in btox}: + print(f"REFUSED {app}: the box walked TO {box.get('to')}, the bench tested TO {bto}") + return 1 + if retest and not arg("--box-evidence"): + print(f"REFUSED {app}: a re-test must cite the box venue (--box-evidence)") return 1 digests = {} - for svc, ref in sorted(bench["to"].items()): + for svc, ref in sorted(bto.items()): d, why = image_digest.resolve(ref) if not d: print(f"INCONCLUSIVE {app}: {svc} {ref}: {why}") return 2 + tested = ref_digest(bench["to"][svc]) + if tested and d != tested: + print(f"REFUSED {app}: {svc} {ref}: the bench tested {tested[:19]}, the registry serves {d[:19]} now — re-run") + return 1 digests[svc] = d conts = (bench["memory"].get("containers") or {}).values() anon = [c.get("anon_peak_pct") for c in conts if isinstance(c.get("anon_peak_pct"), (int, float))] @@ -1035,7 +1115,7 @@ def write_ladder(argv) -> int: # harness v4 (`09` §6.4 part 10): a PostgreSQL major is written ONLY when BOTH venues converted it — # the bench by its own conversion, the box by the PRODUCT's (the walk records the controller's # `CONVERTED` line). The mark is what lets the box convert at all, and the catalog gate reads it. - conv = pg_conversion(bench["from"], bench["to"]) + conv = pg_conversion(bfrom, bto) if conv: bc, xc = bench.get("engine_conversion") or {}, box.get("engine_conversion") or {} want = {k: conv[k] for k in ("service", "engine", "from", "to")} @@ -1046,7 +1126,7 @@ def write_ladder(argv) -> int: print(f"REFUSED {app}: the box did not convert {want} through the product (its record: {xc})") return 1 marks = set(bench.get("marks") or []) - entry = {"from": bench["from"], "to": bench["to"], "digest": digests, "verdict": "proven", + entry = {"from": bfrom, "to": bto, "digest": digests, "verdict": "proven", "tested_at": bench["measured_at"], "harness_version": bench["harness_version"], "evidence": arg("--evidence"), "box_evidence": arg("--box-evidence"), "memory_peak_pct": peak, "memory_basis": "anon" if anon else "cgroup_peak", @@ -1055,6 +1135,12 @@ def write_ladder(argv) -> int: "needs_person": None, "memory_tight": peak > ladder.MEMORY_TIGHT_PCT}} if conv: entry["engine_conversion"] = {k: conv[k] for k in ("service", "engine", "from", "to")} + if retest: + head = (ladder.parse(fy_p.read_text())[0] or [{}])[-1] + entry["digest_from"] = {s: (ref_digest(bench["from"][s]) or (head.get("digest") or {}).get(s)) for s in bto} + if head.get("to") != bto: + print(f"REFUSED {app}: a re-test must follow an entry that names the same refs (the head is {head.get('to')})") + return 1 probs = ladder.check_entry(entry) if probs: print(f"REFUSED {app}: the entry would not be well-formed: {probs}") @@ -1080,18 +1166,19 @@ def write_ladder(argv) -> int: if m: svc = m.group(1) mi = re.match(r"^(\s+image:\s*)(\S+)\s*$", line) - if mi and svc in bench["to"] and mi.group(2) == bench["from"][svc]: - line = mi.group(1) + bench["to"][svc] + if mi and svc in bto and mi.group(2) == bfrom[svc]: + line = mi.group(1) + bto[svc] out.append(line) comp_p.write_text(pg_mounts_for("\n".join(out) + "\n") if conv else "\n".join(out) + "\n") - if ladder.images_in(comp_p.read_text()) != bench["to"]: + if ladder.images_in(comp_p.read_text()) != bto: comp_p.write_text(comp) print(f"REFUSED {app}: the compose could not be moved line by line — restored") return 1 fy = fy_p.read_text() fy = re.sub(r'^catalog_since:.*$', 'catalog_since: "%s"' % _dt.date.today().isoformat(), fy, count=1, flags=re.M) fy_p.write_text(ladder.append_entry(fy, entry)) - print(f"WROTE {app}: {bench['from']} -> {bench['to']} peak {peak}% marks {entry['marks']}") + print(f"WROTE {app}: {'RE-TEST ' if retest else ''}{bfrom} -> {bto} peak {peak}% marks {entry['marks']}" + + (f" digest {entry['digest_from']} -> {digests}" if retest else "")) return 0 @@ -1158,6 +1245,8 @@ def main(argv): argv = argv[2:] if argv and argv[0] == "--move": argv = [add_move_edge(argv[1], argv[2:])] + if argv and argv[0] == "--retest": + argv = [add_retest_edge(argv[1], argv[2:])] if not argv or argv[0] == "--list": for k, v in EDGES.items(): print(f"{k:5s} {v['app']:12s} {v['note']}")