Files
felhom.eu/documentation/audits/the-28-2026-09-22/walk28.py
T
admin 186546d562
gates / gates (push) Successful in 27s
THE TWENTY-EIGHT: every app no drill had touched, walked in one night
All 28 walked on scratch guest 9202 against the private drill catalog. 26 deployed, 6 proven,
5 inconclusive, 14 with no upstream edge, 1 failed honestly (outline 1.9.1->1.10.1, HELD with the
right sentence), 2 undeployable - one (plant-it) by design, refused by the lifecycle gate, proven
live for the first time. Each app also got the half the update night skipped: a restore from its own
copy with the seed read back again - 21 restored, 2 correctly REFUSED per 07 6.2.

R-630 RAISED TO P1 by measurement: a stack with NO probe container does not skip verifying - it
waits out the full health timeout and HOLDS, stopping an app whose three containers read healthy.
The controller's own words: "not healthy within 5m0s (last: no probe container)".

R-633 opened: a remove sent during a restore reports success and leaves a container restarting with
a live public route. The product already refuses that clash for update and for restore, naming the
blocker; remove has no such guard.

R-634 opened: an app can be running, healthy and serving while recorded as deployed=false, and is
then unremovable. Reproducible alone on sparkyfitness; concurrency-linked on two others.

R-631 and R-632 CLOSED. Register 321 -> 323. Seven interventions, six of them my own harness -
named, with what each cost. No product code. The live catalog's image: lines are byte-identical to
the start of the night.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-22 16:00:56 +02:00

239 lines
12 KiB
Python

#!/usr/bin/env python3
"""walk28.py <app> — ONE of the twenty-eight, the full rotation walk, through the product.
Reuses `update-night-2026-09-21/walk.py` wholesale (R-161: do not rewrite the method). What is NEW
here versus the update night is step 6: the app is RESTORED from its own backup and the seed read
back a second time. The update night skipped that half.
Every app gets a line even when it cannot be deployed. `inconclusive` is never collapsed into
`failed` or `proven`, and an app with no non-browser seed route is `inconclusive` WITH WHAT WAS
TRIED — never faked.
"""
import argparse, json, os, re, sys, time, traceback
from datetime import datetime, timezone
HERE = os.path.dirname(os.path.abspath(__file__))
NIGHT = os.path.join(os.path.dirname(HERE), "update-night-2026-09-21")
sys.path.insert(0, NIGHT)
sys.path.insert(0, HERE)
import walk as w # noqa: E402
from fixtures28 import FIXTURES28 # noqa: E402
EDGES = json.load(open(os.path.join(HERE, "edges.json"))) if os.path.exists(
os.path.join(HERE, "edges.json")) else {}
META = json.load(open("/tmp/claude-1000/-mnt-5-hdd-felhom-eu-git/"
"d029e2e6-1762-440e-956d-0760c8aea4b3/scratchpad/drift28/meta.json"))
LOCK = "/tmp/claude-1000/-mnt-5-hdd-felhom-eu-git/d029e2e6-1762-440e-956d-0760c8aea4b3/scratchpad/drill.lock"
def drill_lock(fn, *a, **k):
"""The drill repo is ONE git clone; two workers pushing at once corrupt each other's index."""
for _ in range(600):
try:
fd = os.open(LOCK, os.O_CREAT | os.O_EXCL | os.O_WRONLY)
os.close(fd)
break
except FileExistsError:
time.sleep(1)
try:
return fn(*a, **k)
finally:
try:
os.unlink(LOCK)
except OSError:
pass
def main():
ap = argparse.ArgumentParser()
ap.add_argument("app")
ap.add_argument("--cap", type=int, default=1800)
a = ap.parse_args()
app = a.app
d = os.path.join(HERE, "apps", app)
os.makedirs(d, exist_ok=True)
# per-worker session files: walk.py keeps the cookie in SC, and two workers racing on one
# cookie file log each other out. Assigning the module global is enough — every helper reads
# it at CALL time.
w.SC = os.path.join("/tmp/claude-1000/-mnt-5-hdd-felhom-eu-git/"
"d029e2e6-1762-440e-956d-0760c8aea4b3/scratchpad/w", app)
os.makedirs(w.SC, exist_ok=True)
import shutil
shutil.copy("/tmp/claude-1000/-mnt-5-hdd-felhom-eu-git/"
"d029e2e6-1762-440e-956d-0760c8aea4b3/scratchpad/.ctlpw", w.SC + "/.ctlpw")
t0 = time.time()
m = META.get(app, {})
fx = FIXTURES28.get(app)
sub = getattr(fx, "sub", None) or app
edge = EDGES.get(app)
rec = {"harness_version": 2, "app": app,
"venue": "guest 9202 demo-hp-scratch, controller 0.261.0, drill catalog",
"class": m.get("class") or ("db" if m.get("engines") else "file-leg"),
"from": {}, "to": {}, "verdict": "inconclusive",
"deployed": False, "seed_route": None, "seed_read_before": False,
"seed_read_after_update": False, "backup": None,
"restore_verdict": "not-attempted", "seed_read_after_restore": False,
"healthy_after": False, "migration_observed": None, "removed_clean": None,
"edge": edge, "duration_s": 0,
"measured_at": datetime.now(timezone.utc).isoformat(),
"evidence": f"the-28-2026-09-22/apps/{app}/", "notes": []}
w.say(f"==== {app} (sub={sub}, class={rec['class']}, edge={edge or 'none'})")
try:
w.login()
# ---- 1 deploy at the LIVE pin -------------------------------------------------------
ok = w.deploy(app, sub)
st = w.stack(app)
rec["deployed"] = bool(st.get("deployed"))
rec["from"] = (st.get("app_config") or {}).get("pinned_images") or {}
if not rec["deployed"]:
rec["verdict"] = "could-not-deploy"
rec["notes"].append("deploy never reached `deployed` with a pin")
raise SystemExit(0)
front = w.app_curl(sub, "/")
rec["front_door_after_deploy"] = {"rc": front[0], "code": front[1]}
w.say(f" [1] front door {front[1]} controller state={st.get('state')}")
# ---- 2/3 seed through the app's OWN front door, then read it back (C1) ---------------
tok = None
if fx is None:
rec["seed_route"] = "none written"
rec["notes"].append("no fixture: no non-browser seed route was written for this app")
w.say(" [2] NO FIXTURE — recorded inconclusive for the data half, not faked")
else:
tok = fx.seed(w, sub, w.say)
rec["seed_route"] = getattr(fx, "route", fx.__class__.__name__)
if tok is None:
rec["notes"].append("fixture ran and found no non-browser seed route: "
+ (getattr(fx, "tried", "") or "see log"))
w.say(" [2] fixture found no route — inconclusive for the data half")
else:
rec["seed_read_before"] = bool(fx.verify(w, sub, tok, w.say))
w.say(f" [3] C1 readback before = {rec['seed_read_before']}")
# ---- 4 „Mentés most" -----------------------------------------------------------------
w.backup_now(app)
snaps = w.snapshots(app)
rec["backup"] = {"snapshots_offered": len(snaps),
"first": (snaps[0] if snaps else None)}
w.say(f" [4] backups page offers {len(snaps)} restorable copy(ies)")
# ---- 5 the edge, if one exists --------------------------------------------------------
if edge:
h = drill_lock(w.drill_bump, app, edge["from"], edge["to"])
rec["drill_commit"] = h
secs = w.sync_rescan(app, edge["to"].split(",")[0])
rec["badge_seconds"] = secs
rec["badges"] = w.badges(app)
w.say(f" [5] badge {rec['badges']}")
ph = w.press_update(app)
rec["phases"] = ph
# R-621: the hold destroys the app's logs, so they are captured DURING the wait,
# not after. press_update polls; this is the best moment available to a caller.
logs = w.app_logs(app, 300)
open(os.path.join(d, "app-logs-during-after.txt"), "w").write(logs or "")
for line in (logs or "").split("\n"):
if any(k in line.lower() for k in ("migrat", "upgrad", "schema", "alter table")):
rec["migration_observed"] = line.strip()[:300]
break
obs = w.observables(app)
rec["observables"] = obs
rec["to"] = (w.stack(app).get("app_config") or {}).get("pinned_images") or {}
# The Update can be REFUSED before any phase exists — `409 busy` while a backup or a
# restore is in flight, which is the product guarding itself and is a RESULT, not an
# error. An empty phase list then made `[-1]` raise, which cost rallly its whole walk.
plist = ph if isinstance(ph, list) else (ph.get("phases") or [])
last = plist[-1].get("phase") if plist else None
if not last and isinstance(ph, dict) and ph.get("http") == "409":
last = "refused-" + str((ph.get("refusal") or {}).get("data", {}).get("reason")
or "409")
rec["update_refusal"] = (ph.get("refusal") or {}).get("error")
rec["update_phase_final"] = last
if tok is not None:
rec["seed_read_after_update"] = bool(fx.verify(w, sub, tok, w.say))
st = w.stack(app)
rec["healthy_after"] = st.get("state") in ("running",)
if last == "done" and (tok is None or rec["seed_read_after_update"]):
rec["verdict"] = "proven" if tok is not None else "inconclusive"
elif last == "failed":
rec["verdict"] = "failed"
else:
rec["verdict"] = "no-edge"
# ---- 6 RESTORE — the half the update night skipped -------------------------------------
if snaps:
r = w.restore(app)
rec["restore_raw"] = str(r)[:400]
# A REFUSAL IS NOT A FAILURE, and collapsing the two would have mis-recorded the very
# behaviour `07` §6.2 predicts. A class-A app's local copy holds no readable file leg,
# and the product refuses rather than restoring a database over files it does not have
# — naming the action that DOES work. The refusal travels as a `flash_error` on the
# redirect, so it is read from there and quoted, urldecoded, in the record.
from urllib.parse import unquote_plus
loc = " ".join(r.get("location") or []) if isinstance(r, dict) else ""
# NOT `m` — that name already holds this app's META entry, and shadowing it made the
# teardown fall over with `'re.Match' object has no attribute 'get'` on paperless-ngx.
fm = re.search(r"flash_error=([^&\s]+)", loc)
rec["restore_refusal"] = unquote_plus(fm.group(1)) if fm else None
time.sleep(15)
st = w.stack(app)
if rec["restore_refusal"]:
rec["restore_verdict"] = "refused-with-a-sentence"
elif st.get("state") in ("running", "unhealthy"):
rec["restore_verdict"] = "ok"
else:
rec["restore_verdict"] = "failed"
rec["restore_state_seen"] = st.get("state")
if tok is not None:
rec["seed_read_after_restore"] = bool(fx.verify(w, sub, tok, w.say))
w.say(f" [6] restore -> {rec['restore_verdict']}, seed back = "
f"{rec['seed_read_after_restore']}")
# WAIT FOR THE RESTORE TO SETTLE BEFORE REMOVING. `POST /backup/restore` answers 302
# and works in the background; a remove sent while its `compose up` is still running
# tears down what exists and the restore then RE-CREATES it — the controller records
# the app as removed and a container keeps restarting with a live traefik route.
# Measured on gokapi at 11:34 (R-633). Leaving the race in would mean every later app
# measured the harness rather than the product.
for _ in range(36):
s = w.stack(app)
if not s.get("restore_running") and s.get("state") in (
"running", "unhealthy", "degraded", "stopped", "not_deployed", None):
break
time.sleep(5)
time.sleep(10)
else:
rec["restore_verdict"] = "no-snapshot-offered"
except SystemExit:
pass
except Exception as e:
rec["notes"].append(f"harness error: {type(e).__name__}: {e}")
open(os.path.join(d, "traceback.txt"), "w").write(traceback.format_exc())
w.say(f" !! {type(e).__name__}: {e}")
finally:
# ---- 7 remove, then R-626: check 60 s later that nothing came back --------------------
try:
w.remove(app)
time.sleep(60)
back = w.guest(f"docker ps -a --format '{{{{.Names}}}}' | grep -x '{app}' || true; "
f"ls /opt/docker/stacks/{app}/app.yaml 2>/dev/null || true")
rec["removed_clean"] = (back.strip() == "")
rec["remove_leftovers"] = back.strip()[:300]
w.say(f" [7] 60 s after remove: clean={rec['removed_clean']} {back.strip()[:120]!r}")
# drop the app's images BY NAME — never `prune` (rule 3); 9202's root disk is 28 GB
imgs = [v for v in (m.get("all_images") or {}).values() if v]
if imgs:
w.guest("docker image rm -f " + " ".join(imgs) + " 2>&1 | tail -2")
except Exception as e:
rec["notes"].append(f"teardown error: {type(e).__name__}: {e}")
rec["duration_s"] = round(time.time() - t0, 1)
json.dump(rec, open(os.path.join(d, "verdict.json"), "w"), ensure_ascii=False, indent=2)
open(os.path.join(d, "log.txt"), "w").write("\n".join(w.LOG) + "\n")
w.say(f" [9] verdict {rec['verdict']} ({rec['duration_s']}s) -> apps/{app}/verdict.json")
if __name__ == "__main__":
main()