Files
felhom.eu/documentation/audits/night-2026-09-23/bakeoff.py
T
admin 0030cbd1de
gates / gates (push) Successful in 26s
night 2026-09-23: evidence so far (Parts A, B, C in progress)
The records the catalog's ladder entries cite (apps/<app>/bench and
apps/<app>/verdict.json), the drill tools, the spike, R-626's run,
R-612/R-613 red-proofs, the chaos schedule (seed 20260923) drawn
BEFORE round 1. Docs and register follow at the end of the night.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-23 21:33:54 +02:00

267 lines
14 KiB
Python

#!/usr/bin/env python3
"""bakeoff.py — `09` §3 decision 19: the two ways of keeping the undo's last-second copy, measured on
the same three apps (docmost / PostgreSQL, romm / MariaDB, vikunja / SQLite in a volume), guest 9202,
drill catalog.
F — copy the folder: the app's NAMED volumes copied with its containers stopped; put back on failure.
D — dump and load, as fixed by the morning spike: completion marker first; PostgreSQL drops and
recreates the dump's schemas inside the load's transaction; MariaDB drops every table, then loads.
EVIDENCE, NOT PRODUCT. Product acts go through the endpoints the UI invokes (deploy, backup, sync,
rescan, update, start, remove). The undo has no product path yet, so it is done by hand — the hold is
lifted with the operator CLI + a controller restart (the only exit that exists), and the product's own
Start supplies the app's secrets (they are never decrypted here).
Usage: python3 bakeoff.py <stage> <app> stages: prep fcopy break update undoF undoD cutoff state
"""
import json, os, sys, time
sys.path.insert(0, ".")
import walk as w
from spike import FX, SUB, load, save, ts, break_edge, docmost_seed_b, docmost_verify_b, romm_seed_b, \
romm_verify_b, vik_seed_b, vik_verify_b
EDGE = {"docmost": ("docmost/docmost:0.95.0", "docmost/docmost:0.96.0", 3000, 3999),
"romm": ("rommapp/romm:5.0.0", "rommapp/romm:5.3.0", 8080, 8999),
"vikunja": ("vikunja/vikunja:2.3.0", "vikunja/vikunja:2.6.0", 3456, 3999)}
UNIT = {"docmost": "/mnt/sys_drive/felhom-data/backups/primary/docmost",
"vikunja": "/mnt/sys_drive/felhom-data/backups/primary/vikunja",
"romm": "/mnt/felhom-drives/scratch_hdd/userdata/romm/backups/primary/romm"}
SUFFIX = ".pre-undo"
def seed_b(app, sub, A):
return {"docmost": docmost_seed_b, "romm": romm_seed_b, "vikunja": vik_seed_b}[app](sub, A)
def verify_b(app, sub, A, B):
r = {"docmost": docmost_verify_b, "romm": romm_verify_b, "vikunja": vik_verify_b}[app](sub, A, B)
return all(r.values()) if isinstance(r, dict) else r
def db_state(app):
"""The table count and the migration ledger, asked of the engine (or the SQLite file, read-only)."""
if app == "docmost":
return w.guest("docker exec docmost-postgres psql -U docmost -d docmost -Atc \"select count(*) from information_schema.tables where table_schema='public'\" 2>&1 | tr '\\n' ' '; "
"docker exec docmost-postgres psql -U docmost -d docmost -Atc \"select count(*)||' ledger, newest '||max(name) from kysely_migration\" 2>&1").strip()
if app == "romm":
q = lambda sql: f"docker exec romm-db sh -c 'mariadb -uroot -p\"$MYSQL_ROOT_PASSWORD\" -N -e \"{sql}\" romm' 2>&1 | tr '\\n' ' '"
base_tables = q("select count(*) from information_schema.tables where table_schema=database() and table_type=0x42415345205441424C45")
alembic = q("select version_num from alembic_version")
return w.guest(f'echo "tables=$({base_tables}) alembic=$({alembic})"').strip()
return w.guest("""python3 - <<'PY'
import sqlite3
c=sqlite3.connect('file:/var/lib/docker/volumes/vikunja_vikunja_db/_data/vikunja.db?mode=ro',uri=True)
t=c.execute("select count(*) from sqlite_master where type='table'").fetchone()[0]
m=c.execute("select count(*), max(id) from migration").fetchone()
print(f"tables={t} ledger={m[0]} newest={m[1]}")
PY""").strip()
def volumes(app):
return [v for v in w.guest(f"docker volume ls -q --filter label=com.docker.compose.project={app}").split()
if not v.endswith(SUFFIX)]
def containers(app):
return w.guest(f"docker ps -a --filter label=com.docker.compose.project={app} --format '{{{{.Names}}}}'").split()
def stage_prep(app):
s = {"app": app, "sub": SUB[app]}
w.say(f"=== prep {app}: deploy at the drill FROM pin")
w.say("deployed:", w.deploy(app, s["sub"]))
s["seedA"] = FX[app].seed(w, s["sub"], w.say)
w.say("C1 A:", FX[app].verify(w, s["sub"], s["seedA"], w.say))
w.backup_now(app)
w.wait_app(s["sub"], {"docmost": "/", "romm": "/api/heartbeat", "vikunja": "/api/v1/info"}[app], want=("200",), tries=60)
s["seedB"] = seed_b(app, s["sub"], s["seedA"])
w.say("B reads back:", verify_b(app, s["sub"], s["seedA"], s["seedB"]))
s["volumes"] = volumes(app)
w.say("named volumes:", s["volumes"])
save(app, s)
def stage_fcopy(app):
"""F at safety-dump time: stop, copy every named volume (cp -a into a sibling volume, and
separately a tar, for the rate), start. Measures the EXTRA downtime and the disk."""
s = load(app)
s["state_before_update"] = db_state(app)
w.say("db state before the update:", s["state_before_update"])
vols = s["volumes"]
w.say(w.guest(f"cp /opt/docker/stacks/{app}/docker-compose.yml /root/pre-{app}-compose.yml; grep -m1 'image: {EDGE[app][0]}' /root/pre-{app}-compose.yml"))
script = f"""
set -u
cs=$(docker ps --filter label=com.docker.compose.project={app} --format '{{{{.Names}}}}')
t0=$(date +%s.%N)
docker stop $cs >/dev/null
t1=$(date +%s.%N)
for v in {' '.join(vols)}; do
docker volume rm -f "$v{SUFFIX}" >/dev/null 2>&1
docker volume create "$v{SUFFIX}" >/dev/null
a=$(date +%s.%N)
docker run --rm -v "$v":/from:ro -v "$v{SUFFIX}":/to alpine:3.20 sh -c 'cp -a /from/. /to/ && sync' || echo "COPY FAILED $v"
b=$(date +%s.%N)
bytes=$(docker run --rm -v "$v":/from:ro alpine:3.20 du -sb /from | cut -f1)
files=$(docker run --rm -v "$v":/from:ro alpine:3.20 sh -c 'find /from | wc -l')
cbytes=$(docker run --rm -v "$v{SUFFIX}":/to:ro alpine:3.20 du -sb /to | cut -f1)
cfiles=$(docker run --rm -v "$v{SUFFIX}":/to:ro alpine:3.20 sh -c 'find /to | wc -l')
mkdir -p /root/bk; c=$(date +%s.%N)
docker run --rm -v "$v":/vol:ro -v /root/bk:/out alpine:3.20 tar cf /out/$v.tar -C /vol . ; d=$(date +%s.%N)
tb=$(stat -c %s /root/bk/$v.tar); rm -f /root/bk/$v.tar
python3 -c "print('VOL $v bytes=$bytes files=$files | copy bytes=$cbytes files=$cfiles | cp -a %.2fs | tar %.2fs (%s B)' % ($b-$a, $d-$c, '$tb'))"
done
t2=$(date +%s.%N)
docker start $cs >/dev/null
t3=$(date +%s.%N)
python3 -c "print('stop %.2fs copy(all, cp -a + the tar measurement) %.2fs start %.2fs' % ($t1-$t0, $t2-$t1, $t3-$t2))"
df -B1 --output=avail /var/lib/docker | tail -1 | awk '{{printf "free on the docker root: %.2f GiB\\n", $1/2^30}}'
"""
out = w.guest(script, timeout=1800)
w.say(out)
t0 = time.time()
w.wait_app(s["sub"], {"docmost": "/", "romm": "/api/heartbeat", "vikunja": "/api/v1/info"}[app], want=("200",), tries=60, delay=2)
w.say(f"front door answering again {round(time.time()-t0,1)}s after the start")
s["fcopy"] = out
s["fcopy_at"] = ts()
save(app, s)
def stage_break(app):
s = load(app)
frm, to, p0, p1 = EDGE[app]
s["break_commit"] = break_edge(app, frm, to, p0, p1)
w.say("badge caught up after", w.sync_rescan(app, to), "s")
save(app, s)
def stage_update(app):
s = load(app)
r = w.press_update(app, poll=1.0)
w.say(json.dumps({k: r[k] for k in ("final_phase", "hold_reason", "state", "duration_s")}, ensure_ascii=False))
s.setdefault("updates", []).append(r)
save(app, s)
LIFT = """docker exec felhom-controller /usr/local/bin/felhom-controller --clear-restore-hold {app} 2>&1 | grep -E 'CLEARED|no restore hold'
docker restart felhom-controller >/dev/null; sleep 15"""
# The OLD definition comes from the bake-off's OWN pre-update copy (/root/pre-<app>-compose.yml, taken
# in fcopy), NEVER from the recovery unit: measured 2026-09-23 10:59, the unit was re-captured 10 s
# after the hold was lifted — with the NEW definition — and a pin-back that read it started the new
# version again. That is R-639 seen live, and why the product undo keeps its own copies.
PINBACK = """S=/opt/docker/stacks/{app}; P=/root/pre-{app}-compose.yml
grep -m1 'image: .*{frm}' $P >/dev/null || {{ echo "PRE-UPDATE COPY MISSING OR WRONG: $P"; exit 1; }}
cp $P $S/applied-compose.yml; cp $P $S/docker-compose.yml
sed -i 's#^\\(\\s*{svc}: \\){to}$#\\1{frm}#' $S/app.yaml; sed -n '/^pinned_images:/,$p' $S/app.yaml | head -4"""
def lift_and_pinback(app):
frm, to, _, _ = EDGE[app]
t = time.time()
w.say(w.guest(LIFT.format(app=app)))
w.say(w.guest(PINBACK.format(unit=UNIT[app], app=app, svc=app, frm=frm, to=to)))
w.login()
return round(time.time() - t, 1)
def stage_undoF(app):
s = load(app)
w.say(f"=== undo by F: {app}")
w.say("hold lift + pin back took", lift_and_pinback(app), "s (not part of a product undo)")
vols = s["volumes"]
script = f"""
t0=$(date +%s.%N)
cs=$(docker ps -a --filter label=com.docker.compose.project={app} -q); [ -n "$cs" ] && docker stop $cs >/dev/null
for v in {' '.join(vols)}; do
docker run --rm -v "$v{SUFFIX}":/from:ro -v "$v":/to alpine:3.20 sh -c 'rm -rf /to/..?* /to/.[!.]* /to/* ; cp -a /from/. /to/ && sync' || echo "RESTORE FAILED $v"
done
python3 -c "import time;print('volumes put back in %.2fs' % (time.time()-$t0))"
"""
w.say(w.guest(script, timeout=1800))
t0 = time.time()
c, d = w.ctl("POST", f"/api/stacks/{app}/start")
w.say("product start ->", c)
up = w.wait_app(s["sub"], {"docmost": "/", "romm": "/api/heartbeat", "vikunja": "/api/v1/info"}[app], want=("200",), tries=60, delay=2)
th = round(time.time() - t0, 1)
A = FX[app].verify(w, s["sub"], s["seedA"], w.say)
B = verify_b(app, s["sub"], s["seedA"], s["seedB"])
st = db_state(app)
w.say(f"F RESULT: healthy={up} after {th}s A={A} B={B} db now: {st} (before the update: {s['state_before_update']})")
s["undoF"] = {"healthy": up, "health_s": th, "A": A, "B": B, "db": st}
save(app, s)
def stage_undoD(app):
"""D, on the SECOND failed update: its safety dump (written by the product, phase 3) is the copy."""
s = load(app)
w.say(f"=== undo by D: {app}")
w.say("hold lift + pin back took", lift_and_pinback(app), "s (not part of a product undo)")
if app == "vikunja":
w.say("vikunja has no database server: the product wrote NO safety dump (R-641). D's copy for it "
"is a volume tar at safety-dump time, which is F with extra steps — recorded, not re-run.")
return
c, d = w.ctl("POST", f"/api/stacks/{app}/start") # the product supplies the env; the app half will refuse
time.sleep(8)
if app == "docmost":
load_cmd = """{ echo 'DROP SCHEMA public CASCADE; CREATE SCHEMA public;'; cat "$D"; } | docker exec -i docmost-postgres psql -v ON_ERROR_STOP=1 --single-transaction -U docmost -d docmost > /root/dload.out 2>&1"""
marker = "-- PostgreSQL database dump complete"
else:
load_cmd = """{ echo 'SET FOREIGN_KEY_CHECKS=0;'; docker exec romm-db sh -c 'mariadb -uroot -p"$MYSQL_ROOT_PASSWORD" -N -e "select concat(\\"DROP TABLE IF EXISTS \\`\\",table_name,\\"\\`;\\") from information_schema.tables where table_schema=\\"romm\\" and table_type=\\"BASE TABLE\\""' ; cat "$D"; } | docker exec -i romm-db sh -c 'mariadb -uroot -p"$MYSQL_ROOT_PASSWORD" romm' > /root/dload.out 2>&1"""
marker = "-- Dump completed"
script = f"""
D=$(ls -t {UNIT[app]}/db-dumps/pre-restore-*.sql | head -1); echo "undo copy: $(basename $D) $(stat -c %s $D) B"
docker stop {app} >/dev/null 2>&1; docker update --restart=no {app} >/dev/null
t0=$(date +%s.%N)
if tail -n 5 "$D" | grep -q -- '{marker}'; then echo "marker present"; else echo "MARKER ABSENT - refusing"; exit 0; fi
{load_cmd}; echo "load rc=$?"; tail -2 /root/dload.out
python3 -c "import time;print('marker check + load %.2fs' % (time.time()-$t0))"
docker update --restart=unless-stopped {app} >/dev/null; docker start {app} >/dev/null
"""
w.say(w.guest(script, timeout=900))
t0 = time.time()
up = w.wait_app(s["sub"], {"docmost": "/", "romm": "/api/heartbeat"}[app], want=("200",), tries=60, delay=2)
th = round(time.time() - t0, 1)
A = FX[app].verify(w, s["sub"], s["seedA"], w.say)
B = verify_b(app, s["sub"], s["seedA"], s["seedB"])
st = db_state(app)
w.say(f"D RESULT: healthy={up} after {th}s A={A} B={B} db now: {st} (before the update: {s['state_before_update']})")
s["undoD"] = {"healthy": up, "health_s": th, "A": A, "B": B, "db": st}
save(app, s)
def stage_cutoff(app):
"""The cut-off copy, both methods, DETECTED before anything is loaded or swapped.
F: a copy killed half-way (timeout) — compared with the source and checked for the finished-marker
the build would write only after cp exits 0. D: the safety dump cut in half — the marker check."""
s = load(app)
big = max(s["volumes"], key=lambda v: int(w.guest(f"docker run --rm -v {v}:/v:ro alpine:3.20 du -sb /v | cut -f1").strip() or 0))
script = f"""
v={big}
docker volume rm -f $v.cut >/dev/null 2>&1; docker volume create $v.cut >/dev/null
cs=$(docker ps --filter label=com.docker.compose.project={app} --format '{{{{.Names}}}}'); docker stop $cs >/dev/null
# Kill the copy CONTAINER, not the client (killing `docker run` leaves the container copying — measured
# on docmost 2026-09-23, rc 137 with a complete copy and a marker).
docker run -d --name cutcopy -v $v:/from:ro -v $v.cut:/to alpine:3.20 sh -c 'cp -a /from/. /to/ && touch /to/.felhom-copy-complete' >/dev/null
sleep 0.02; docker kill cutcopy >/dev/null 2>&1; echo "copy container exit=$(docker wait cutcopy) (137 = killed)"; docker rm -f cutcopy >/dev/null 2>&1
echo "source bytes=$(docker run --rm -v $v:/f:ro alpine:3.20 du -sb /f | cut -f1) cut copy bytes=$(docker run --rm -v $v.cut:/f:ro alpine:3.20 du -sb /f | cut -f1)"
echo "finished-marker in the cut copy: $(docker run --rm -v $v.cut:/f:ro alpine:3.20 sh -c 'ls /f/.felhom-copy-complete 2>/dev/null | wc -l') -> F refuses to swap"
docker volume rm -f $v.cut >/dev/null; docker start $cs >/dev/null
D=$(ls -t {UNIT[app]}/db-dumps/pre-restore-*.sql 2>/dev/null | head -1)
if [ -n "$D" ]; then head -c $(( $(stat -c %s $D) / 2 )) $D > /root/cut.sql
echo "D: whole copy marker: $(tail -n5 $D | grep -cE -- '-- (PostgreSQL database dump complete|Dump completed)') cut copy marker: $(tail -n5 /root/cut.sql | grep -cE -- '-- (PostgreSQL database dump complete|Dump completed)') -> D refuses to load"
rm -f /root/cut.sql
else echo "D: no safety dump exists for this app (R-641)"; fi
"""
w.say(f"=== cut-off copy: {app} (largest volume {big})")
out = w.guest(script, timeout=600)
w.say(out)
s["cutoff"] = out
save(app, s)
if __name__ == "__main__":
w.login()
{"prep": stage_prep, "fcopy": stage_fcopy, "break": stage_break, "update": stage_update,
"undoF": stage_undoF, "undoD": stage_undoD, "cutoff": stage_cutoff,
"state": lambda a: w.say(db_state(a))}[sys.argv[1]](sys.argv[2])