Files
felhom.eu/documentation/audits/night-2026-09-23/chaos.py
T
admin 3e58c184f6
gates / gates (push) Successful in 27s
night shift 2026-09-23: the record, the register, the morning note
DRILL-night-2026-09-23.md: Parts A-E. 09 §3 decisions 21 (operator word),
22 and 23 (CC unattended, operator may reverse); §6.4 parts 4 and 6
(catalog half) shipped; §6.1a residuals R-658/R-659. Register 330 -> 336:
R-651..R-660 opened (R-658 and R-659 P1), R-650/R-640/R-499/R-626 closed.
Capability map, nightly rotation (opengist), STATUS (one question: the
floor), CONTEXT, REPORT. The floor stays 0.266.0.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-24 00:24:04 +02:00

414 lines
20 KiB
Python

#!/usr/bin/env python3
"""chaos.py — Part D, the chaos hour on the undo. Guest 9202, controller v0.267.0, drill catalog.
python3 chaos.py setup # six apps at their FROM pins, seeded (A), one whole-box backup, seed B
python3 chaos.py round <N> # one round of chaos_schedule.py, evidence written as it happens
python3 chaos.py readback # every app's seeds, read through its own front door
EVIDENCE, NOT PRODUCT. Every act is the endpoint the UI invokes; every accident is something a real
box suffers (a power cut, a killed process, a full disk, a memory hog, a backup, a second press).
The app a round acts on is fixed by the table below, NOT drawn: an app's state after round N decides
what round N+1 can do to it, so the pairing is written here and in the findings doc before round 1.
The five things recorded every round: what the household saw (the app page in BOTH languages), what
the box did by itself (phases + the controller's own log lines), time to steady, which events fired
(9202 has no hub, so R-620's "DROPPED event" lines are the witness), and — judged in the findings doc
against `08-alarm-ladder.md` — what should have fired and did not. Plus the data read back.
"""
import json, os, re, subprocess, sys, threading, time
sys.path.insert(0, ".")
import walk as w
import fixtures as fx
import live
from spike import docmost_seed_b, docmost_verify_b, romm_seed_b, romm_verify_b, vik_seed_b, vik_verify_b, \
break_edge
from chaos_schedule import draw
HERE = os.path.dirname(os.path.abspath(__file__))
STATE = os.path.join(HERE, "chaos-state.json")
EVD = os.path.join(HERE, "chaos")
os.makedirs(EVD, exist_ok=True)
# the six (brief Part D): two PostgreSQL, one MariaDB, one volume-only, one file-leg, one big database
SIX = {
"docmost": {"fx": fx.Docmost(), "class": "PostgreSQL + the BIG database (300 extra spaces)", "from": "docmost/docmost:0.95.0", "to": "docmost/docmost:0.96.0", "port": 3000},
"adventurelog": {"fx": fx.FIXTURES["adventurelog"], "class": "PostgreSQL (PostGIS)",
"from": "ghcr.io/seanmorley15/adventurelog-backend:v0.12.1,ghcr.io/seanmorley15/adventurelog-frontend:v0.12.1",
"to": "ghcr.io/seanmorley15/adventurelog-backend:v0.13.0,ghcr.io/seanmorley15/adventurelog-frontend:v0.13.0", "port": 8000},
"romm": {"fx": fx.Romm(), "class": "MariaDB", "from": "rommapp/romm:5.3.0", "to": "rommapp/romm:5.3.1", "port": 8080},
"vikunja": {"fx": fx.Vikunja(), "class": "volume-only (SQLite + files in volumes)", "from": "vikunja/vikunja:2.5.0", "to": "vikunja/vikunja:2.6.0", "port": 3456},
"navidrome": {"fx": fx.Navidrome(), "class": "volume-only (SQLite) + a drive folder", "from": "deluan/navidrome:0.64.0", "to": "deluan/navidrome:0.64.1", "port": 4533},
"nextcloud": {"fx": fx.Nextcloud(), "class": "FILE-LEG (user files on the drive) + MariaDB", "from": "nextcloud:34.0.1-apache", "to": "nextcloud:34.0.4-apache", "port": 80},
}
EXTRA = {"actualbudget": fx.ActualBudget(), "papra": fx.Papra()}
SEED_B = {"docmost": (docmost_seed_b, docmost_verify_b), "romm": (romm_seed_b, romm_verify_b),
"vikunja": (vik_seed_b, vik_verify_b)}
# round -> the app it acts on (and, for two_updates, the second app)
TARGET = {1: "actualbudget", 2: "docmost", 3: "vikunja", 4: "adventurelog", 5: "papra", 6: "romm",
7: "docmost", 8: "navidrome", 9: "vikunja", 10: "actualbudget", 11: "nextcloud", 12: "papra"}
SECOND = {11: "romm"} # its database-engine step 11.4 -> 11.8, a tested good step (catalog 431cdec)
EXTRA_PINS = {"romm": ["mariadb:11.4"]}
KILL_LOG = os.path.join(HERE, "chaos-kills.json")
def load():
return json.load(open(STATE)) if os.path.exists(STATE) else {}
def save(s):
json.dump(s, open(STATE, "w"), indent=2, ensure_ascii=False)
def fixture(app):
return SIX[app]["fx"] if app in SIX else EXTRA[app]
def sub_of(app):
return fixture(app).sub
def hp(cmd, timeout=300):
return w.sh(["ssh", "-o", "ConnectTimeout=20", w.HP, "export LC_ALL=C; " + cmd], timeout=timeout)
def wait_controller(cap=900):
"""After a cut or a restart: the controller answers and knows its stacks again."""
t0 = time.time()
while time.time() - t0 < cap:
try:
w.login()
code, d = w.ctl("GET", "/api/stacks")
if code == "200":
return round(time.time() - t0, 1)
except SystemExit:
pass
time.sleep(5)
return None
# ---------------------------------------------------------------------------------------- accidents
def accident(kind, rnd, app, note):
t = time.strftime("%H:%M:%S")
if kind == "none":
return
w.say(f" >>> ACCIDENT {kind} at {t}")
if kind == "power_cut":
r1 = hp("pct stop 9202", 180); r2 = hp("pct start 9202", 180)
note["accident"] = f"pct stop rc={r1.returncode}, pct start rc={r2.returncode} at {t}"
elif kind == "controller_kill":
kills = json.load(open(KILL_LOG)) if os.path.exists(KILL_LOG) else []
if len(kills) >= 3 or (kills and time.time() - kills[-1] < 20 * 60):
note["accident"] = f"controller_kill SKIPPED — the budget rule (max 3, >= 20 min apart): {kills}"
w.say(" >>> " + note["accident"]); return
out = w.guest("p=$(docker inspect -f '{{.State.Pid}}' felhom-controller); kill -9 $p; echo killed $p")
kills.append(time.time()); json.dump(kills, open(KILL_LOG, "w"))
note["accident"] = f"kill -9 of the controller's PID: {out.strip()} at {t}"
elif kind == "docker_restart":
out = w.guest("systemctl restart docker; echo rc=$?", timeout=300)
note["accident"] = f"systemctl restart docker: {out.strip()} at {t}"
elif kind == "disk_fill":
out = w.guest("""set -e
fs=/var/lib/docker; avail=$(df -B1 --output=avail $fs | tail -1)
keep=$(( 3 * 1024 * 1024 * 1024 )) # the update floor is 2 GB; leave 1 GB above it
n=$(( avail - keep )); [ $n -gt 0 ] && fallocate -l $n $fs/chaos-fill
df -h $fs | tail -1""", timeout=120)
note["accident"] = f"filled /var/lib/docker to 3 GB free (floor 2 GB + 1 GB): {out.strip()} at {t}"
elif kind == "memory_hog":
out = w.guest("docker run -d --rm --name chaos-hog alpine sh -c 'head -c 1000m /dev/zero | tail & sleep 300; kill %1' ; sleep 20; docker stats --no-stream --format '{{.Name}} {{.MemUsage}}' chaos-hog")
note["accident"] = f"1 GB hog for 5 min: {out.strip()} at {t}"
elif kind == "box_backup":
code, d = w.ctl("POST", "/api/backup/run")
note["accident"] = f"POST /api/backup/run -> {code} {str(d)[:160]} at {t}"
elif kind == "two_updates":
second = SECOND.get(rnd)
code, d = w.ctl("POST", f"/api/stacks/{second}/update")
note["accident"] = f"a second Update on {second} -> {code} {str(d)[:200]} at {t}"
note["second_app"] = second
w.say(" >>> " + note.get("accident", ""))
def undo_accident(kind):
if kind == "disk_fill":
w.guest("rm -f /var/lib/docker/chaos-fill; df -h /var/lib/docker | tail -1")
if kind == "memory_hog":
w.guest("docker rm -f chaos-hog >/dev/null 2>&1; true")
# ---------------------------------------------------------------------------------------- observing
def events_since(t_iso):
"""R-620: a disabled notifier says what it DROPS — once per type at WARN, then at DEBUG."""
code, d = w.ctl("GET", "/api/debug/logs?level=DEBUG&lines=6000")
ents = ((d.get("data") or {}).get("entries") or []) if isinstance(d, dict) else []
out = []
for e in ents:
m = re.search(r"DROPPED event (\S+) \(severity (\w+)\)", e.get("message", ""), re.I) # WARN once, then DEBUG "dropped event"
if m and e.get("timestamp", "") >= t_iso:
out.append(f"{e['timestamp']} {m.group(1)} ({m.group(2)})")
return out
def box_log(t_iso, app):
code, d = w.ctl("GET", "/api/debug/logs?level=INFO&lines=4000")
ents = ((d.get("data") or {}).get("entries") or []) if isinstance(d, dict) else []
keys = (app, "update recovery", "UNDO", "undo", "held", "HOLD", "restore", "Restore", "deploy", "floor",
"disk", "space", "backup")
return [f"{e['timestamp']} {e.get('level')} {e.get('message','')[:300]}" for e in ents
if e.get("timestamp", "") >= t_iso and any(k in e.get("message", "") for k in keys)][-60:]
def household(app):
out = {"badges": w.badges(app), "lines": live.page_lines(app)}
st = w.stack(app)
out["state"] = {k: st.get(k) for k in ("deployed", "state", "update_phase", "hold_reason", "update_error")}
return out
def readback(app, s):
a = s.get("apps", {}).get(app, {})
if not a.get("A"):
return {"A": None, "note": "no seed A recorded"}
f = fixture(app)
w.wait_app(sub_of(app), "/", tries=60, delay=3)
res = {"A": bool(f.verify(w, sub_of(app), a["A"], w.say))}
if app in SEED_B and a.get("B"):
r = SEED_B[app][1](sub_of(app), a["A"], a["B"])
res["B"] = all(r.values()) if isinstance(r, dict) else bool(r)
return res
def steady(app, t0, cap=1500, want_deployed=True):
last = None
while time.time() - t0 < cap:
try:
st = w.stack(app)
except Exception:
st = {}
cur = (st.get("deployed"), st.get("state"), st.get("updating"), st.get("update_phase"), st.get("hold_reason"))
if cur != last:
w.say(f" +{time.time()-t0:6.1f}s {app}: deployed={cur[0]} state={cur[1]} updating={cur[2]} phase={cur[3]} hold={str(cur[4])[:80]!r}")
last = cur
if st and not st.get("updating") and not st.get("deploying"):
if (want_deployed and st.get("deployed") and st.get("state") in ("running", "stopped", "unhealthy", "degraded")) \
or (not want_deployed and not st.get("deployed")) or st.get("hold_reason"):
return round(time.time() - t0, 1), st
time.sleep(2)
return None, w.stack(app)
# ---------------------------------------------------------------------------------------- actions
def drill_template_at(app, frm_first):
for i in range(36):
if f"image: {frm_first}" in w.guest(f"cat /opt/docker/stacks/{app}/docker-compose.yml"):
return i
w.sync_rescan(); time.sleep(5)
return None
def drill_wait_ref(app, ref):
for i in range(36):
if ref in w.guest(f"cat /opt/docker/stacks/{app}/docker-compose.yml"):
return i
w.sync_rescan(); time.sleep(5)
return None
def act(action, app, rnd, note, s, arm=lambda: None):
"""Run the action. Returns when the app is steady (or the cap). The accident fires in a thread,
ARMED at the moment of the press itself (round 2 showed why: the catalog took ~7 min to reach the
box, and an accident timed from the round's start landed before the update ever began)."""
t0 = time.time()
if action == "install":
vals = w.deploy_values(app, sub_of(app))
arm()
code, d = w.ctl("POST", f"/api/stacks/{app}/deploy", {"values": vals})
note["act"] = f"deploy -> {code} {str(d)[:120]}"
secs, st = steady(app, t0)
if st.get("deployed"):
w.wait_app(sub_of(app), "/", tries=60, delay=3)
A = fixture(app).seed(w, sub_of(app), w.say)
s.setdefault("apps", {}).setdefault(app, {})["A"] = A
save(s)
return secs, st
if action == "remove":
arm()
code = w.remove(app)
note["act"] = f"remove -> {code}"
return steady(app, t0, want_deployed=False)
if action == "restore":
arm()
res = w.restore(app)
note["act"] = {k: res.get(k) for k in ("ok", "why", "snapshot_id", "http", "seconds", "state_after", "hold_after")}
return steady(app, t0)
frm, to, port = SIX[app]["from"], SIX[app]["to"], SIX[app]["port"]
if action == "good_update":
h = w.drill_bump(app, frm, to)
else: # fail_health / cutoff: the real image move + a probe port the app does not answer
h = break_edge(app, frm, to, port, 8999 if port != 8999 else 8998)
note["drill_commit"] = h
# For an INSTALLED app the stack dir's compose is the applied one, not the catalog's — waiting on
# it cost ~7 min in rounds 2 and 4. The badge's own input (`catalog_images`) is the right wait.
note["badge_catchup_s"] = w.sync_rescan(expect_app=app, expect_ref=to.split(",")[0], tries=60, delay=5)
on_phase = None
if action == "cutoff":
done = {}
def on_phase(ph):
if ph == "verifying" and not done:
out = w.guest(f"""c=$(docker volume ls -q --filter label=felhom.undo-copy-of={app} | head -1); echo "copy: $c"
docker run --rm -v $c:/c alpine sh -c 'rm -f /c/felhom-undo-complete; ls /c | head -5'""")
done["out"] = out
note["cutoff"] = out.strip()
w.say(" >>> finished-marker removed from one undo copy:\n" + out)
return False
arm()
phases, st = live.press(app, poll=0.5, on_phase=on_phase)
note["phases"] = phases
return steady(app, t0)
def repair_drill(action, app):
"""After a broken-edge round: the drill template goes back to the app's OLD image and its REAL
probe, so the app is not offered the broken step for the rest of the hour."""
if action not in ("fail_health", "cutoff"):
return
import fcntl
with open(f"{w.SC}/drill.lock", "w") as lk:
fcntl.flock(lk, fcntl.LOCK_EX)
w.sh(["git", "-C", w.DRILL, "pull", "-q", "--rebase", "origin", "main"], timeout=120)
comp = f"{w.DRILL}/templates/{app}/docker-compose.yml"
c = open(comp).read()
for f1, t1 in zip(SIX[app]["from"].split(","), SIX[app]["to"].split(",")):
c = c.replace(f"image: {t1}", f"image: {f1}")
open(comp, "w").write(c)
fy = f"{w.DRILL}/templates/{app}/.felhom.yml"
f = open(fy).read()
f = re.sub(r"(healthcheck:\n(?:.*\n){0,8}?\s+port: )899[89]\b", lambda m: m.group(1) + str(SIX[app]["port"]), f)
open(fy, "w").write(f)
w.sh(["git", "-C", w.DRILL, "commit", "-qam", f"DRILL chaos: {app} back to its old image and real probe"])
w.sh(["git", "-C", w.DRILL, "push", "-q", "origin", "main"], timeout=120)
# ---------------------------------------------------------------------------------------- stages
def setup():
import fcntl
s = load()
w.login()
with open(f"{w.SC}/drill.lock", "w") as lk:
fcntl.flock(lk, fcntl.LOCK_EX)
w.sh(["git", "-C", w.DRILL, "pull", "-q", "--rebase", "origin", "main"], timeout=120)
for app, a in SIX.items():
comp = f"{w.DRILL}/templates/{app}/docker-compose.yml"
c = open(comp).read()
for frm in a["from"].split(",") + EXTRA_PINS.get(app, []):
c = re.sub(r"image: " + re.escape(frm.rsplit(":", 1)[0]) + r":\S+", "image: " + frm, c, count=1)
open(comp, "w").write(c)
# and the REAL probe, whatever an earlier drill left there
fy = f"{w.DRILL}/templates/{app}/.felhom.yml"
f = open(fy).read()
f = re.sub(r"(healthcheck:\n(?:.*\n){0,8}?\s+port: )899[89]\b", lambda m: m.group(1) + str(a["port"]), f)
open(fy, "w").write(f)
w.sh(["git", "-C", w.DRILL, "commit", "-qam", "DRILL chaos setup: the six at their FROM pins, real probes"])
w.sh(["git", "-C", w.DRILL, "push", "-q", "origin", "main"], timeout=120)
for app, a in SIX.items():
w.say(f"== setup {app} ({a['class']})")
drill_template_at(app, a["from"].split(",")[0])
if not w.deploy(app, sub_of(app)):
w.say(f" {app}: deploy FAILED"); continue
w.wait_app(sub_of(app), "/", tries=90, delay=3)
A = a["fx"].seed(w, sub_of(app), w.say)
s.setdefault("apps", {}).setdefault(app, {})["A"] = A
save(s)
# the big database: 300 more spaces in docmost, through its own API
A = s["apps"]["docmost"]["A"]
n = 0
for i in range(300):
if docmost_seed_b(sub_of("docmost"), A):
n += 1
s["apps"]["docmost"]["big_extra_spaces"] = n
w.say(f" docmost: {n} extra spaces created (the big database)")
save(s)
w.say("== one whole-box backup on 9202 (the restore rounds need a copy)")
t0 = time.time(); code, d = w.ctl("POST", "/api/backup/run"); w.say(f" backup -> {code} {str(d)[:120]}")
time.sleep(20)
for _ in range(360):
c2, d2 = w.ctl("GET", "/api/backup/status")
dd = (d2.get("data") or {}) if isinstance(d2, dict) else {}
if not dd.get("running"):
break
time.sleep(5)
w.say(f" backup finished after {round(time.time()-t0)}s: {str(dd)[:200]}")
for app in SEED_B:
B = SEED_B[app][0](sub_of(app), s["apps"][app]["A"])
s["apps"][app]["B"] = B
w.say(f" seed B for {app} (after the backup): {bool(B)}")
save(s)
s["setup_readback"] = {app: readback(app, s) for app in SIX}
w.say("setup readback: " + json.dumps(s["setup_readback"]))
save(s)
open(os.path.join(EVD, "00-setup.log"), "w").write("\n".join(w.LOG) + "\n")
def run_round(n):
s = load()
sched = {r["round"]: r for r in draw()[0]}[n]
app = TARGET[n]
w.login()
t_iso = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
w.say(f"==== ROUND {n}: {sched['action']} on {app} x {sched['accident']} at +{sched['delay_s']}s ({t_iso})")
note = {"round": n, **sched, "app": app, "started": t_iso}
note["before"] = household(app) if w.stack(app).get("deployed") else {"state": "not deployed"}
if n in SECOND: # the second app gets its good step in the drill catalog before the round
note["second_drill_commit"] = w.drill_bump(SECOND[n], "mariadb:11.4", "mariadb:11.8")
note["second_badge_catchup_s"] = w.sync_rescan(expect_app=SECOND[n], expect_ref="mariadb:11.8", tries=60, delay=5)
t0 = time.time()
timer = threading.Timer(sched["delay_s"], accident, args=(sched["accident"], n, app, note))
def arm():
note["armed_at"] = time.strftime("%H:%M:%S")
timer.start()
try:
secs, st = act(sched["action"], app, n, note, s, arm)
except Exception as e:
note["act_exception"] = repr(e)
secs, st = None, {}
if timer.is_alive() or note.get("armed_at"):
timer.join()
else:
note["accident"] = "NOT FIRED — the action ended before it was armed"
if sched["accident"] in ("power_cut", "docker_restart", "controller_kill"):
note["controller_back_s"] = wait_controller()
secs2, st = steady(app, t0, want_deployed=(sched["action"] != "remove"))
secs = secs2 or secs
note["time_to_steady_s"] = round(time.time() - t0, 1) if secs is None else secs
undo_accident(sched["accident"])
if note.get("second_app"):
s2, st2 = steady(note["second_app"], t0)
note["second_app_end"] = {k: st2.get(k) for k in ("state", "update_phase", "hold_reason")}
note["second_app_readback"] = readback(note["second_app"], s)
note["after"] = household(app) if w.stack(app).get("deployed") else {"state": "not deployed",
"leftovers": w.guest(f"docker ps -a --filter label=com.docker.compose.project={app} --format '{{{{.Names}}}}'; docker volume ls -q | grep '^{app}_' || true").strip()}
note["observables"] = w.observables(app) if w.stack(app).get("deployed") else None
note["readback"] = readback(app, s) if w.stack(app).get("deployed") else None
note["events"] = events_since(t_iso)
note["box_log"] = box_log(t_iso, app)
note["undo_copies_left"] = w.guest(f"docker volume ls -q --filter label=felhom.undo-copy-of={app} | wc -l").strip()
# the WHOLE controller log of the round, off the machine now (R-320): round 7's power cut took the
# container's earlier log with it, and round 1's app_start_failed could no longer be tied to an app
open(os.path.join(EVD, f"round-{n:02d}-controller.log"), "w").write(
w.guest(f"docker logs --since {t_iso} felhom-controller 2>&1 | grep -vE 'auth: valid session|router.go:81' | tail -3000"))
repair_drill(sched["action"], app)
json.dump(note, open(os.path.join(EVD, f"round-{n:02d}.json"), "w"), indent=2, ensure_ascii=False)
open(os.path.join(EVD, f"round-{n:02d}.log"), "w").write("\n".join(w.LOG) + "\n")
w.say(f"==== ROUND {n} END: steady in {note['time_to_steady_s']}s; readback={note['readback']}; "
f"events={len(note['events'])}")
if __name__ == "__main__":
if sys.argv[1] == "setup":
setup()
elif sys.argv[1] == "round":
run_round(int(sys.argv[2]))
elif sys.argv[1] == "readback":
w.login(); s = load()
print(json.dumps({a: readback(a, s) for a in list(SIX) + list(EXTRA) if w.stack(a).get("deployed")}, indent=1))