3e58c184f6
gates / gates (push) Successful in 27s
DRILL-night-2026-09-23.md: Parts A-E. 09 §3 decisions 21 (operator word), 22 and 23 (CC unattended, operator may reverse); §6.4 parts 4 and 6 (catalog half) shipped; §6.1a residuals R-658/R-659. Register 330 -> 336: R-651..R-660 opened (R-658 and R-659 P1), R-650/R-640/R-499/R-626 closed. Capability map, nightly rotation (opengist), STATUS (one question: the floor), CONTEXT, REPORT. The floor stays 0.266.0. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
414 lines
20 KiB
Python
414 lines
20 KiB
Python
#!/usr/bin/env python3
|
|
"""chaos.py — Part D, the chaos hour on the undo. Guest 9202, controller v0.267.0, drill catalog.
|
|
|
|
python3 chaos.py setup # six apps at their FROM pins, seeded (A), one whole-box backup, seed B
|
|
python3 chaos.py round <N> # one round of chaos_schedule.py, evidence written as it happens
|
|
python3 chaos.py readback # every app's seeds, read through its own front door
|
|
|
|
EVIDENCE, NOT PRODUCT. Every act is the endpoint the UI invokes; every accident is something a real
|
|
box suffers (a power cut, a killed process, a full disk, a memory hog, a backup, a second press).
|
|
The app a round acts on is fixed by the table below, NOT drawn: an app's state after round N decides
|
|
what round N+1 can do to it, so the pairing is written here and in the findings doc before round 1.
|
|
|
|
The five things recorded every round: what the household saw (the app page in BOTH languages), what
|
|
the box did by itself (phases + the controller's own log lines), time to steady, which events fired
|
|
(9202 has no hub, so R-620's "DROPPED event" lines are the witness), and — judged in the findings doc
|
|
against `08-alarm-ladder.md` — what should have fired and did not. Plus the data read back.
|
|
"""
|
|
import json, os, re, subprocess, sys, threading, time
|
|
sys.path.insert(0, ".")
|
|
import walk as w
|
|
import fixtures as fx
|
|
import live
|
|
from spike import docmost_seed_b, docmost_verify_b, romm_seed_b, romm_verify_b, vik_seed_b, vik_verify_b, \
|
|
break_edge
|
|
from chaos_schedule import draw
|
|
|
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
STATE = os.path.join(HERE, "chaos-state.json")
|
|
EVD = os.path.join(HERE, "chaos")
|
|
os.makedirs(EVD, exist_ok=True)
|
|
|
|
# the six (brief Part D): two PostgreSQL, one MariaDB, one volume-only, one file-leg, one big database
|
|
SIX = {
|
|
"docmost": {"fx": fx.Docmost(), "class": "PostgreSQL + the BIG database (300 extra spaces)", "from": "docmost/docmost:0.95.0", "to": "docmost/docmost:0.96.0", "port": 3000},
|
|
"adventurelog": {"fx": fx.FIXTURES["adventurelog"], "class": "PostgreSQL (PostGIS)",
|
|
"from": "ghcr.io/seanmorley15/adventurelog-backend:v0.12.1,ghcr.io/seanmorley15/adventurelog-frontend:v0.12.1",
|
|
"to": "ghcr.io/seanmorley15/adventurelog-backend:v0.13.0,ghcr.io/seanmorley15/adventurelog-frontend:v0.13.0", "port": 8000},
|
|
"romm": {"fx": fx.Romm(), "class": "MariaDB", "from": "rommapp/romm:5.3.0", "to": "rommapp/romm:5.3.1", "port": 8080},
|
|
"vikunja": {"fx": fx.Vikunja(), "class": "volume-only (SQLite + files in volumes)", "from": "vikunja/vikunja:2.5.0", "to": "vikunja/vikunja:2.6.0", "port": 3456},
|
|
"navidrome": {"fx": fx.Navidrome(), "class": "volume-only (SQLite) + a drive folder", "from": "deluan/navidrome:0.64.0", "to": "deluan/navidrome:0.64.1", "port": 4533},
|
|
"nextcloud": {"fx": fx.Nextcloud(), "class": "FILE-LEG (user files on the drive) + MariaDB", "from": "nextcloud:34.0.1-apache", "to": "nextcloud:34.0.4-apache", "port": 80},
|
|
}
|
|
EXTRA = {"actualbudget": fx.ActualBudget(), "papra": fx.Papra()}
|
|
SEED_B = {"docmost": (docmost_seed_b, docmost_verify_b), "romm": (romm_seed_b, romm_verify_b),
|
|
"vikunja": (vik_seed_b, vik_verify_b)}
|
|
# round -> the app it acts on (and, for two_updates, the second app)
|
|
TARGET = {1: "actualbudget", 2: "docmost", 3: "vikunja", 4: "adventurelog", 5: "papra", 6: "romm",
|
|
7: "docmost", 8: "navidrome", 9: "vikunja", 10: "actualbudget", 11: "nextcloud", 12: "papra"}
|
|
SECOND = {11: "romm"} # its database-engine step 11.4 -> 11.8, a tested good step (catalog 431cdec)
|
|
EXTRA_PINS = {"romm": ["mariadb:11.4"]}
|
|
KILL_LOG = os.path.join(HERE, "chaos-kills.json")
|
|
|
|
|
|
def load():
|
|
return json.load(open(STATE)) if os.path.exists(STATE) else {}
|
|
|
|
|
|
def save(s):
|
|
json.dump(s, open(STATE, "w"), indent=2, ensure_ascii=False)
|
|
|
|
|
|
def fixture(app):
|
|
return SIX[app]["fx"] if app in SIX else EXTRA[app]
|
|
|
|
|
|
def sub_of(app):
|
|
return fixture(app).sub
|
|
|
|
|
|
def hp(cmd, timeout=300):
|
|
return w.sh(["ssh", "-o", "ConnectTimeout=20", w.HP, "export LC_ALL=C; " + cmd], timeout=timeout)
|
|
|
|
|
|
def wait_controller(cap=900):
|
|
"""After a cut or a restart: the controller answers and knows its stacks again."""
|
|
t0 = time.time()
|
|
while time.time() - t0 < cap:
|
|
try:
|
|
w.login()
|
|
code, d = w.ctl("GET", "/api/stacks")
|
|
if code == "200":
|
|
return round(time.time() - t0, 1)
|
|
except SystemExit:
|
|
pass
|
|
time.sleep(5)
|
|
return None
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------- accidents
|
|
def accident(kind, rnd, app, note):
|
|
t = time.strftime("%H:%M:%S")
|
|
if kind == "none":
|
|
return
|
|
w.say(f" >>> ACCIDENT {kind} at {t}")
|
|
if kind == "power_cut":
|
|
r1 = hp("pct stop 9202", 180); r2 = hp("pct start 9202", 180)
|
|
note["accident"] = f"pct stop rc={r1.returncode}, pct start rc={r2.returncode} at {t}"
|
|
elif kind == "controller_kill":
|
|
kills = json.load(open(KILL_LOG)) if os.path.exists(KILL_LOG) else []
|
|
if len(kills) >= 3 or (kills and time.time() - kills[-1] < 20 * 60):
|
|
note["accident"] = f"controller_kill SKIPPED — the budget rule (max 3, >= 20 min apart): {kills}"
|
|
w.say(" >>> " + note["accident"]); return
|
|
out = w.guest("p=$(docker inspect -f '{{.State.Pid}}' felhom-controller); kill -9 $p; echo killed $p")
|
|
kills.append(time.time()); json.dump(kills, open(KILL_LOG, "w"))
|
|
note["accident"] = f"kill -9 of the controller's PID: {out.strip()} at {t}"
|
|
elif kind == "docker_restart":
|
|
out = w.guest("systemctl restart docker; echo rc=$?", timeout=300)
|
|
note["accident"] = f"systemctl restart docker: {out.strip()} at {t}"
|
|
elif kind == "disk_fill":
|
|
out = w.guest("""set -e
|
|
fs=/var/lib/docker; avail=$(df -B1 --output=avail $fs | tail -1)
|
|
keep=$(( 3 * 1024 * 1024 * 1024 )) # the update floor is 2 GB; leave 1 GB above it
|
|
n=$(( avail - keep )); [ $n -gt 0 ] && fallocate -l $n $fs/chaos-fill
|
|
df -h $fs | tail -1""", timeout=120)
|
|
note["accident"] = f"filled /var/lib/docker to 3 GB free (floor 2 GB + 1 GB): {out.strip()} at {t}"
|
|
elif kind == "memory_hog":
|
|
out = w.guest("docker run -d --rm --name chaos-hog alpine sh -c 'head -c 1000m /dev/zero | tail & sleep 300; kill %1' ; sleep 20; docker stats --no-stream --format '{{.Name}} {{.MemUsage}}' chaos-hog")
|
|
note["accident"] = f"1 GB hog for 5 min: {out.strip()} at {t}"
|
|
elif kind == "box_backup":
|
|
code, d = w.ctl("POST", "/api/backup/run")
|
|
note["accident"] = f"POST /api/backup/run -> {code} {str(d)[:160]} at {t}"
|
|
elif kind == "two_updates":
|
|
second = SECOND.get(rnd)
|
|
code, d = w.ctl("POST", f"/api/stacks/{second}/update")
|
|
note["accident"] = f"a second Update on {second} -> {code} {str(d)[:200]} at {t}"
|
|
note["second_app"] = second
|
|
w.say(" >>> " + note.get("accident", ""))
|
|
|
|
|
|
def undo_accident(kind):
|
|
if kind == "disk_fill":
|
|
w.guest("rm -f /var/lib/docker/chaos-fill; df -h /var/lib/docker | tail -1")
|
|
if kind == "memory_hog":
|
|
w.guest("docker rm -f chaos-hog >/dev/null 2>&1; true")
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------- observing
|
|
def events_since(t_iso):
|
|
"""R-620: a disabled notifier says what it DROPS — once per type at WARN, then at DEBUG."""
|
|
code, d = w.ctl("GET", "/api/debug/logs?level=DEBUG&lines=6000")
|
|
ents = ((d.get("data") or {}).get("entries") or []) if isinstance(d, dict) else []
|
|
out = []
|
|
for e in ents:
|
|
m = re.search(r"DROPPED event (\S+) \(severity (\w+)\)", e.get("message", ""), re.I) # WARN once, then DEBUG "dropped event"
|
|
if m and e.get("timestamp", "") >= t_iso:
|
|
out.append(f"{e['timestamp']} {m.group(1)} ({m.group(2)})")
|
|
return out
|
|
|
|
|
|
def box_log(t_iso, app):
|
|
code, d = w.ctl("GET", "/api/debug/logs?level=INFO&lines=4000")
|
|
ents = ((d.get("data") or {}).get("entries") or []) if isinstance(d, dict) else []
|
|
keys = (app, "update recovery", "UNDO", "undo", "held", "HOLD", "restore", "Restore", "deploy", "floor",
|
|
"disk", "space", "backup")
|
|
return [f"{e['timestamp']} {e.get('level')} {e.get('message','')[:300]}" for e in ents
|
|
if e.get("timestamp", "") >= t_iso and any(k in e.get("message", "") for k in keys)][-60:]
|
|
|
|
|
|
def household(app):
|
|
out = {"badges": w.badges(app), "lines": live.page_lines(app)}
|
|
st = w.stack(app)
|
|
out["state"] = {k: st.get(k) for k in ("deployed", "state", "update_phase", "hold_reason", "update_error")}
|
|
return out
|
|
|
|
|
|
def readback(app, s):
|
|
a = s.get("apps", {}).get(app, {})
|
|
if not a.get("A"):
|
|
return {"A": None, "note": "no seed A recorded"}
|
|
f = fixture(app)
|
|
w.wait_app(sub_of(app), "/", tries=60, delay=3)
|
|
res = {"A": bool(f.verify(w, sub_of(app), a["A"], w.say))}
|
|
if app in SEED_B and a.get("B"):
|
|
r = SEED_B[app][1](sub_of(app), a["A"], a["B"])
|
|
res["B"] = all(r.values()) if isinstance(r, dict) else bool(r)
|
|
return res
|
|
|
|
|
|
def steady(app, t0, cap=1500, want_deployed=True):
|
|
last = None
|
|
while time.time() - t0 < cap:
|
|
try:
|
|
st = w.stack(app)
|
|
except Exception:
|
|
st = {}
|
|
cur = (st.get("deployed"), st.get("state"), st.get("updating"), st.get("update_phase"), st.get("hold_reason"))
|
|
if cur != last:
|
|
w.say(f" +{time.time()-t0:6.1f}s {app}: deployed={cur[0]} state={cur[1]} updating={cur[2]} phase={cur[3]} hold={str(cur[4])[:80]!r}")
|
|
last = cur
|
|
if st and not st.get("updating") and not st.get("deploying"):
|
|
if (want_deployed and st.get("deployed") and st.get("state") in ("running", "stopped", "unhealthy", "degraded")) \
|
|
or (not want_deployed and not st.get("deployed")) or st.get("hold_reason"):
|
|
return round(time.time() - t0, 1), st
|
|
time.sleep(2)
|
|
return None, w.stack(app)
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------- actions
|
|
def drill_template_at(app, frm_first):
|
|
for i in range(36):
|
|
if f"image: {frm_first}" in w.guest(f"cat /opt/docker/stacks/{app}/docker-compose.yml"):
|
|
return i
|
|
w.sync_rescan(); time.sleep(5)
|
|
return None
|
|
|
|
|
|
def drill_wait_ref(app, ref):
|
|
for i in range(36):
|
|
if ref in w.guest(f"cat /opt/docker/stacks/{app}/docker-compose.yml"):
|
|
return i
|
|
w.sync_rescan(); time.sleep(5)
|
|
return None
|
|
|
|
|
|
def act(action, app, rnd, note, s, arm=lambda: None):
|
|
"""Run the action. Returns when the app is steady (or the cap). The accident fires in a thread,
|
|
ARMED at the moment of the press itself (round 2 showed why: the catalog took ~7 min to reach the
|
|
box, and an accident timed from the round's start landed before the update ever began)."""
|
|
t0 = time.time()
|
|
if action == "install":
|
|
vals = w.deploy_values(app, sub_of(app))
|
|
arm()
|
|
code, d = w.ctl("POST", f"/api/stacks/{app}/deploy", {"values": vals})
|
|
note["act"] = f"deploy -> {code} {str(d)[:120]}"
|
|
secs, st = steady(app, t0)
|
|
if st.get("deployed"):
|
|
w.wait_app(sub_of(app), "/", tries=60, delay=3)
|
|
A = fixture(app).seed(w, sub_of(app), w.say)
|
|
s.setdefault("apps", {}).setdefault(app, {})["A"] = A
|
|
save(s)
|
|
return secs, st
|
|
if action == "remove":
|
|
arm()
|
|
code = w.remove(app)
|
|
note["act"] = f"remove -> {code}"
|
|
return steady(app, t0, want_deployed=False)
|
|
if action == "restore":
|
|
arm()
|
|
res = w.restore(app)
|
|
note["act"] = {k: res.get(k) for k in ("ok", "why", "snapshot_id", "http", "seconds", "state_after", "hold_after")}
|
|
return steady(app, t0)
|
|
frm, to, port = SIX[app]["from"], SIX[app]["to"], SIX[app]["port"]
|
|
if action == "good_update":
|
|
h = w.drill_bump(app, frm, to)
|
|
else: # fail_health / cutoff: the real image move + a probe port the app does not answer
|
|
h = break_edge(app, frm, to, port, 8999 if port != 8999 else 8998)
|
|
note["drill_commit"] = h
|
|
# For an INSTALLED app the stack dir's compose is the applied one, not the catalog's — waiting on
|
|
# it cost ~7 min in rounds 2 and 4. The badge's own input (`catalog_images`) is the right wait.
|
|
note["badge_catchup_s"] = w.sync_rescan(expect_app=app, expect_ref=to.split(",")[0], tries=60, delay=5)
|
|
on_phase = None
|
|
if action == "cutoff":
|
|
done = {}
|
|
|
|
def on_phase(ph):
|
|
if ph == "verifying" and not done:
|
|
out = w.guest(f"""c=$(docker volume ls -q --filter label=felhom.undo-copy-of={app} | head -1); echo "copy: $c"
|
|
docker run --rm -v $c:/c alpine sh -c 'rm -f /c/felhom-undo-complete; ls /c | head -5'""")
|
|
done["out"] = out
|
|
note["cutoff"] = out.strip()
|
|
w.say(" >>> finished-marker removed from one undo copy:\n" + out)
|
|
return False
|
|
arm()
|
|
phases, st = live.press(app, poll=0.5, on_phase=on_phase)
|
|
note["phases"] = phases
|
|
return steady(app, t0)
|
|
|
|
|
|
def repair_drill(action, app):
|
|
"""After a broken-edge round: the drill template goes back to the app's OLD image and its REAL
|
|
probe, so the app is not offered the broken step for the rest of the hour."""
|
|
if action not in ("fail_health", "cutoff"):
|
|
return
|
|
import fcntl
|
|
with open(f"{w.SC}/drill.lock", "w") as lk:
|
|
fcntl.flock(lk, fcntl.LOCK_EX)
|
|
w.sh(["git", "-C", w.DRILL, "pull", "-q", "--rebase", "origin", "main"], timeout=120)
|
|
comp = f"{w.DRILL}/templates/{app}/docker-compose.yml"
|
|
c = open(comp).read()
|
|
for f1, t1 in zip(SIX[app]["from"].split(","), SIX[app]["to"].split(",")):
|
|
c = c.replace(f"image: {t1}", f"image: {f1}")
|
|
open(comp, "w").write(c)
|
|
fy = f"{w.DRILL}/templates/{app}/.felhom.yml"
|
|
f = open(fy).read()
|
|
f = re.sub(r"(healthcheck:\n(?:.*\n){0,8}?\s+port: )899[89]\b", lambda m: m.group(1) + str(SIX[app]["port"]), f)
|
|
open(fy, "w").write(f)
|
|
w.sh(["git", "-C", w.DRILL, "commit", "-qam", f"DRILL chaos: {app} back to its old image and real probe"])
|
|
w.sh(["git", "-C", w.DRILL, "push", "-q", "origin", "main"], timeout=120)
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------- stages
|
|
def setup():
|
|
import fcntl
|
|
s = load()
|
|
w.login()
|
|
with open(f"{w.SC}/drill.lock", "w") as lk:
|
|
fcntl.flock(lk, fcntl.LOCK_EX)
|
|
w.sh(["git", "-C", w.DRILL, "pull", "-q", "--rebase", "origin", "main"], timeout=120)
|
|
for app, a in SIX.items():
|
|
comp = f"{w.DRILL}/templates/{app}/docker-compose.yml"
|
|
c = open(comp).read()
|
|
for frm in a["from"].split(",") + EXTRA_PINS.get(app, []):
|
|
c = re.sub(r"image: " + re.escape(frm.rsplit(":", 1)[0]) + r":\S+", "image: " + frm, c, count=1)
|
|
open(comp, "w").write(c)
|
|
# and the REAL probe, whatever an earlier drill left there
|
|
fy = f"{w.DRILL}/templates/{app}/.felhom.yml"
|
|
f = open(fy).read()
|
|
f = re.sub(r"(healthcheck:\n(?:.*\n){0,8}?\s+port: )899[89]\b", lambda m: m.group(1) + str(a["port"]), f)
|
|
open(fy, "w").write(f)
|
|
w.sh(["git", "-C", w.DRILL, "commit", "-qam", "DRILL chaos setup: the six at their FROM pins, real probes"])
|
|
w.sh(["git", "-C", w.DRILL, "push", "-q", "origin", "main"], timeout=120)
|
|
for app, a in SIX.items():
|
|
w.say(f"== setup {app} ({a['class']})")
|
|
drill_template_at(app, a["from"].split(",")[0])
|
|
if not w.deploy(app, sub_of(app)):
|
|
w.say(f" {app}: deploy FAILED"); continue
|
|
w.wait_app(sub_of(app), "/", tries=90, delay=3)
|
|
A = a["fx"].seed(w, sub_of(app), w.say)
|
|
s.setdefault("apps", {}).setdefault(app, {})["A"] = A
|
|
save(s)
|
|
# the big database: 300 more spaces in docmost, through its own API
|
|
A = s["apps"]["docmost"]["A"]
|
|
n = 0
|
|
for i in range(300):
|
|
if docmost_seed_b(sub_of("docmost"), A):
|
|
n += 1
|
|
s["apps"]["docmost"]["big_extra_spaces"] = n
|
|
w.say(f" docmost: {n} extra spaces created (the big database)")
|
|
save(s)
|
|
w.say("== one whole-box backup on 9202 (the restore rounds need a copy)")
|
|
t0 = time.time(); code, d = w.ctl("POST", "/api/backup/run"); w.say(f" backup -> {code} {str(d)[:120]}")
|
|
time.sleep(20)
|
|
for _ in range(360):
|
|
c2, d2 = w.ctl("GET", "/api/backup/status")
|
|
dd = (d2.get("data") or {}) if isinstance(d2, dict) else {}
|
|
if not dd.get("running"):
|
|
break
|
|
time.sleep(5)
|
|
w.say(f" backup finished after {round(time.time()-t0)}s: {str(dd)[:200]}")
|
|
for app in SEED_B:
|
|
B = SEED_B[app][0](sub_of(app), s["apps"][app]["A"])
|
|
s["apps"][app]["B"] = B
|
|
w.say(f" seed B for {app} (after the backup): {bool(B)}")
|
|
save(s)
|
|
s["setup_readback"] = {app: readback(app, s) for app in SIX}
|
|
w.say("setup readback: " + json.dumps(s["setup_readback"]))
|
|
save(s)
|
|
open(os.path.join(EVD, "00-setup.log"), "w").write("\n".join(w.LOG) + "\n")
|
|
|
|
|
|
def run_round(n):
|
|
s = load()
|
|
sched = {r["round"]: r for r in draw()[0]}[n]
|
|
app = TARGET[n]
|
|
w.login()
|
|
t_iso = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
w.say(f"==== ROUND {n}: {sched['action']} on {app} x {sched['accident']} at +{sched['delay_s']}s ({t_iso})")
|
|
note = {"round": n, **sched, "app": app, "started": t_iso}
|
|
note["before"] = household(app) if w.stack(app).get("deployed") else {"state": "not deployed"}
|
|
if n in SECOND: # the second app gets its good step in the drill catalog before the round
|
|
note["second_drill_commit"] = w.drill_bump(SECOND[n], "mariadb:11.4", "mariadb:11.8")
|
|
note["second_badge_catchup_s"] = w.sync_rescan(expect_app=SECOND[n], expect_ref="mariadb:11.8", tries=60, delay=5)
|
|
t0 = time.time()
|
|
timer = threading.Timer(sched["delay_s"], accident, args=(sched["accident"], n, app, note))
|
|
|
|
def arm():
|
|
note["armed_at"] = time.strftime("%H:%M:%S")
|
|
timer.start()
|
|
try:
|
|
secs, st = act(sched["action"], app, n, note, s, arm)
|
|
except Exception as e:
|
|
note["act_exception"] = repr(e)
|
|
secs, st = None, {}
|
|
if timer.is_alive() or note.get("armed_at"):
|
|
timer.join()
|
|
else:
|
|
note["accident"] = "NOT FIRED — the action ended before it was armed"
|
|
if sched["accident"] in ("power_cut", "docker_restart", "controller_kill"):
|
|
note["controller_back_s"] = wait_controller()
|
|
secs2, st = steady(app, t0, want_deployed=(sched["action"] != "remove"))
|
|
secs = secs2 or secs
|
|
note["time_to_steady_s"] = round(time.time() - t0, 1) if secs is None else secs
|
|
undo_accident(sched["accident"])
|
|
if note.get("second_app"):
|
|
s2, st2 = steady(note["second_app"], t0)
|
|
note["second_app_end"] = {k: st2.get(k) for k in ("state", "update_phase", "hold_reason")}
|
|
note["second_app_readback"] = readback(note["second_app"], s)
|
|
note["after"] = household(app) if w.stack(app).get("deployed") else {"state": "not deployed",
|
|
"leftovers": w.guest(f"docker ps -a --filter label=com.docker.compose.project={app} --format '{{{{.Names}}}}'; docker volume ls -q | grep '^{app}_' || true").strip()}
|
|
note["observables"] = w.observables(app) if w.stack(app).get("deployed") else None
|
|
note["readback"] = readback(app, s) if w.stack(app).get("deployed") else None
|
|
note["events"] = events_since(t_iso)
|
|
note["box_log"] = box_log(t_iso, app)
|
|
note["undo_copies_left"] = w.guest(f"docker volume ls -q --filter label=felhom.undo-copy-of={app} | wc -l").strip()
|
|
# the WHOLE controller log of the round, off the machine now (R-320): round 7's power cut took the
|
|
# container's earlier log with it, and round 1's app_start_failed could no longer be tied to an app
|
|
open(os.path.join(EVD, f"round-{n:02d}-controller.log"), "w").write(
|
|
w.guest(f"docker logs --since {t_iso} felhom-controller 2>&1 | grep -vE 'auth: valid session|router.go:81' | tail -3000"))
|
|
repair_drill(sched["action"], app)
|
|
json.dump(note, open(os.path.join(EVD, f"round-{n:02d}.json"), "w"), indent=2, ensure_ascii=False)
|
|
open(os.path.join(EVD, f"round-{n:02d}.log"), "w").write("\n".join(w.LOG) + "\n")
|
|
w.say(f"==== ROUND {n} END: steady in {note['time_to_steady_s']}s; readback={note['readback']}; "
|
|
f"events={len(note['events'])}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
if sys.argv[1] == "setup":
|
|
setup()
|
|
elif sys.argv[1] == "round":
|
|
run_round(int(sys.argv[2]))
|
|
elif sys.argv[1] == "readback":
|
|
w.login(); s = load()
|
|
print(json.dumps({a: readback(a, s) for a in list(SIX) + list(EXTRA) if w.stack(a).get("deployed")}, indent=1))
|