test record: an image move must carry its proof (09 decision 13, part 4)
gates / gates (push) Successful in 1s
gates / gates (push) Successful in 1s
update_ladder: in .felhom.yml, one JSON entry per line (spiked live on controller v0.266.0 and v0.267.0 first). Two gates: check-test-record.py (static, CI too) and check-test-record-move.py (history + registry for moved refs only). 16 decoys, 3 red-proofs. The ONLY writer is upgrade-test.py --write-ladder (bench AND box proven, digests resolved). Harness v3: box fixtures on the bench, files_may_change. Backfill: the 21 moves of 2026-09-22, 21 proven from their records. No image: line moved. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,193 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""ladder.py — the test record (`update_ladder:`) in `.felhom.yml`: read it, check it, write it.
|
||||
|
||||
`09-update-architecture.md` §3 decision 13 — *the test decides, not the tag*: the catalog holds only
|
||||
tested steps, and an image move with no test record is refused at push time. §6.4 part 4 is the
|
||||
build; part 5 (the box climbing the ladder) and the digest half of part 6 on the box are NOT this.
|
||||
|
||||
THE FORMAT, chosen for the two readers it has (spiked live 2026-09-23 on controller v0.266.0 and
|
||||
v0.267.0 — the controller ignores the unknown top-level key and deploys, probes and badges as before):
|
||||
|
||||
update_ladder:
|
||||
- {"from": {...}, "to": {...}, "digest": {...}, "verdict": "proven", ...}
|
||||
|
||||
ONE JSON OBJECT PER LINE. JSON is a subset of YAML's flow style, so the controller's YAML parser
|
||||
reads it; and the catalog CI runner has NO PyYAML (it carries python3 and git only), so the gate
|
||||
reads it with `json.loads` — no parser that can be missing, no degraded mode. A line under
|
||||
`update_ladder:` that is not exactly ` - {json}` is a conviction, never a skip.
|
||||
|
||||
AN ENTRY (all keys required unless marked):
|
||||
from, to {service: image ref} for EVERY service with an image: line, before / after
|
||||
digest {service: "sha256:<64 hex>"} for every service in `to` — what the registry
|
||||
served for that ref when the entry was written (decision 17)
|
||||
verdict "proven" | "unrecorded" (backfill only: a live move with no record found)
|
||||
tested_at RFC 3339, or null for "unrecorded"
|
||||
harness_version int (2 = the memory watch), or null for "unrecorded"
|
||||
evidence path of the verdict record(s), relative to the workspace root
|
||||
memory_peak_pct the memory watch's worst container peak in % of its limit; null only on a
|
||||
backfilled entry (harness v1 had no watch)
|
||||
marks {"files_may_change": bool, "needs_person": null | "<why>", "memory_tight": bool}
|
||||
backfilled (optional) "YYYY-MM-DD" — written by the backfill from an EXISTING record,
|
||||
never by a new test; a new move may not carry it
|
||||
|
||||
Every path that reads or writes the format is here, so the gate and the writer cannot disagree.
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
|
||||
LADDER_KEY_RE = re.compile(r"^update_ladder:\s*$")
|
||||
ENTRY_RE = re.compile(r"^ - (\{.*\})\s*$")
|
||||
SERVICE_RE = re.compile(r"^ ([A-Za-z0-9_-]+):\s*$")
|
||||
IMAGE_RE = re.compile(r"^\s+image:\s*[\"']?([^\s\"'#]+)")
|
||||
DIGEST_RE = re.compile(r"^sha256:[0-9a-f]{64}$")
|
||||
TS_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$")
|
||||
DATE_RE = re.compile(r"^\d{4}-\d{2}-\d{2}$")
|
||||
VERDICTS = ("proven", "unrecorded")
|
||||
MEMORY_TIGHT_PCT = 80.0
|
||||
|
||||
|
||||
def images_in(compose_text):
|
||||
"""{service: image} — per service, from that service's OWN `image:` line (the same reading
|
||||
check-engine-major.py and check-catalog-since.py make)."""
|
||||
out, cur = {}, None
|
||||
for line in compose_text.splitlines():
|
||||
m = SERVICE_RE.match(line)
|
||||
if m:
|
||||
cur = m.group(1)
|
||||
continue
|
||||
mi = IMAGE_RE.match(line)
|
||||
if mi and cur and cur not in out:
|
||||
out[cur] = mi.group(1)
|
||||
return out
|
||||
|
||||
|
||||
def parse(felhom_text):
|
||||
"""(entries, raw_lines, errors). No ladder → ([], [], []). A malformed line is an ERROR."""
|
||||
lines = felhom_text.splitlines()
|
||||
start = None
|
||||
for i, l in enumerate(lines):
|
||||
if LADDER_KEY_RE.match(l):
|
||||
if start is not None:
|
||||
return [], [], ["update_ladder: appears twice"]
|
||||
start = i
|
||||
if start is None:
|
||||
return [], [], []
|
||||
entries, raws, errors = [], [], []
|
||||
for l in lines[start + 1:]:
|
||||
if not l.strip() or l.lstrip().startswith("#"):
|
||||
continue
|
||||
if not l.startswith(" "):
|
||||
break # the next top-level key: the block has ended
|
||||
m = ENTRY_RE.match(l)
|
||||
if not m:
|
||||
errors.append("not a one-line JSON entry under update_ladder: %r" % l[:120])
|
||||
continue
|
||||
try:
|
||||
e = json.loads(m.group(1))
|
||||
except ValueError as ex:
|
||||
errors.append("entry is not valid JSON (%s): %r" % (ex, l[:120]))
|
||||
continue
|
||||
if not isinstance(e, dict):
|
||||
errors.append("entry is not an object: %r" % l[:120])
|
||||
continue
|
||||
entries.append(e)
|
||||
raws.append(m.group(1))
|
||||
if not entries and not errors:
|
||||
errors.append("update_ladder: is present but holds no entry")
|
||||
return entries, raws, errors
|
||||
|
||||
|
||||
def check_entry(e):
|
||||
"""Problems with ONE entry's shape, as sentences. Empty list = well-formed."""
|
||||
p = []
|
||||
for k in ("from", "to", "digest", "verdict", "tested_at", "harness_version", "evidence",
|
||||
"memory_peak_pct", "marks"):
|
||||
if k not in e:
|
||||
p.append("missing key %r" % k)
|
||||
if p:
|
||||
return p
|
||||
for k in ("from", "to", "digest"):
|
||||
if not isinstance(e[k], dict) or not e[k]:
|
||||
p.append("%r must be a non-empty {service: value} object" % k)
|
||||
if p:
|
||||
return p
|
||||
v = e["verdict"]
|
||||
if v not in VERDICTS:
|
||||
p.append("verdict %r is not allowed in a ladder (only %s — a failed or inconclusive test is "
|
||||
"evidence, never a step a box may take)" % (v, "/".join(VERDICTS)))
|
||||
backfilled = e.get("backfilled")
|
||||
if backfilled is not None and not (isinstance(backfilled, str) and DATE_RE.match(backfilled)):
|
||||
p.append("backfilled must be a YYYY-MM-DD date")
|
||||
if v == "unrecorded" and backfilled is None:
|
||||
p.append("verdict 'unrecorded' exists only for the backfill of a move made before the gate")
|
||||
for svc, ref in e["to"].items():
|
||||
d = e["digest"].get(svc)
|
||||
if not (isinstance(d, str) and DIGEST_RE.match(d)):
|
||||
p.append("no sha256 digest for service %r (%s)" % (svc, ref))
|
||||
for svc in e["digest"]:
|
||||
if svc not in e["to"]:
|
||||
p.append("digest names a service %r that `to` does not" % svc)
|
||||
marks = e["marks"]
|
||||
if not isinstance(marks, dict) or set(marks) != {"files_may_change", "needs_person", "memory_tight"}:
|
||||
p.append("marks must be exactly {files_may_change, needs_person, memory_tight}")
|
||||
marks = {}
|
||||
if v == "proven":
|
||||
if not (isinstance(e["tested_at"], str) and TS_RE.match(e["tested_at"])):
|
||||
p.append("a proven entry needs tested_at as an RFC 3339 time")
|
||||
if not (isinstance(e["evidence"], str) and e["evidence"].strip()):
|
||||
p.append("a proven entry must cite its evidence")
|
||||
peak = e["memory_peak_pct"]
|
||||
if backfilled is None:
|
||||
if not isinstance(e["harness_version"], int) or e["harness_version"] < 2:
|
||||
p.append("a new proven entry needs harness_version >= 2 (the memory watch)")
|
||||
if not isinstance(peak, (int, float)) or isinstance(peak, bool):
|
||||
p.append("a new proven entry needs memory_peak_pct from the memory watch")
|
||||
if isinstance(peak, (int, float)) and not isinstance(peak, bool) and marks:
|
||||
tight = peak > MEMORY_TIGHT_PCT
|
||||
if bool(marks.get("memory_tight")) != tight:
|
||||
p.append("marks.memory_tight=%s disagrees with memory_peak_pct=%s (tight above %d%%)"
|
||||
% (marks.get("memory_tight"), peak, MEMORY_TIGHT_PCT))
|
||||
return p
|
||||
|
||||
|
||||
def entry_line(e):
|
||||
"""The one line the writer emits for an entry — key order fixed so diffs stay readable."""
|
||||
order = ["from", "to", "digest", "verdict", "tested_at", "harness_version", "evidence",
|
||||
"box_evidence", "memory_peak_pct", "marks", "backfilled", "note"]
|
||||
ordered = {k: e[k] for k in order if k in e}
|
||||
for k in e:
|
||||
if k not in ordered:
|
||||
ordered[k] = e[k]
|
||||
return " - " + json.dumps(ordered, ensure_ascii=False, separators=(", ", ": "))
|
||||
|
||||
|
||||
LADDER_HEADER = (
|
||||
"\n# update_ladder — the test record: one tested step per line, oldest first (JSON flow mappings,\n"
|
||||
"# `09-update-architecture.md` §6.4 part 4). WRITTEN BY scripts/upgrade-test.py, never by hand;\n"
|
||||
"# gated by scripts/check-test-record.py. An image: move without a proven entry here is refused.\n"
|
||||
"update_ladder:\n")
|
||||
|
||||
|
||||
def append_entry(felhom_text, e):
|
||||
"""Return felhom_text with `e` appended as the ladder's newest line (creating the block at the
|
||||
END of the file when absent — it is a top-level key and nothing may follow it inside the block)."""
|
||||
line = entry_line(e)
|
||||
lines = felhom_text.splitlines()
|
||||
start = None
|
||||
for i, l in enumerate(lines):
|
||||
if LADDER_KEY_RE.match(l):
|
||||
start = i
|
||||
if start is None:
|
||||
body = felhom_text.rstrip("\n") + "\n" + LADDER_HEADER + line + "\n"
|
||||
return body
|
||||
end = start + 1
|
||||
while end < len(lines) and (not lines[end].strip() or lines[end].startswith(" ")
|
||||
or lines[end].lstrip().startswith("#")):
|
||||
end += 1
|
||||
# insert after the last entry line of the block
|
||||
last = start
|
||||
for j in range(start + 1, end):
|
||||
if ENTRY_RE.match(lines[j]):
|
||||
last = j
|
||||
lines.insert(last + 1, line)
|
||||
return "\n".join(lines) + "\n"
|
||||
Reference in New Issue
Block a user