#!/usr/bin/env python3 # -*- coding: utf-8 -*- """register_table.py — read OPEN-ITEMS.md's rows the way a reader sees them (2026-10-03). A helper, not a gate: `register_shape_gate.py` and `closed_register_gate.py` both import it, so the two gates cannot disagree about where a row's cells are. WHY A HELPER EXISTS. Both gates used to split a row on every `|`. `register_shape_gate.py`'s own docstring records why that cannot count cells: register prose carries literal pipes. Measured on 2026-10-03 against the register at `9305288` (442 rows): **every literal pipe but five sat inside inline code** (`--format '{{.Names}}|{{.Image}}'`, `owner: CC | …` inside backticks). A split that skips pipes inside backticks and escaped `\\|` gave the declared column count for 437 rows; the five it did not were genuine shape defects (a stray extra cell, two rows written five columns wide), and they were repaired in the same commit. So the cell count CAN be computed after all — by reading code spans the way a Markdown renderer does. WHAT IT CANNOT SEE: a pipe in plain prose outside backticks still makes an extra cell. That is a real rendering defect (the table shows an extra column), so a gate convicting it is right. THE COLUMN COMES FROM THE HEADER, never from a fixed index. Each register table declares its columns (`| ID | Category | Sev | What | State | … |`); `rows()` remembers the last header it saw and reports each row's cells by NAME. A gate that hard-codes "the state is cell 3" is wrong the day someone adds a column — the closed gate's original `cells[3]` assumption is exactly that. """ import io import re ROW = re.compile(r"^\| \*\*(R-\d+[a-z]?)\*\* \|") HEADER = re.compile(r"^\|\s*ID\s*\|") # A verdict ends at the first separator. `READY — RE-RANKED UP (R-86 closed)` leads with READY. SEPARATOR = re.compile(u"[—–(:,.;]|\\s-\\s") # THE TWO VOCABULARIES. A leading verdict is the first word of the state cell (emphasis stripped). # OPEN_STATES is the whole list a register row may lead with; anything else is either finished # (CLOSED_FAMILY — the row belongs in CLOSED-ITEMS.md) or unknown (a word nobody defined). OPEN_STATES = ("READY", "OPEN", "BLOCKED", "WATCHING", "WAITING-ON-OPERATOR", "NARROWED", "DEFERRED", "VERIFY") CLOSED_FAMILY = re.compile(r"^(CLOSED|SHIPPED|FIXED|KILLED|DONE|SUPERSEDED|OBSOLETE|WITHDRAWN|" r"MERGED|DISCHARGED|RULED|MOVED|BANKED|PROVEN-LIVE|EXECUTED|ANSWERED|" r"FOLDED|DECIDED|RESOLVED|DOCUMENTED|PASSED|COMPLETE|COMPLETED|✅)", re.I) def split_cells(line): """The cells of a markdown table row, skipping pipes inside `code` and escaped `\\|`. Returns the inner cells only (the empty strings outside the outer pipes are dropped), or None when the row does not end with `|` — such a row has lost its last cell. """ s = line.rstrip() if not s.endswith("|"): return None out, cur, code, i = [], [], False, 0 while i < len(s): ch = s[i] if ch == "\\" and i + 1 < len(s) and s[i + 1] == "|": cur.append("\\|") i += 2 continue if ch == "`": code = not code if ch == "|" and not code: out.append("".join(cur)) cur = [] else: cur.append(ch) i += 1 out.append("".join(cur)) return [c.strip() for c in out[1:-1]] def leading_verdict(cell): """The verdict word(s) before the first separator, with Markdown emphasis stripped.""" text = cell.replace("*", "").replace("~", "").replace("`", "").strip() return SEPARATOR.split(text)[0].strip() def first_word(cell): v = leading_verdict(cell) return v.split()[0].upper() if v.split() else "" def rows(path): """Yield (line_no, id, columns, cells, line) for every register row. `columns` is the header of the table the row sits in (a list of names, or None when no header has been seen yet); `cells` is split_cells(line) (None when the row lost its last pipe). """ columns = None for n, line in enumerate(io.open(path, encoding="utf-8"), 1): line = line.rstrip("\n") if HEADER.match(line): columns = split_cells(line) continue m = ROW.match(line) if not m: continue yield n, m.group(1), columns, split_cells(line), line def cell(columns, cells, name): """The named cell of a row, or None when the table has no such column or the row is short.""" if not columns or cells is None or name not in columns: return None i = columns.index(name) return cells[i] if i < len(cells) else None