From 94555614ab5222b9792ca047456eace81c1f0807 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Tue, 1 Sep 2026 12:42:42 +0200 Subject: [PATCH] observations_gate: normalise emphasis before matching a marker (R-419 follow-up) My first fix was too strict. It anchored a marker to a line start or a bare '. ', which misses the commonest real shape - a bolded sentence followed by a bolded marker: '...them.** **FILED: R-427**'. CAUGHT BY THE GATE CONVICTING THE VERY REPORT THAT DOCUMENTS IT, on two of its own observations. That is the both-directions check working: a gate that rejects the decoy AND the genuine article is worse than the hole it replaced. Code spans are stripped FIRST and that order matters - a marker inside backticks is being talked about, never used, and removing it is what makes the R-419 decoy fail. Emphasis is stripped second so '**FILED: R-1**' and 'FILED: R-1' are the same thing to the anchor. Decoy suite re-run: the R-419 decoy and a backticked-only mention are still REFUSED; both genuine marker shapes pass. --- scripts/observations_gate.py | 26 ++++++++++++++++++++------ 1 file changed, 20 insertions(+), 6 deletions(-) diff --git a/scripts/observations_gate.py b/scripts/observations_gate.py index b94a9dfe..9485103c 100644 --- a/scripts/observations_gate.py +++ b/scripts/observations_gate.py @@ -89,15 +89,29 @@ ITEM_RE = re.compile(r"^\s*(\d+)[.)]\s+(.*)$") # where a real marker is written and where a mention inside prose never is. `**FILED: R-417**` at # the end of a sentence is the common real shape, so a marker is also accepted after a sentence # boundary — but never mid-sentence and never inside backticks. -_MARK = r"(?:^|(?<=[.!?)]\s)|(?<=[.!?)]\s\s))[\s>*_\-]*" -FILED_RE = re.compile(_MARK + r"\*{0,2}FILED:\*{0,2}\s*(R-\d+)", - re.IGNORECASE | re.MULTILINE) -NOT_A_FINDING_RE = re.compile(_MARK + r"\*{0,2}NOT-A-FINDING:\*{0,2}\s*(\S.*)$", - re.IGNORECASE | re.MULTILINE) +# Emphasis is NORMALISED AWAY before matching (see normalise_for_markers), so the anchor only has to +# describe where a marker sits in a SENTENCE: at the start of a line, or after a sentence ends. The +# common real shape is a bolded sentence followed by a bolded marker — `...them.** **FILED: R-427**` — +# and an anchor that did not allow the emphasis run between them rejected genuine markers. That was +# caught by this gate convicting the very report that documents it. +_MARK = r"(?:^|(?<=[.!?)]\s))\s*" +FILED_RE = re.compile(_MARK + r"FILED:\s*(R-\d+)", re.IGNORECASE | re.MULTILINE) +NOT_A_FINDING_RE = re.compile(_MARK + r"NOT-A-FINDING:\s*(\S.*)$", re.IGNORECASE | re.MULTILINE) # A marker written inside backticks is being TALKED ABOUT, never used. Strip inline code spans # before matching — this is what makes the R-419 decoy fail. CODE_SPAN_RE = re.compile(r"`[^`]*`") +EMPHASIS_RE = re.compile(r"[*_]+") + + +def normalise_for_markers(body): + """Strip inline code spans, then emphasis, before looking for a marker. + + CODE SPANS GO FIRST AND THAT ORDER MATTERS: a marker inside backticks is being TALKED ABOUT, + never used, and removing it is what makes the R-419 decoy fail. Emphasis goes second so that + `**FILED: R-427**` and `FILED: R-427` are the same thing to the anchor below. + """ + return EMPHASIS_RE.sub("", CODE_SPAN_RE.sub("", body)) REGISTERS = [ os.path.join(ROOT, "documentation", "backlog", "OPEN-ITEMS.md"), @@ -204,7 +218,7 @@ def main(): convictions = [] satisfied = [] for num, body in items: - body = CODE_SPAN_RE.sub("", body) # R-419: a marker inside backticks is a mention + body = normalise_for_markers(body) # R-419 filed = FILED_RE.findall(body) declared = NOT_A_FINDING_RE.findall(body) first_line = body.split("\n")[0].strip()