diff --git a/scripts/observations_gate.py b/scripts/observations_gate.py index b94a9dfe..9485103c 100644 --- a/scripts/observations_gate.py +++ b/scripts/observations_gate.py @@ -89,15 +89,29 @@ ITEM_RE = re.compile(r"^\s*(\d+)[.)]\s+(.*)$") # where a real marker is written and where a mention inside prose never is. `**FILED: R-417**` at # the end of a sentence is the common real shape, so a marker is also accepted after a sentence # boundary — but never mid-sentence and never inside backticks. -_MARK = r"(?:^|(?<=[.!?)]\s)|(?<=[.!?)]\s\s))[\s>*_\-]*" -FILED_RE = re.compile(_MARK + r"\*{0,2}FILED:\*{0,2}\s*(R-\d+)", - re.IGNORECASE | re.MULTILINE) -NOT_A_FINDING_RE = re.compile(_MARK + r"\*{0,2}NOT-A-FINDING:\*{0,2}\s*(\S.*)$", - re.IGNORECASE | re.MULTILINE) +# Emphasis is NORMALISED AWAY before matching (see normalise_for_markers), so the anchor only has to +# describe where a marker sits in a SENTENCE: at the start of a line, or after a sentence ends. The +# common real shape is a bolded sentence followed by a bolded marker — `...them.** **FILED: R-427**` — +# and an anchor that did not allow the emphasis run between them rejected genuine markers. That was +# caught by this gate convicting the very report that documents it. +_MARK = r"(?:^|(?<=[.!?)]\s))\s*" +FILED_RE = re.compile(_MARK + r"FILED:\s*(R-\d+)", re.IGNORECASE | re.MULTILINE) +NOT_A_FINDING_RE = re.compile(_MARK + r"NOT-A-FINDING:\s*(\S.*)$", re.IGNORECASE | re.MULTILINE) # A marker written inside backticks is being TALKED ABOUT, never used. Strip inline code spans # before matching — this is what makes the R-419 decoy fail. CODE_SPAN_RE = re.compile(r"`[^`]*`") +EMPHASIS_RE = re.compile(r"[*_]+") + + +def normalise_for_markers(body): + """Strip inline code spans, then emphasis, before looking for a marker. + + CODE SPANS GO FIRST AND THAT ORDER MATTERS: a marker inside backticks is being TALKED ABOUT, + never used, and removing it is what makes the R-419 decoy fail. Emphasis goes second so that + `**FILED: R-427**` and `FILED: R-427` are the same thing to the anchor below. + """ + return EMPHASIS_RE.sub("", CODE_SPAN_RE.sub("", body)) REGISTERS = [ os.path.join(ROOT, "documentation", "backlog", "OPEN-ITEMS.md"), @@ -204,7 +218,7 @@ def main(): convictions = [] satisfied = [] for num, body in items: - body = CODE_SPAN_RE.sub("", body) # R-419: a marker inside backticks is a mention + body = normalise_for_markers(body) # R-419 filed = FILED_RE.findall(body) declared = NOT_A_FINDING_RE.findall(body) first_line = body.split("\n")[0].strip()