557629d2bf
gates / gates (push) Successful in 2m3s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
107 lines
4.9 KiB
Python
107 lines
4.9 KiB
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
"""hu_grep.py — count the lines holding a Hungarian (accented) string, and REFUSE to say "0" untested (R-364).
|
|
|
|
Usage: python3 scripts/hu_grep.py PATTERN --anchor ASCII_TEXT [--negative TEXT] PATH [PATH ...]
|
|
Exit 0 found (prints the count and each file:line) · 1 a TESTED zero · 2 REFUSED (the zero is not evidence)
|
|
|
|
WHY. Accented-text search is an instrument that silently transforms its input. Three near-misses are on record:
|
|
`ssh → pct exec → bash -c` mangled a pattern (2026-07-20); `kubectl exec … sh -c grep` returned 0 for three
|
|
strings that WERE there (2026-08-13); `tar -tf` printed `őszibarack.md` as `\\305\\221szibarack.md` (2026-08-21).
|
|
Each time the zero looked exactly like "absent". Discipline alone failed, so the judgement lives here:
|
|
|
|
* the file bytes are read by Python, never through a shell, and matched as UTF-8 bytes;
|
|
* a pattern that ARRIVED transformed — octal escapes like `\\305\\221`, or U+FFFD — is refused outright;
|
|
* for a pattern with any byte >= 0x80, a ZERO is reported only when
|
|
- the ASCII ANCHOR (text you know is in these files) is found — the instrument can see the files; and
|
|
- the NEGATIVE control (default: a string nobody writes) returns zero — the instrument can say no; and
|
|
- the same pattern in the other Unicode normal form (NFC <-> NFD) is also absent — an "ő" typed as one
|
|
code point does not match "o" + combining mark, and that zero would be a false one.
|
|
Otherwise it prints REFUSED with the reason and exits 2.
|
|
|
|
An ASCII pattern is a plain count (exit 1 on zero) — the transformations above do not touch it.
|
|
Pure Python, stdlib only: runs on the BusyBox CI runner and on any box with python3.
|
|
"""
|
|
import argparse
|
|
import os
|
|
import re
|
|
import sys
|
|
import unicodedata
|
|
|
|
DEFAULT_NEGATIVE = "zz-hu-grep-negative-control-qx7"
|
|
OCTAL_ESCAPE = re.compile(r"\\[0-3][0-7]{2}")
|
|
|
|
|
|
def iter_files(paths):
|
|
for p in paths:
|
|
if os.path.isdir(p):
|
|
for dp, dns, fns in os.walk(p):
|
|
dns[:] = sorted(d for d in dns if d not in (".git", "node_modules", "__pycache__"))
|
|
for f in sorted(fns):
|
|
yield os.path.join(dp, f)
|
|
else:
|
|
yield p
|
|
|
|
|
|
def hits(needle, paths):
|
|
"""[(path, line_no)] for every line whose bytes contain needle (a str, matched as UTF-8 bytes)."""
|
|
b = needle.encode("utf-8")
|
|
out = []
|
|
for f in iter_files(paths):
|
|
try:
|
|
with open(f, "rb") as fh:
|
|
for n, line in enumerate(fh, 1):
|
|
if b in line:
|
|
out.append((f, n))
|
|
except OSError as e:
|
|
raise SystemExit("REFUSED: cannot read %s (%s) — an unreadable file is not an absent string" % (f, e))
|
|
return out
|
|
|
|
|
|
def judge(pattern, anchor, negative, paths):
|
|
"""(exit_code, message, found_lines)."""
|
|
if "�" in pattern or OCTAL_ESCAPE.search(pattern):
|
|
return 2, ("REFUSED: the pattern arrived TRANSFORMED (%r) — octal escapes or U+FFFD mean a shell or a "
|
|
"listing re-encoded it. Pass the real characters." % pattern), []
|
|
found = hits(pattern, paths)
|
|
if found:
|
|
return 0, "%d line(s)" % len(found), found
|
|
if all(ord(c) < 0x80 for c in pattern):
|
|
return 1, "0 line(s)", []
|
|
# A zero for an accented pattern: only with all three controls behaving.
|
|
if not anchor or any(ord(c) >= 0x80 for c in anchor):
|
|
return 2, "REFUSED: an accented pattern needs --anchor with ASCII text known to be in these files", []
|
|
a = hits(anchor, paths)
|
|
if not a:
|
|
return 2, ("REFUSED: the anchor %r was not found either — the instrument cannot see these files, so "
|
|
"its zero is not evidence" % anchor), []
|
|
if hits(negative, paths):
|
|
return 2, "REFUSED: the negative control %r was FOUND — this instrument cannot say no" % negative, []
|
|
for form in ("NFC", "NFD"):
|
|
other = unicodedata.normalize(form, pattern)
|
|
if other != pattern:
|
|
h = hits(other, paths)
|
|
if h:
|
|
return 2, ("REFUSED: 0 for the pattern as typed, but %d line(s) hold its %s form — the files and "
|
|
"the pattern use different Unicode normal forms" % (len(h), form)), []
|
|
return 1, ("0 line(s) — a TESTED zero: anchor %r found on %d line(s), negative control 0, other normal "
|
|
"form 0" % (anchor, len(a))), []
|
|
|
|
|
|
def main(argv=None):
|
|
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
|
ap.add_argument("pattern")
|
|
ap.add_argument("paths", nargs="+")
|
|
ap.add_argument("--anchor", default="")
|
|
ap.add_argument("--negative", default=DEFAULT_NEGATIVE)
|
|
a = ap.parse_args(argv)
|
|
rc, msg, found = judge(a.pattern, a.anchor, a.negative, a.paths)
|
|
for f, n in found:
|
|
print("%s:%d" % (f, n))
|
|
print(msg)
|
|
return rc
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|