#!/usr/bin/env python3 # -*- coding: utf-8 -*- """hu_grep.py — count the lines holding a Hungarian (accented) string, and REFUSE to say "0" untested (R-364). Usage: python3 scripts/hu_grep.py PATTERN --anchor ASCII_TEXT [--negative TEXT] PATH [PATH ...] Exit 0 found (prints the count and each file:line) · 1 a TESTED zero · 2 REFUSED (the zero is not evidence) WHY. Accented-text search is an instrument that silently transforms its input. Three near-misses are on record: `ssh → pct exec → bash -c` mangled a pattern (2026-07-20); `kubectl exec … sh -c grep` returned 0 for three strings that WERE there (2026-08-13); `tar -tf` printed `őszibarack.md` as `\\305\\221szibarack.md` (2026-08-21). Each time the zero looked exactly like "absent". Discipline alone failed, so the judgement lives here: * the file bytes are read by Python, never through a shell, and matched as UTF-8 bytes; * a pattern that ARRIVED transformed — octal escapes like `\\305\\221`, or U+FFFD — is refused outright; * for a pattern with any byte >= 0x80, a ZERO is reported only when - the ASCII ANCHOR (text you know is in these files) is found — the instrument can see the files; and - the NEGATIVE control (default: a string nobody writes) returns zero — the instrument can say no; and - the same pattern in the other Unicode normal form (NFC <-> NFD) is also absent — an "ő" typed as one code point does not match "o" + combining mark, and that zero would be a false one. Otherwise it prints REFUSED with the reason and exits 2. An ASCII pattern is a plain count (exit 1 on zero) — the transformations above do not touch it. Pure Python, stdlib only: runs on the BusyBox CI runner and on any box with python3. """ import argparse import os import re import sys import unicodedata DEFAULT_NEGATIVE = "zz-hu-grep-negative-control-qx7" OCTAL_ESCAPE = re.compile(r"\\[0-3][0-7]{2}") def iter_files(paths): for p in paths: if os.path.isdir(p): for dp, dns, fns in os.walk(p): dns[:] = sorted(d for d in dns if d not in (".git", "node_modules", "__pycache__")) for f in sorted(fns): yield os.path.join(dp, f) else: yield p def hits(needle, paths): """[(path, line_no)] for every line whose bytes contain needle (a str, matched as UTF-8 bytes).""" b = needle.encode("utf-8") out = [] for f in iter_files(paths): try: with open(f, "rb") as fh: for n, line in enumerate(fh, 1): if b in line: out.append((f, n)) except OSError as e: raise SystemExit("REFUSED: cannot read %s (%s) — an unreadable file is not an absent string" % (f, e)) return out def judge(pattern, anchor, negative, paths): """(exit_code, message, found_lines).""" if "�" in pattern or OCTAL_ESCAPE.search(pattern): return 2, ("REFUSED: the pattern arrived TRANSFORMED (%r) — octal escapes or U+FFFD mean a shell or a " "listing re-encoded it. Pass the real characters." % pattern), [] found = hits(pattern, paths) if found: return 0, "%d line(s)" % len(found), found if all(ord(c) < 0x80 for c in pattern): return 1, "0 line(s)", [] # A zero for an accented pattern: only with all three controls behaving. if not anchor or any(ord(c) >= 0x80 for c in anchor): return 2, "REFUSED: an accented pattern needs --anchor with ASCII text known to be in these files", [] a = hits(anchor, paths) if not a: return 2, ("REFUSED: the anchor %r was not found either — the instrument cannot see these files, so " "its zero is not evidence" % anchor), [] if hits(negative, paths): return 2, "REFUSED: the negative control %r was FOUND — this instrument cannot say no" % negative, [] for form in ("NFC", "NFD"): other = unicodedata.normalize(form, pattern) if other != pattern: h = hits(other, paths) if h: return 2, ("REFUSED: 0 for the pattern as typed, but %d line(s) hold its %s form — the files and " "the pattern use different Unicode normal forms" % (len(h), form)), [] return 1, ("0 line(s) — a TESTED zero: anchor %r found on %d line(s), negative control 0, other normal " "form 0" % (anchor, len(a))), [] def main(argv=None): ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) ap.add_argument("pattern") ap.add_argument("paths", nargs="+") ap.add_argument("--anchor", default="") ap.add_argument("--negative", default=DEFAULT_NEGATIVE) a = ap.parse_args(argv) rc, msg, found = judge(a.pattern, a.anchor, a.negative, a.paths) for f, n in found: print("%s:%d" % (f, n)) print(msg) return rc if __name__ == "__main__": sys.exit(main())