v0.247.0: i18n spike — the dashboard can speak English, Hungarian byte-identical
gates / gates (push) Successful in 19s

Message bundles (internal/i18n) expanded into templates before parsing, one
template set per language. Launcher, /backups, /apps/<slug> and the layout
converted; household language setting, POST /settings/language, ?lang= override,
report field. Parity test against fixtures captured from unconverted templates;
copy gates read templates expanded; new i18n_missing_gate.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-17 14:50:57 +02:00
parent 89dd3e94b1
commit 612c417024
49 changed files with 11734 additions and 248 deletions
+3
View File
@@ -86,6 +86,9 @@ GATES = [
("observations", SHARED_OBSERVATIONS, [REPO], True, True),
# R-470/R-472 — the newest release header states its MinAgent; the hub's floor reads it. Fast.
("minagent-header", os.path.join(SCRIPTS, "minagent_header_gate.py"), [], True, True),
# v0.247.0 — the i18n bundles: keys exist, no orphans, the English gap and the formal-form debt
# only shrink (ratchets), no pleading English. Fast: stdlib file reads.
("i18n", os.path.join(SCRIPTS, "i18n_missing_gate.py"), [], True, True),
# R-404 — ADVISORY. Reports the golden debt where it is created; never refuses.
("golden-notice", GOLDEN_NOTICE, [REPO], True, False),
]
+15 -1
View File
@@ -7,6 +7,9 @@ Exit 1 if any emoji/pictograph remains in web+setup templates.
"""
import io, os, sys, unicodedata
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import i18n_bundle # noqa: E402
ROOTS = [
os.path.join("internal", "web", "templates"),
os.path.join("internal", "setup", "templates"),
@@ -35,8 +38,19 @@ ALLOW = set("✓✗✔✘•●○■▶") # ✓ ✗ ✔ ✘ • ● ○ ■
def scan(path):
# v0.247.0: the copy of a converted template lives in the i18n bundle -- scan every language's
# expansion, so an emoji in a translation is caught where it would render.
hits = []
for lineno, line in enumerate(io.open(path, encoding="utf-8"), 1):
for lang in i18n_bundle.LANGS:
for h in _scan_text(i18n_bundle.read_template(path, lang)):
if h not in hits:
hits.append(h)
return hits
def _scan_text(text):
hits = []
for lineno, line in enumerate(text.split("\n"), 1):
for ch in line:
if ch in ALLOW:
continue
+48
View File
@@ -0,0 +1,48 @@
# -*- coding: utf-8 -*-
"""i18n_bundle.py -- read a dashboard template the way the controller renders it.
Since v0.247.0 a template's copy lives in internal/i18n/locales/<lang>.json and the template carries
`{{T "key"}}` markers, expanded before parsing (internal/i18n/i18n.go). A gate that reads the raw
template no longer sees that copy -- the retrieval-promise gate went STALE the moment the first page
was converted, which is how this module was found to be needed. Gates that judge copy, markup or
template actions call expand()/read_template() so they judge what a household actually receives, in
every language.
The marker pattern is the Go one, character for character (markerRe); an undefined key is left in
place, exactly as the loader leaves it.
"""
import io
import json
import os
import re
_HERE = os.path.dirname(os.path.abspath(__file__))
LOCALES = os.path.join(_HERE, "..", "internal", "i18n", "locales")
LANGS = ("hu", "en")
MARKER = re.compile(r'\{\{\s*T\s+"([A-Za-z0-9_.\-]+)"\s*\}\}')
_cache = {}
def load(lang):
if lang not in _cache:
p = os.path.join(LOCALES, lang + ".json")
_cache[lang] = json.load(io.open(p, encoding="utf-8")) if os.path.exists(p) else {}
return _cache[lang]
def expand(text, lang):
hu, own = load("hu"), load(lang)
def rep(m):
k = m.group(1)
if k in own:
return own[k]
if k in hu:
return hu[k]
return m.group(0)
return MARKER.sub(rep, text)
def read_template(path, lang="hu"):
return expand(io.open(path, encoding="utf-8").read(), lang)
+445
View File
@@ -0,0 +1,445 @@
#!/usr/bin/env python3
"""i18n_extract.py -- move the Hungarian copy of a dashboard template into the message bundle.
python3 scripts/i18n_extract.py internal/web/templates/launcher.html [more.html ...]
For each file: every Hungarian text run, attribute value and inline-JS string is replaced IN PLACE by
a `{{T "<file>.<slug>"}}` marker, and the exact text it replaced is written to
internal/i18n/locales/hu.json under that key. Nothing else in the file moves -- surrounding
whitespace, tags and template actions stay where they were.
Why that is safe (and what proves it): internal/i18n expands every marker back to the bundle text
BEFORE html/template parses the file (see internal/i18n/i18n.go). The Hungarian template set is
therefore parsed from the original bytes, in the original escaping contexts. The parity test
(internal/web/i18n_parity_test.go) renders the converted pages and compares them byte-for-byte with
fixtures captured from the unconverted templates -- this tool is not trusted, it is checked.
A "run" is the unit a translator needs: text, value actions ({{.Name}}) and inline tags
(<strong>, <a>, <br>, ...) up to the next block tag or control action ({{if}}, {{else}}, {{end}},
{{range}}, comments). Word order therefore stays inside one message. English values of JS-context
keys must not contain a bare quote, backslash or newline -- pinned by TestI18nJSContextValuesAreSafe.
Idempotent: a file already carrying markers keeps them; a run that already is a marker is skipped.
Source is ASCII-only; Hungarian letters are written as \\u escapes.
"""
import collections
import json
import os
import re
import sys
import unicodedata
HERE = os.path.dirname(os.path.abspath(__file__))
CTRL = os.path.dirname(HERE)
HU_JSON = os.path.join(CTRL, "internal", "i18n", "locales", "hu.json")
HU_LETTERS = set("\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170")
# ASCII-only Hungarian: a word with no accented letter is invisible to the letter test. Case-insensitive.
# The last line was added after a review of what the first pass left behind (Megszakadt, Elavult,
# Csak x86, Pi kompatibilis, Rendszermonitor, Most nem, sikertelen, ismeretlen).
ASCII_HU = re.compile(r"\b(Fut|Nincs|Igen|Nem|Hiba|Mentve|Rendben|Adatok|Napi|Heti|Havi|Mindig|Soha|Tegnap|"
r"Perc|Letiltva|Bekapcsolva|Kikapcsolva|Folyamatban|Sikertelen|Sikeres|Tartalom|"
r"Kapcsolat|Vissza|Megosztva|Kezdeti|Rendszer|Csatlakozva|Kulcs|Megnyit|Elrejt|"
r"Ismeretlen|Szabad|Foglalt|Kijelentkezes|Napok|nap|perc|ora|"
r"Megszakadt|Elavult|Csak|kompatibilis|Rendszermonitor|Most nem)\b", re.I)
INLINE = {"strong", "em", "b", "i", "br", "code", "small", "a", "kbd", "u"} # NOT span: it is layout here
VOID = {"br"}
SKIP_ATTRS = {"class", "id", "href", "src", "style", "name", "type", "for", "action", "method", "rel",
"target", "width", "height", "autocomplete", "role", "hidden", "onclick", "onerror",
"onchange", "onsubmit", "oninput", "aria-controls", "aria-expanded", "content", "charset",
"lang", "value"}
CONTROL = re.compile(r"^\{\{-?\s*(/\*|if\b|else\b|end\b|range\b|with\b|define\b|block\b|template\b|break\b|continue\b|\$\w+\s*:?=)")
MARKER = re.compile(r'^\{\{\s*T\s+"[^"]+"\s*\}\}$')
def is_hu(s):
return any(c in HU_LETTERS for c in s) or bool(ASCII_HU.search(s))
def slugify(text):
t = re.sub(r"\{\{.*?\}\}", " ", text)
t = re.sub(r"<[^>]*>", " ", t)
t = re.sub(r"&[a-z]+;|&#\d+;", " ", t)
t = unicodedata.normalize("NFKD", t)
t = "".join(c for c in t if not unicodedata.combining(c)).lower()
words = re.findall(r"[a-z0-9]+", t)
slug = "_".join(words[:5])[:40].strip("_")
return slug or "text"
# ---------------------------------------------------------------------------------------------------
# tokenizer: (kind, start, end) kind in action | tag | comment | text | script | style
# ---------------------------------------------------------------------------------------------------
def find_action_end(src, i):
if src.startswith("{{/*", i) or src.startswith("{{- /*", i):
j = src.find("*/", i)
j = src.find("}}", j)
else:
j = src.find("}}", i + 2)
return len(src) if j < 0 else j + 2
def tokenize(src):
toks, i, n = [], 0, len(src)
while i < n:
if src.startswith("{{", i):
j = find_action_end(src, i)
toks.append(("action", i, j))
i = j
elif src.startswith("<!--", i):
j = src.find("-->", i)
j = n if j < 0 else j + 3
toks.append(("comment", i, j))
i = j
elif src[i] == "<" and i + 1 < n and (src[i + 1].isalpha() or src[i + 1] in "/!"):
j, inq = i + 1, False
while j < n:
if src.startswith("{{", j):
j = find_action_end(src, j)
continue
if src[j] == '"':
inq = not inq
elif src[j] == ">" and not inq:
break
j += 1
j = min(j + 1, n)
toks.append(("tag", i, j))
name = tag_name(src[i:j])
if name in ("script", "style") and not src[i:j].startswith("</"):
close = re.compile(r"</%s\s*>" % name, re.I).search(src, j)
k = close.start() if close else n
toks.append((name, j, k))
i = k
else:
i = j
else:
j = i
while j < n and not src.startswith("{{", j) and not (
src[j] == "<" and j + 1 < n and (src[j + 1].isalpha() or src[j + 1] in "/!")):
j += 1
toks.append(("text", i, j))
i = j
return toks
def tag_name(tag):
m = re.match(r"</?\s*([a-zA-Z0-9]+)", tag)
return m.group(1).lower() if m else ""
# ---------------------------------------------------------------------------------------------------
# runs
# ---------------------------------------------------------------------------------------------------
class Ctx:
def __init__(self, prefix, bundle, stats):
self.prefix, self.bundle, self.stats = prefix, bundle, stats
def key_for(self, text):
base = f"{self.prefix}.{slugify(text)}"
key, k = base, 2
while key in self.bundle and self.bundle[key] != text:
key, k = f"{base}_{k}", k + 1
self.bundle[key] = text
return key
def edits_for_fragment(src, base, ctx, allow_attrs=True):
"""Return [(start, end, replacement)] for one HTML fragment (a template, or a JS string's body)."""
toks = tokenize(src)
edits = []
run = []
def flush():
nonlocal run
r = run
run = []
# trim edges: whitespace text, unbalanced/void tags
while r:
k, s, e = r[0]
if k == "text" and not src[s:e].strip():
r = r[1:]
continue
if k == "tag" and not balanced_open(src, r, 0):
r = r[1:]
continue
break
while r:
k, s, e = r[-1]
if k == "text" and not src[s:e].strip():
r = r[:-1]
continue
if k == "tag" and not balanced_close(src, r, len(r) - 1):
r = r[:-1]
continue
break
# a wrapper pair around the whole run (<span class="x">...</span>) stays outside the message
while len(r) >= 2 and r[0][0] == "tag" and r[-1][0] == "tag" and balanced_open(src, r, 0) \
and matching_close(src, r, 0) == len(r) - 1:
r = strip_ws(src, r[1:-1])
if not r:
return
segs = [r]
if not inline_balanced(src, r):
segs = [[t] for t in r if t[0] != "tag"]
for seg in segs:
s0, e0 = seg[0][1], seg[-1][2]
chunk = src[s0:e0]
visible = "".join(src[s:e] for k, s, e in seg if k == "text")
if not is_hu(visible):
continue
lead = len(chunk) - len(chunk.lstrip())
trail = len(chunk) - len(chunk.rstrip())
core = chunk.strip()
if MARKER.match(core):
continue
key = ctx.key_for(core)
ctx.stats["run"] += 1
edits.append((base + s0 + lead, base + e0 - trail, '{{T "%s"}}' % key))
for tok in toks:
kind, s, e = tok
text = src[s:e]
if kind == "text":
run.append(tok)
elif kind == "action":
if CONTROL.match(text) or MARKER.match(text):
flush()
else:
run.append(tok)
elif kind == "tag" and tag_name(text) in INLINE and not has_hu_attr(text) and not is_button_link(src, tok):
run.append(tok)
else:
flush()
if kind == "tag" and allow_attrs:
edits.extend(attr_edits(text, base + s, ctx))
elif kind == "script":
edits.extend(js_edits(text, base + s, ctx))
flush()
return edits
def is_button_link(src, tok):
"""An <a class="btn ..."> is a control of its own, not a word inside a sentence. Its closing </a>
is found by walking forward; both the opening and closing tag break the run."""
k, s, e = tok
t = src[s:e]
if tag_name(t) != "a":
return False
if t.startswith("</"):
opener = src.rfind("<a", 0, s)
return opener >= 0 and bool(re.search(r'class="[^"]*\bbtn\b', src[opener:src.find(">", opener) + 1]))
return bool(re.search(r'class="[^"]*\bbtn\b', t))
def has_hu_attr(tag):
return any(is_hu(v) and a.lower() not in SKIP_ATTRS for a, v in attrs(tag))
def attrs(tag):
out = []
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
pos = 0
while True:
m = pat.search(tag, pos)
if not m:
break
j = m.end()
start = j
while j < len(tag):
if tag.startswith("{{", j):
j = find_action_end(tag, j)
continue
if tag[j] == '"':
break
j += 1
out.append((m.group(1), tag[start:j]))
pos = j + 1
return out
def attr_edits(tag, base, ctx):
edits = []
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
pos = 0
while True:
m = pat.search(tag, pos)
if not m:
break
j = start = m.end()
while j < len(tag):
if tag.startswith("{{", j):
j = find_action_end(tag, j)
continue
if tag[j] == '"':
break
j += 1
name, val = m.group(1).lower(), tag[start:j]
pos = j + 1
if name in SKIP_ATTRS or not is_hu(re.sub(r"\{\{.*?\}\}", "", val)):
continue
if re.search(r"\{\{-?\s*(if|else|end|range|with)\b", val) or MARKER.match(val.strip()):
continue # conditional attribute copy: left for a hand edit (reported)
key = ctx.key_for(val)
ctx.stats["attr"] += 1
edits.append((base + start, base + j, '{{T "%s"}}' % key))
return edits
def js_edits(body, base, ctx):
"""String literals inside a <script> body. Comments, regex literals and template actions are
skipped so an apostrophe in a comment or a quote inside {{if eq .X "y"}} does not open a string."""
edits, i, n = [], 0, len(body)
prev = ""
while i < n:
c = body[i]
if body.startswith("{{", i):
i = find_action_end(body, i)
continue
if body.startswith("//", i):
j = body.find("\n", i)
i = n if j < 0 else j
continue
if body.startswith("/*", i):
j = body.find("*/", i + 2)
i = n if j < 0 else j + 2
continue
if c == "/" and prev in "(,=:[!&|?{};":
j = i + 1
while j < n and body[j] != "/" and body[j] != "\n":
j += 2 if body[j] == "\\" else 1
i = j + 1
prev = "/"
continue
if c in "'\"":
j = i + 1
while j < n and body[j] != c and body[j] != "\n":
if body.startswith("{{", j):
j = find_action_end(body, j)
continue
j += 2 if body[j] == "\\" else 1
content = body[i + 1:j]
if is_hu(re.sub(r"\{\{.*?\}\}", "", content)):
# A JS string is often a PIECE of markup: '\')">T\u00f6rl\u00e9s</button>' starts inside a
# tag. Everything up to the first '>' that precedes any '<' is that tag's tail, not text.
skip = 0
gt, lt = content.find(">"), content.find("<")
if gt >= 0 and (lt < 0 or gt < lt):
skip = gt + 1
sub = edits_for_fragment(content[skip:], base + i + 1 + skip, ctx, allow_attrs=False)
ctx.stats["js"] += len(sub)
edits.extend(sub)
i = j + 1
prev = "s"
continue
if not c.isspace():
prev = c
i += 1
return edits
def strip_ws(src, r):
while r and r[0][0] == "text" and not src[r[0][1]:r[0][2]].strip():
r = r[1:]
while r and r[-1][0] == "text" and not src[r[-1][1]:r[-1][2]].strip():
r = r[:-1]
return r
def matching_close(src, run, idx):
name = tag_name(src[run[idx][1]:run[idx][2]])
depth = 0
for j in range(idx, len(run)):
k2, s2, e2 = run[j]
if k2 != "tag" or tag_name(src[s2:e2]) != name:
continue
depth += -1 if src[s2:e2].startswith("</") else 1
if depth == 0:
return j
return -1
def balanced_open(src, run, idx):
k, s, e = run[idx]
t = src[s:e]
name = tag_name(t)
if name not in INLINE or name in VOID or t.startswith("</") or t.endswith("/>"):
return False
depth = 0
for k2, s2, e2 in run[idx:]:
if k2 != "tag" or tag_name(src[s2:e2]) != name:
continue
depth += -1 if src[s2:e2].startswith("</") else 1
if depth == 0:
return True
return False
def balanced_close(src, run, idx):
k, s, e = run[idx]
t = src[s:e]
name = tag_name(t)
if name not in INLINE or name in VOID or not t.startswith("</"):
return False
depth = 0
for k2, s2, e2 in reversed(run[:idx + 1]):
if k2 != "tag" or tag_name(src[s2:e2]) != name:
continue
depth += 1 if src[s2:e2].startswith("</") else -1
if depth == 0:
return True
return False
def inline_balanced(src, run):
depth = collections.Counter()
for k, s, e in run:
if k != "tag":
continue
t = src[s:e]
name = tag_name(t)
if name in VOID or t.endswith("/>"):
continue
depth[name] += -1 if t.startswith("</") else 1
if depth[name] < 0:
return False
return all(v == 0 for v in depth.values())
def main(argv):
if not argv:
print(__doc__)
return 2
bundle = collections.OrderedDict()
if os.path.exists(HU_JSON):
with open(HU_JSON, encoding="utf-8") as f:
bundle.update(json.load(f, object_pairs_hook=collections.OrderedDict))
for path in argv:
src = open(path, encoding="utf-8").read()
prefix = os.path.splitext(os.path.basename(path))[0]
stats = collections.Counter()
ctx = Ctx(prefix, bundle, stats)
edits = sorted(edits_for_fragment(src, 0, ctx), key=lambda e: e[0])
out, last = [], 0
for s, e, rep in edits:
if s < last:
raise SystemExit(f"{path}: overlapping edit at {s}")
out.append(src[last:s])
out.append(rep)
last = e
out.append(src[last:])
with open(path, "w", encoding="utf-8") as f:
f.write("".join(out))
print(f"{path}: {stats['run']} runs, {stats['attr']} attributes, {stats['js']} of the runs inside JS strings")
os.makedirs(os.path.dirname(HU_JSON), exist_ok=True)
with open(HU_JSON, "w", encoding="utf-8") as f:
json.dump(bundle, f, ensure_ascii=False, indent=2)
f.write("\n")
print(f"{HU_JSON}: {len(bundle)} keys")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))
+137
View File
@@ -0,0 +1,137 @@
# -*- coding: utf-8 -*-
"""i18n_missing_gate.py -- the message bundles are complete, consistent and in the product's voice.
Run from controller/: python3 scripts/i18n_missing_gate.py
Exit 0 clean · 1 convicted · 2 inconclusive (bundles or templates not found).
Design: felhom.eu/documentation/architecture/10-localisation.md. The Go tests pin the runtime
(parity, fallback, parameters, JS-context safety); this gate pins the SOURCE, before a build:
1. KEYS EXIST. Every `{{T "key"}}` marker in any dashboard template names a key in hu.json.
(The loader refuses an undefined key at startup; this names the key at push time instead.)
2. NO ORPHANS. Every en.json key translates a hu.json key (".one"/".other" count for their base).
3. THE ENGLISH GAP ONLY SHRINKS. The number of hu keys with no English is printed and compared with
EN_MISSING_CEILING. More than the ceiling convicts (a slice converted copy and did not translate
it); LESS also convicts, until the ceiling is lowered to match -- a ratchet that cannot be
loosened by forgetting it. Raising it is a decision recorded in 10-localisation.md.
4. VOICE, per language.
en -- second person, plain: no "please", no "kindly" (the household is not being begged).
hu -- the product speaks in „te". Formal („ön") forms already in the converted copy are known
debt (R-516) and COUNTED against HU_FORMAL_CEILING with the same ratchet: a new formal form
convicts; fixing one requires lowering the ceiling. The Hungarian copy itself is not changed by
the release that introduced this gate (parity rule), which is why this is a ratchet and not a ban.
Hungarian is matched by ASCII-folded stems (the workspace rule for Hungarian search), with a
positive and a negative control printed on every run.
"""
import io
import json
import os
import re
import sys
import unicodedata
HERE = os.path.dirname(os.path.abspath(__file__))
CTRL = os.path.dirname(HERE)
LOCALES = os.path.join(CTRL, "internal", "i18n", "locales")
TEMPLATE_ROOTS = [os.path.join(CTRL, "internal", "web", "templates")]
MARKER = re.compile(r'\{\{\s*T\s+"([A-Za-z0-9_.\-]+)"\s*\}\}')
EN_MISSING_CEILING = 0
HU_FORMAL_CEILING = 6
EN_FORBIDDEN = re.compile(r"\b(please|kindly)\b", re.I)
# Formal-address stems, ASCII-folded. Each is a verb form or phrase that only the „ön" register uses.
HU_FORMAL = [
"olvassa be", "kerjuk", "vegye fel", "ujratelepiti az", "importalnia", "biztosan kikapcsolja",
"ellenorizze", "csatlakoztassa", "adjon hozza", "nezze meg", "kattintson", "valassza ki",
]
def fold(s):
s = unicodedata.normalize("NFKD", s)
return "".join(c for c in s if not unicodedata.combining(c)).lower()
def load(lang):
p = os.path.join(LOCALES, lang + ".json")
if not os.path.exists(p):
return None
return json.load(io.open(p, encoding="utf-8"))
def templates():
out = []
for root in TEMPLATE_ROOTS:
for dirpath, _dirs, names in os.walk(root):
for n in sorted(names):
if n.endswith(".html"):
out.append(os.path.join(dirpath, n))
return sorted(out)
def main():
hu, en = load("hu"), load("en")
tpls = templates()
if hu is None or en is None or not tpls:
print("i18n gate INCONCLUSIVE: bundles or templates not found under %s" % CTRL)
return 2
# Controls for the ASCII-folded matcher.
if "kerjuk" not in fold("Kérjük, vegye fel"):
print("i18n gate INCONCLUSIVE: positive control failed (folding does not strip accents)")
return 2
if any(stem in fold("Csatlakoztasd újra a meghajtót") for stem in HU_FORMAL):
print("i18n gate INCONCLUSIVE: negative control failed (an informal te-form sentence matched a formal stem)")
return 2
problems = []
used = 0
for path in tpls:
rel = os.path.relpath(path, CTRL)
for i, line in enumerate(io.open(path, encoding="utf-8"), 1):
for m in MARKER.finditer(line):
used += 1
if m.group(1) not in hu:
problems.append("%s:%d marker key %r is not in hu.json" % (rel, i, m.group(1)))
for k in en:
base = re.sub(r"\.(one|other)$", "", k)
if k not in hu and base not in hu:
problems.append("en.json key %r has no Hungarian source" % k)
if EN_FORBIDDEN.search(en[k]):
problems.append("en.json %r says %r -- plain second person, no pleading" % (k, EN_FORBIDDEN.search(en[k]).group(0)))
missing = [k for k in hu if k not in en and (k + ".other") not in en]
formal = [(k, stem) for k in hu for stem in HU_FORMAL if stem in fold(hu[k])]
print("i18n: %d markers in %d templates; hu %d keys, en %d keys" % (used, len(tpls), len(hu), len(en)))
print("i18n: English missing %d (ceiling %d); Hungarian formal forms %d (ceiling %d, R-516)"
% (len(missing), EN_MISSING_CEILING, len(formal), HU_FORMAL_CEILING))
for k, stem in formal:
print(" formal: %s (%s)" % (k, stem))
if len(missing) > EN_MISSING_CEILING:
problems.append("%d Hungarian keys have no English, ceiling is %d: %s"
% (len(missing), EN_MISSING_CEILING, ", ".join(missing[:10])))
elif len(missing) < EN_MISSING_CEILING:
problems.append("English gap shrank to %d -- lower EN_MISSING_CEILING from %d to match (the ratchet)"
% (len(missing), EN_MISSING_CEILING))
if len(formal) > HU_FORMAL_CEILING:
problems.append("%d formal (on-form) forms in hu.json, ceiling %d -- the product speaks in the te-form" % (len(formal), HU_FORMAL_CEILING))
elif len(formal) < HU_FORMAL_CEILING:
problems.append("formal forms fell to %d -- lower HU_FORMAL_CEILING from %d to match (the ratchet)"
% (len(formal), HU_FORMAL_CEILING))
if problems:
for p in problems:
print(" " + p)
print("I18N GATE FAILED: %d problem(s)" % len(problems))
return 1
print("i18n gate OK")
return 0
if __name__ == "__main__":
sys.exit(main())
+1 -1
View File
@@ -33,7 +33,7 @@ SIGNATURE = {
"ˇ": "ˇ (caron — cp1250 round-trip)",
}
EXTS = (".html", ".css", ".js", ".go")
EXTS = (".html", ".css", ".js", ".go", ".json") # .json: the i18n bundles (v0.247.0) are copy
def files():
+11 -4
View File
@@ -11,6 +11,9 @@ Exit 1 if any native confirm(/prompt(/alert( call remains in the templates.
"""
import io, os, re, sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import i18n_bundle # noqa: E402
ROOTS = [
os.path.join("internal", "web", "templates"),
os.path.join("internal", "setup", "templates"),
@@ -41,10 +44,14 @@ def main():
for root in ROOTS:
for path in _html_files(root):
fn = os.path.relpath(path, root)
for lineno, line in enumerate(io.open(path, encoding="utf-8"), 1):
if NATIVE.search(line):
total += 1
print("%s:%d %s" % (fn, lineno, line.strip()[:120].encode('ascii', 'backslashreplace').decode()))
# v0.247.0: judge each language's expansion -- JS copy lives in the i18n bundle now.
seen = set()
for lang in i18n_bundle.LANGS:
for lineno, line in enumerate(i18n_bundle.read_template(path, lang).split("\n"), 1):
if NATIVE.search(line) and (lineno, line) not in seen:
seen.add((lineno, line))
total += 1
print("%s:%d [%s] %s" % (fn, lineno, lang, line.strip()[:120].encode('ascii', 'backslashreplace').decode()))
if total:
print("NATIVE CONFIRM GATE FAILED: %d native confirm()/prompt()/alert() call(s) remain" % total)
sys.exit(1)
+8 -1
View File
@@ -27,6 +27,9 @@ import os
import re
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import i18n_bundle # noqa: E402
_HERE = os.path.dirname(os.path.abspath(__file__))
TEMPLATES = os.path.join(_HERE, "..", "internal", "web", "templates")
@@ -119,7 +122,11 @@ def scan():
sources.append((os.path.basename(gp), gp, GO_COMMENT))
files = files + [os.path.basename(gp)]
for name, path, stripper in sources:
text = stripper.sub("", open(path, encoding="utf-8").read())
# v0.247.0: a template's copy lives in the i18n bundle; judge the page as it renders in
# Hungarian (the stems are Hungarian). English retrieval claims are not scanned yet -- R-row
# in 10-localisation.md; the English bundle covers three pages today.
raw = open(path, encoding="utf-8").read()
text = stripper.sub("", i18n_bundle.expand(raw, "hu") if path.endswith(".html") else raw)
for stem in STEMS:
for m in re.finditer(re.escape(stem) + r"[a-záéíóöőúüű]*", text):
line = text[: m.start()].count("\n") + 1
+14 -1
View File
@@ -39,6 +39,9 @@ import os
import re
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import i18n_bundle # noqa: E402
HERE = os.path.dirname(os.path.abspath(__file__))
CTRL = os.path.dirname(HERE)
TPL = os.path.join(CTRL, "internal", "web", "templates")
@@ -71,8 +74,18 @@ ALLOWLIST = {
def check(path):
# v0.247.0: bundle values carry template actions ({{.RecoveryAbandonDate}}), so a secret-named
# expression can arrive through a translation. Judge every language's expansion; dedupe.
out = []
for lang in i18n_bundle.LANGS:
for c in _check_src(i18n_bundle.read_template(path, lang)):
if c not in out:
out.append(c)
return out
def _check_src(src):
convictions = []
src = open(path, encoding="utf-8").read()
for m in ACTION.finditer(src):
expr = m.group(1).strip()
# A TEMPLATE comment `{{/* ... */}}` is stripped by html/template and never reaches the
+33
View File
@@ -44,6 +44,7 @@ COVERS = {
"debug-routes": "a live dispatcher case commented out - the button survives, the handler dies",
"golden-notice": "R-410 in the other direction: an empty dir must not count as a bake",
"minagent-header": "the word MinAgent in prose / a code span, not the header line (test_minagent_header_gate.py)",
"i18n": "an undefined marker key, a pleading English value, a shrinking/growing gap (v0.247.0)",
}
fails = []
@@ -133,6 +134,38 @@ def _comment_out_the_partial(src):
swapped("app-row-dedup/comment", "app_row_dedup_gate.py",
os.path.join(TPL, "dashboard.html"), _comment_out_the_partial)
# --- i18n (v0.247.0): copy moved OUT of the templates into the bundles. Two families of decoy: ---
# --- the new gate itself, and every copy gate that must still see copy that now lives there. ---
LOC = os.path.join(CTRL, "internal", "i18n", "locales")
EN_JSON, HU_JSON = os.path.join(LOC, "en.json"), os.path.join(LOC, "hu.json")
LAUNCHER = os.path.join(TPL, "launcher.html")
def _one(old, new):
def t(src):
assert src.count(old) == 1, "decoy anchor %r not found exactly once — the decoy cannot be built" % old
return src.replace(old, new)
return t
swapped("i18n/undefined-key", "i18n_missing_gate.py", LAUNCHER,
_one('{{T "launcher.link_masolasa"}}', '{{T "launcher.no_such_key"}}'))
swapped("i18n/pleading", "i18n_missing_gate.py", EN_JSON,
_one('"launcher.link_masolasa": "Copy link"', '"launcher.link_masolasa": "Please copy the link"'))
swapped("i18n/gap-grew", "i18n_missing_gate.py", EN_JSON,
_one(' "launcher.link_masolasa": "Copy link",\n', ''))
swapped("i18n/new-formal-form", "i18n_missing_gate.py", HU_JSON,
_one('"launcher.link_masolasa": "Link másolása"', '"launcher.link_masolasa": "Kattintson a link másolásához"'))
# Scope: the copy gates read the page AS RENDERED, so copy that lives in a bundle is still judged.
swapped("emoji/bundle", "emoji_gate.py", EN_JSON,
_one('"launcher.link_masolasa": "Copy link"', '"launcher.link_masolasa": "Copy link \U0001F600"'))
swapped("native-confirm/bundle", "native_confirm_gate.py", EN_JSON,
_one('"launcher.masolva": "Copied"', '"launcher.masolva": "Copied\u0027+confirm(\u0027x\u0027)+\u0027"'))
swapped("retrieval-promise/bundle", "retrieval_promise_gate.py", HU_JSON,
_one('"launcher.link_masolasa": "Link másolása"', '"launcher.link_masolasa": "A mentésed bármikor visszaállítható."'))
swapped("secret-markup/bundle", "secret_in_markup_gate.py", EN_JSON,
_one('"launcher.link_masolasa": "Copy link"', '"launcher.link_masolasa": "Copy {{.RetrievalPassword}}"'))
# --- golden-notice: R-410's decoy, in the other direction. It is ADVISORY, so rc is never the ---
# --- question — what it COUNTED is. ---
ran += 1