v0.247.0: i18n spike — the dashboard can speak English, Hungarian byte-identical
gates / gates (push) Successful in 19s
gates / gates (push) Successful in 19s
Message bundles (internal/i18n) expanded into templates before parsing, one template set per language. Launcher, /backups, /apps/<slug> and the layout converted; household language setting, POST /settings/language, ?lang= override, report field. Parity test against fixtures captured from unconverted templates; copy gates read templates expanded; new i18n_missing_gate. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -86,6 +86,9 @@ GATES = [
|
||||
("observations", SHARED_OBSERVATIONS, [REPO], True, True),
|
||||
# R-470/R-472 — the newest release header states its MinAgent; the hub's floor reads it. Fast.
|
||||
("minagent-header", os.path.join(SCRIPTS, "minagent_header_gate.py"), [], True, True),
|
||||
# v0.247.0 — the i18n bundles: keys exist, no orphans, the English gap and the formal-form debt
|
||||
# only shrink (ratchets), no pleading English. Fast: stdlib file reads.
|
||||
("i18n", os.path.join(SCRIPTS, "i18n_missing_gate.py"), [], True, True),
|
||||
# R-404 — ADVISORY. Reports the golden debt where it is created; never refuses.
|
||||
("golden-notice", GOLDEN_NOTICE, [REPO], True, False),
|
||||
]
|
||||
|
||||
@@ -7,6 +7,9 @@ Exit 1 if any emoji/pictograph remains in web+setup templates.
|
||||
"""
|
||||
import io, os, sys, unicodedata
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import i18n_bundle # noqa: E402
|
||||
|
||||
ROOTS = [
|
||||
os.path.join("internal", "web", "templates"),
|
||||
os.path.join("internal", "setup", "templates"),
|
||||
@@ -35,8 +38,19 @@ ALLOW = set("✓✗✔✘•●○■▶") # ✓ ✗ ✔ ✘ • ● ○ ■
|
||||
|
||||
|
||||
def scan(path):
|
||||
# v0.247.0: the copy of a converted template lives in the i18n bundle -- scan every language's
|
||||
# expansion, so an emoji in a translation is caught where it would render.
|
||||
hits = []
|
||||
for lineno, line in enumerate(io.open(path, encoding="utf-8"), 1):
|
||||
for lang in i18n_bundle.LANGS:
|
||||
for h in _scan_text(i18n_bundle.read_template(path, lang)):
|
||||
if h not in hits:
|
||||
hits.append(h)
|
||||
return hits
|
||||
|
||||
|
||||
def _scan_text(text):
|
||||
hits = []
|
||||
for lineno, line in enumerate(text.split("\n"), 1):
|
||||
for ch in line:
|
||||
if ch in ALLOW:
|
||||
continue
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""i18n_bundle.py -- read a dashboard template the way the controller renders it.
|
||||
|
||||
Since v0.247.0 a template's copy lives in internal/i18n/locales/<lang>.json and the template carries
|
||||
`{{T "key"}}` markers, expanded before parsing (internal/i18n/i18n.go). A gate that reads the raw
|
||||
template no longer sees that copy -- the retrieval-promise gate went STALE the moment the first page
|
||||
was converted, which is how this module was found to be needed. Gates that judge copy, markup or
|
||||
template actions call expand()/read_template() so they judge what a household actually receives, in
|
||||
every language.
|
||||
|
||||
The marker pattern is the Go one, character for character (markerRe); an undefined key is left in
|
||||
place, exactly as the loader leaves it.
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
LOCALES = os.path.join(_HERE, "..", "internal", "i18n", "locales")
|
||||
LANGS = ("hu", "en")
|
||||
MARKER = re.compile(r'\{\{\s*T\s+"([A-Za-z0-9_.\-]+)"\s*\}\}')
|
||||
|
||||
_cache = {}
|
||||
|
||||
|
||||
def load(lang):
|
||||
if lang not in _cache:
|
||||
p = os.path.join(LOCALES, lang + ".json")
|
||||
_cache[lang] = json.load(io.open(p, encoding="utf-8")) if os.path.exists(p) else {}
|
||||
return _cache[lang]
|
||||
|
||||
|
||||
def expand(text, lang):
|
||||
hu, own = load("hu"), load(lang)
|
||||
|
||||
def rep(m):
|
||||
k = m.group(1)
|
||||
if k in own:
|
||||
return own[k]
|
||||
if k in hu:
|
||||
return hu[k]
|
||||
return m.group(0)
|
||||
return MARKER.sub(rep, text)
|
||||
|
||||
|
||||
def read_template(path, lang="hu"):
|
||||
return expand(io.open(path, encoding="utf-8").read(), lang)
|
||||
@@ -0,0 +1,445 @@
|
||||
#!/usr/bin/env python3
|
||||
"""i18n_extract.py -- move the Hungarian copy of a dashboard template into the message bundle.
|
||||
|
||||
python3 scripts/i18n_extract.py internal/web/templates/launcher.html [more.html ...]
|
||||
|
||||
For each file: every Hungarian text run, attribute value and inline-JS string is replaced IN PLACE by
|
||||
a `{{T "<file>.<slug>"}}` marker, and the exact text it replaced is written to
|
||||
internal/i18n/locales/hu.json under that key. Nothing else in the file moves -- surrounding
|
||||
whitespace, tags and template actions stay where they were.
|
||||
|
||||
Why that is safe (and what proves it): internal/i18n expands every marker back to the bundle text
|
||||
BEFORE html/template parses the file (see internal/i18n/i18n.go). The Hungarian template set is
|
||||
therefore parsed from the original bytes, in the original escaping contexts. The parity test
|
||||
(internal/web/i18n_parity_test.go) renders the converted pages and compares them byte-for-byte with
|
||||
fixtures captured from the unconverted templates -- this tool is not trusted, it is checked.
|
||||
|
||||
A "run" is the unit a translator needs: text, value actions ({{.Name}}) and inline tags
|
||||
(<strong>, <a>, <br>, ...) up to the next block tag or control action ({{if}}, {{else}}, {{end}},
|
||||
{{range}}, comments). Word order therefore stays inside one message. English values of JS-context
|
||||
keys must not contain a bare quote, backslash or newline -- pinned by TestI18nJSContextValuesAreSafe.
|
||||
|
||||
Idempotent: a file already carrying markers keeps them; a run that already is a marker is skipped.
|
||||
Source is ASCII-only; Hungarian letters are written as \\u escapes.
|
||||
"""
|
||||
import collections
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
CTRL = os.path.dirname(HERE)
|
||||
HU_JSON = os.path.join(CTRL, "internal", "i18n", "locales", "hu.json")
|
||||
|
||||
HU_LETTERS = set("\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170")
|
||||
# ASCII-only Hungarian: a word with no accented letter is invisible to the letter test. Case-insensitive.
|
||||
# The last line was added after a review of what the first pass left behind (Megszakadt, Elavult,
|
||||
# Csak x86, Pi kompatibilis, Rendszermonitor, Most nem, sikertelen, ismeretlen).
|
||||
ASCII_HU = re.compile(r"\b(Fut|Nincs|Igen|Nem|Hiba|Mentve|Rendben|Adatok|Napi|Heti|Havi|Mindig|Soha|Tegnap|"
|
||||
r"Perc|Letiltva|Bekapcsolva|Kikapcsolva|Folyamatban|Sikertelen|Sikeres|Tartalom|"
|
||||
r"Kapcsolat|Vissza|Megosztva|Kezdeti|Rendszer|Csatlakozva|Kulcs|Megnyit|Elrejt|"
|
||||
r"Ismeretlen|Szabad|Foglalt|Kijelentkezes|Napok|nap|perc|ora|"
|
||||
r"Megszakadt|Elavult|Csak|kompatibilis|Rendszermonitor|Most nem)\b", re.I)
|
||||
|
||||
INLINE = {"strong", "em", "b", "i", "br", "code", "small", "a", "kbd", "u"} # NOT span: it is layout here
|
||||
VOID = {"br"}
|
||||
SKIP_ATTRS = {"class", "id", "href", "src", "style", "name", "type", "for", "action", "method", "rel",
|
||||
"target", "width", "height", "autocomplete", "role", "hidden", "onclick", "onerror",
|
||||
"onchange", "onsubmit", "oninput", "aria-controls", "aria-expanded", "content", "charset",
|
||||
"lang", "value"}
|
||||
CONTROL = re.compile(r"^\{\{-?\s*(/\*|if\b|else\b|end\b|range\b|with\b|define\b|block\b|template\b|break\b|continue\b|\$\w+\s*:?=)")
|
||||
MARKER = re.compile(r'^\{\{\s*T\s+"[^"]+"\s*\}\}$')
|
||||
|
||||
|
||||
def is_hu(s):
|
||||
return any(c in HU_LETTERS for c in s) or bool(ASCII_HU.search(s))
|
||||
|
||||
|
||||
def slugify(text):
|
||||
t = re.sub(r"\{\{.*?\}\}", " ", text)
|
||||
t = re.sub(r"<[^>]*>", " ", t)
|
||||
t = re.sub(r"&[a-z]+;|&#\d+;", " ", t)
|
||||
t = unicodedata.normalize("NFKD", t)
|
||||
t = "".join(c for c in t if not unicodedata.combining(c)).lower()
|
||||
words = re.findall(r"[a-z0-9]+", t)
|
||||
slug = "_".join(words[:5])[:40].strip("_")
|
||||
return slug or "text"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
# tokenizer: (kind, start, end) kind in action | tag | comment | text | script | style
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
|
||||
def find_action_end(src, i):
|
||||
if src.startswith("{{/*", i) or src.startswith("{{- /*", i):
|
||||
j = src.find("*/", i)
|
||||
j = src.find("}}", j)
|
||||
else:
|
||||
j = src.find("}}", i + 2)
|
||||
return len(src) if j < 0 else j + 2
|
||||
|
||||
|
||||
def tokenize(src):
|
||||
toks, i, n = [], 0, len(src)
|
||||
while i < n:
|
||||
if src.startswith("{{", i):
|
||||
j = find_action_end(src, i)
|
||||
toks.append(("action", i, j))
|
||||
i = j
|
||||
elif src.startswith("<!--", i):
|
||||
j = src.find("-->", i)
|
||||
j = n if j < 0 else j + 3
|
||||
toks.append(("comment", i, j))
|
||||
i = j
|
||||
elif src[i] == "<" and i + 1 < n and (src[i + 1].isalpha() or src[i + 1] in "/!"):
|
||||
j, inq = i + 1, False
|
||||
while j < n:
|
||||
if src.startswith("{{", j):
|
||||
j = find_action_end(src, j)
|
||||
continue
|
||||
if src[j] == '"':
|
||||
inq = not inq
|
||||
elif src[j] == ">" and not inq:
|
||||
break
|
||||
j += 1
|
||||
j = min(j + 1, n)
|
||||
toks.append(("tag", i, j))
|
||||
name = tag_name(src[i:j])
|
||||
if name in ("script", "style") and not src[i:j].startswith("</"):
|
||||
close = re.compile(r"</%s\s*>" % name, re.I).search(src, j)
|
||||
k = close.start() if close else n
|
||||
toks.append((name, j, k))
|
||||
i = k
|
||||
else:
|
||||
i = j
|
||||
else:
|
||||
j = i
|
||||
while j < n and not src.startswith("{{", j) and not (
|
||||
src[j] == "<" and j + 1 < n and (src[j + 1].isalpha() or src[j + 1] in "/!")):
|
||||
j += 1
|
||||
toks.append(("text", i, j))
|
||||
i = j
|
||||
return toks
|
||||
|
||||
|
||||
def tag_name(tag):
|
||||
m = re.match(r"</?\s*([a-zA-Z0-9]+)", tag)
|
||||
return m.group(1).lower() if m else ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
# runs
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
|
||||
class Ctx:
|
||||
def __init__(self, prefix, bundle, stats):
|
||||
self.prefix, self.bundle, self.stats = prefix, bundle, stats
|
||||
|
||||
def key_for(self, text):
|
||||
base = f"{self.prefix}.{slugify(text)}"
|
||||
key, k = base, 2
|
||||
while key in self.bundle and self.bundle[key] != text:
|
||||
key, k = f"{base}_{k}", k + 1
|
||||
self.bundle[key] = text
|
||||
return key
|
||||
|
||||
|
||||
def edits_for_fragment(src, base, ctx, allow_attrs=True):
|
||||
"""Return [(start, end, replacement)] for one HTML fragment (a template, or a JS string's body)."""
|
||||
toks = tokenize(src)
|
||||
edits = []
|
||||
run = []
|
||||
|
||||
def flush():
|
||||
nonlocal run
|
||||
r = run
|
||||
run = []
|
||||
# trim edges: whitespace text, unbalanced/void tags
|
||||
while r:
|
||||
k, s, e = r[0]
|
||||
if k == "text" and not src[s:e].strip():
|
||||
r = r[1:]
|
||||
continue
|
||||
if k == "tag" and not balanced_open(src, r, 0):
|
||||
r = r[1:]
|
||||
continue
|
||||
break
|
||||
while r:
|
||||
k, s, e = r[-1]
|
||||
if k == "text" and not src[s:e].strip():
|
||||
r = r[:-1]
|
||||
continue
|
||||
if k == "tag" and not balanced_close(src, r, len(r) - 1):
|
||||
r = r[:-1]
|
||||
continue
|
||||
break
|
||||
# a wrapper pair around the whole run (<span class="x">...</span>) stays outside the message
|
||||
while len(r) >= 2 and r[0][0] == "tag" and r[-1][0] == "tag" and balanced_open(src, r, 0) \
|
||||
and matching_close(src, r, 0) == len(r) - 1:
|
||||
r = strip_ws(src, r[1:-1])
|
||||
if not r:
|
||||
return
|
||||
segs = [r]
|
||||
if not inline_balanced(src, r):
|
||||
segs = [[t] for t in r if t[0] != "tag"]
|
||||
for seg in segs:
|
||||
s0, e0 = seg[0][1], seg[-1][2]
|
||||
chunk = src[s0:e0]
|
||||
visible = "".join(src[s:e] for k, s, e in seg if k == "text")
|
||||
if not is_hu(visible):
|
||||
continue
|
||||
lead = len(chunk) - len(chunk.lstrip())
|
||||
trail = len(chunk) - len(chunk.rstrip())
|
||||
core = chunk.strip()
|
||||
if MARKER.match(core):
|
||||
continue
|
||||
key = ctx.key_for(core)
|
||||
ctx.stats["run"] += 1
|
||||
edits.append((base + s0 + lead, base + e0 - trail, '{{T "%s"}}' % key))
|
||||
|
||||
for tok in toks:
|
||||
kind, s, e = tok
|
||||
text = src[s:e]
|
||||
if kind == "text":
|
||||
run.append(tok)
|
||||
elif kind == "action":
|
||||
if CONTROL.match(text) or MARKER.match(text):
|
||||
flush()
|
||||
else:
|
||||
run.append(tok)
|
||||
elif kind == "tag" and tag_name(text) in INLINE and not has_hu_attr(text) and not is_button_link(src, tok):
|
||||
run.append(tok)
|
||||
else:
|
||||
flush()
|
||||
if kind == "tag" and allow_attrs:
|
||||
edits.extend(attr_edits(text, base + s, ctx))
|
||||
elif kind == "script":
|
||||
edits.extend(js_edits(text, base + s, ctx))
|
||||
flush()
|
||||
return edits
|
||||
|
||||
|
||||
def is_button_link(src, tok):
|
||||
"""An <a class="btn ..."> is a control of its own, not a word inside a sentence. Its closing </a>
|
||||
is found by walking forward; both the opening and closing tag break the run."""
|
||||
k, s, e = tok
|
||||
t = src[s:e]
|
||||
if tag_name(t) != "a":
|
||||
return False
|
||||
if t.startswith("</"):
|
||||
opener = src.rfind("<a", 0, s)
|
||||
return opener >= 0 and bool(re.search(r'class="[^"]*\bbtn\b', src[opener:src.find(">", opener) + 1]))
|
||||
return bool(re.search(r'class="[^"]*\bbtn\b', t))
|
||||
|
||||
|
||||
def has_hu_attr(tag):
|
||||
return any(is_hu(v) and a.lower() not in SKIP_ATTRS for a, v in attrs(tag))
|
||||
|
||||
|
||||
def attrs(tag):
|
||||
out = []
|
||||
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
|
||||
pos = 0
|
||||
while True:
|
||||
m = pat.search(tag, pos)
|
||||
if not m:
|
||||
break
|
||||
j = m.end()
|
||||
start = j
|
||||
while j < len(tag):
|
||||
if tag.startswith("{{", j):
|
||||
j = find_action_end(tag, j)
|
||||
continue
|
||||
if tag[j] == '"':
|
||||
break
|
||||
j += 1
|
||||
out.append((m.group(1), tag[start:j]))
|
||||
pos = j + 1
|
||||
return out
|
||||
|
||||
|
||||
def attr_edits(tag, base, ctx):
|
||||
edits = []
|
||||
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
|
||||
pos = 0
|
||||
while True:
|
||||
m = pat.search(tag, pos)
|
||||
if not m:
|
||||
break
|
||||
j = start = m.end()
|
||||
while j < len(tag):
|
||||
if tag.startswith("{{", j):
|
||||
j = find_action_end(tag, j)
|
||||
continue
|
||||
if tag[j] == '"':
|
||||
break
|
||||
j += 1
|
||||
name, val = m.group(1).lower(), tag[start:j]
|
||||
pos = j + 1
|
||||
if name in SKIP_ATTRS or not is_hu(re.sub(r"\{\{.*?\}\}", "", val)):
|
||||
continue
|
||||
if re.search(r"\{\{-?\s*(if|else|end|range|with)\b", val) or MARKER.match(val.strip()):
|
||||
continue # conditional attribute copy: left for a hand edit (reported)
|
||||
key = ctx.key_for(val)
|
||||
ctx.stats["attr"] += 1
|
||||
edits.append((base + start, base + j, '{{T "%s"}}' % key))
|
||||
return edits
|
||||
|
||||
|
||||
def js_edits(body, base, ctx):
|
||||
"""String literals inside a <script> body. Comments, regex literals and template actions are
|
||||
skipped so an apostrophe in a comment or a quote inside {{if eq .X "y"}} does not open a string."""
|
||||
edits, i, n = [], 0, len(body)
|
||||
prev = ""
|
||||
while i < n:
|
||||
c = body[i]
|
||||
if body.startswith("{{", i):
|
||||
i = find_action_end(body, i)
|
||||
continue
|
||||
if body.startswith("//", i):
|
||||
j = body.find("\n", i)
|
||||
i = n if j < 0 else j
|
||||
continue
|
||||
if body.startswith("/*", i):
|
||||
j = body.find("*/", i + 2)
|
||||
i = n if j < 0 else j + 2
|
||||
continue
|
||||
if c == "/" and prev in "(,=:[!&|?{};":
|
||||
j = i + 1
|
||||
while j < n and body[j] != "/" and body[j] != "\n":
|
||||
j += 2 if body[j] == "\\" else 1
|
||||
i = j + 1
|
||||
prev = "/"
|
||||
continue
|
||||
if c in "'\"":
|
||||
j = i + 1
|
||||
while j < n and body[j] != c and body[j] != "\n":
|
||||
if body.startswith("{{", j):
|
||||
j = find_action_end(body, j)
|
||||
continue
|
||||
j += 2 if body[j] == "\\" else 1
|
||||
content = body[i + 1:j]
|
||||
if is_hu(re.sub(r"\{\{.*?\}\}", "", content)):
|
||||
# A JS string is often a PIECE of markup: '\')">T\u00f6rl\u00e9s</button>' starts inside a
|
||||
# tag. Everything up to the first '>' that precedes any '<' is that tag's tail, not text.
|
||||
skip = 0
|
||||
gt, lt = content.find(">"), content.find("<")
|
||||
if gt >= 0 and (lt < 0 or gt < lt):
|
||||
skip = gt + 1
|
||||
sub = edits_for_fragment(content[skip:], base + i + 1 + skip, ctx, allow_attrs=False)
|
||||
ctx.stats["js"] += len(sub)
|
||||
edits.extend(sub)
|
||||
i = j + 1
|
||||
prev = "s"
|
||||
continue
|
||||
if not c.isspace():
|
||||
prev = c
|
||||
i += 1
|
||||
return edits
|
||||
|
||||
|
||||
def strip_ws(src, r):
|
||||
while r and r[0][0] == "text" and not src[r[0][1]:r[0][2]].strip():
|
||||
r = r[1:]
|
||||
while r and r[-1][0] == "text" and not src[r[-1][1]:r[-1][2]].strip():
|
||||
r = r[:-1]
|
||||
return r
|
||||
|
||||
|
||||
def matching_close(src, run, idx):
|
||||
name = tag_name(src[run[idx][1]:run[idx][2]])
|
||||
depth = 0
|
||||
for j in range(idx, len(run)):
|
||||
k2, s2, e2 = run[j]
|
||||
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
||||
continue
|
||||
depth += -1 if src[s2:e2].startswith("</") else 1
|
||||
if depth == 0:
|
||||
return j
|
||||
return -1
|
||||
|
||||
|
||||
def balanced_open(src, run, idx):
|
||||
k, s, e = run[idx]
|
||||
t = src[s:e]
|
||||
name = tag_name(t)
|
||||
if name not in INLINE or name in VOID or t.startswith("</") or t.endswith("/>"):
|
||||
return False
|
||||
depth = 0
|
||||
for k2, s2, e2 in run[idx:]:
|
||||
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
||||
continue
|
||||
depth += -1 if src[s2:e2].startswith("</") else 1
|
||||
if depth == 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def balanced_close(src, run, idx):
|
||||
k, s, e = run[idx]
|
||||
t = src[s:e]
|
||||
name = tag_name(t)
|
||||
if name not in INLINE or name in VOID or not t.startswith("</"):
|
||||
return False
|
||||
depth = 0
|
||||
for k2, s2, e2 in reversed(run[:idx + 1]):
|
||||
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
||||
continue
|
||||
depth += 1 if src[s2:e2].startswith("</") else -1
|
||||
if depth == 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def inline_balanced(src, run):
|
||||
depth = collections.Counter()
|
||||
for k, s, e in run:
|
||||
if k != "tag":
|
||||
continue
|
||||
t = src[s:e]
|
||||
name = tag_name(t)
|
||||
if name in VOID or t.endswith("/>"):
|
||||
continue
|
||||
depth[name] += -1 if t.startswith("</") else 1
|
||||
if depth[name] < 0:
|
||||
return False
|
||||
return all(v == 0 for v in depth.values())
|
||||
|
||||
|
||||
def main(argv):
|
||||
if not argv:
|
||||
print(__doc__)
|
||||
return 2
|
||||
bundle = collections.OrderedDict()
|
||||
if os.path.exists(HU_JSON):
|
||||
with open(HU_JSON, encoding="utf-8") as f:
|
||||
bundle.update(json.load(f, object_pairs_hook=collections.OrderedDict))
|
||||
for path in argv:
|
||||
src = open(path, encoding="utf-8").read()
|
||||
prefix = os.path.splitext(os.path.basename(path))[0]
|
||||
stats = collections.Counter()
|
||||
ctx = Ctx(prefix, bundle, stats)
|
||||
edits = sorted(edits_for_fragment(src, 0, ctx), key=lambda e: e[0])
|
||||
out, last = [], 0
|
||||
for s, e, rep in edits:
|
||||
if s < last:
|
||||
raise SystemExit(f"{path}: overlapping edit at {s}")
|
||||
out.append(src[last:s])
|
||||
out.append(rep)
|
||||
last = e
|
||||
out.append(src[last:])
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write("".join(out))
|
||||
print(f"{path}: {stats['run']} runs, {stats['attr']} attributes, {stats['js']} of the runs inside JS strings")
|
||||
os.makedirs(os.path.dirname(HU_JSON), exist_ok=True)
|
||||
with open(HU_JSON, "w", encoding="utf-8") as f:
|
||||
json.dump(bundle, f, ensure_ascii=False, indent=2)
|
||||
f.write("\n")
|
||||
print(f"{HU_JSON}: {len(bundle)} keys")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,137 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""i18n_missing_gate.py -- the message bundles are complete, consistent and in the product's voice.
|
||||
|
||||
Run from controller/: python3 scripts/i18n_missing_gate.py
|
||||
Exit 0 clean · 1 convicted · 2 inconclusive (bundles or templates not found).
|
||||
|
||||
Design: felhom.eu/documentation/architecture/10-localisation.md. The Go tests pin the runtime
|
||||
(parity, fallback, parameters, JS-context safety); this gate pins the SOURCE, before a build:
|
||||
|
||||
1. KEYS EXIST. Every `{{T "key"}}` marker in any dashboard template names a key in hu.json.
|
||||
(The loader refuses an undefined key at startup; this names the key at push time instead.)
|
||||
2. NO ORPHANS. Every en.json key translates a hu.json key (".one"/".other" count for their base).
|
||||
3. THE ENGLISH GAP ONLY SHRINKS. The number of hu keys with no English is printed and compared with
|
||||
EN_MISSING_CEILING. More than the ceiling convicts (a slice converted copy and did not translate
|
||||
it); LESS also convicts, until the ceiling is lowered to match -- a ratchet that cannot be
|
||||
loosened by forgetting it. Raising it is a decision recorded in 10-localisation.md.
|
||||
4. VOICE, per language.
|
||||
en -- second person, plain: no "please", no "kindly" (the household is not being begged).
|
||||
hu -- the product speaks in „te". Formal („ön") forms already in the converted copy are known
|
||||
debt (R-516) and COUNTED against HU_FORMAL_CEILING with the same ratchet: a new formal form
|
||||
convicts; fixing one requires lowering the ceiling. The Hungarian copy itself is not changed by
|
||||
the release that introduced this gate (parity rule), which is why this is a ratchet and not a ban.
|
||||
|
||||
Hungarian is matched by ASCII-folded stems (the workspace rule for Hungarian search), with a
|
||||
positive and a negative control printed on every run.
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
CTRL = os.path.dirname(HERE)
|
||||
LOCALES = os.path.join(CTRL, "internal", "i18n", "locales")
|
||||
TEMPLATE_ROOTS = [os.path.join(CTRL, "internal", "web", "templates")]
|
||||
MARKER = re.compile(r'\{\{\s*T\s+"([A-Za-z0-9_.\-]+)"\s*\}\}')
|
||||
|
||||
EN_MISSING_CEILING = 0
|
||||
HU_FORMAL_CEILING = 6
|
||||
|
||||
EN_FORBIDDEN = re.compile(r"\b(please|kindly)\b", re.I)
|
||||
|
||||
# Formal-address stems, ASCII-folded. Each is a verb form or phrase that only the „ön" register uses.
|
||||
HU_FORMAL = [
|
||||
"olvassa be", "kerjuk", "vegye fel", "ujratelepiti az", "importalnia", "biztosan kikapcsolja",
|
||||
"ellenorizze", "csatlakoztassa", "adjon hozza", "nezze meg", "kattintson", "valassza ki",
|
||||
]
|
||||
|
||||
|
||||
def fold(s):
|
||||
s = unicodedata.normalize("NFKD", s)
|
||||
return "".join(c for c in s if not unicodedata.combining(c)).lower()
|
||||
|
||||
|
||||
def load(lang):
|
||||
p = os.path.join(LOCALES, lang + ".json")
|
||||
if not os.path.exists(p):
|
||||
return None
|
||||
return json.load(io.open(p, encoding="utf-8"))
|
||||
|
||||
|
||||
def templates():
|
||||
out = []
|
||||
for root in TEMPLATE_ROOTS:
|
||||
for dirpath, _dirs, names in os.walk(root):
|
||||
for n in sorted(names):
|
||||
if n.endswith(".html"):
|
||||
out.append(os.path.join(dirpath, n))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def main():
|
||||
hu, en = load("hu"), load("en")
|
||||
tpls = templates()
|
||||
if hu is None or en is None or not tpls:
|
||||
print("i18n gate INCONCLUSIVE: bundles or templates not found under %s" % CTRL)
|
||||
return 2
|
||||
|
||||
# Controls for the ASCII-folded matcher.
|
||||
if "kerjuk" not in fold("Kérjük, vegye fel"):
|
||||
print("i18n gate INCONCLUSIVE: positive control failed (folding does not strip accents)")
|
||||
return 2
|
||||
if any(stem in fold("Csatlakoztasd újra a meghajtót") for stem in HU_FORMAL):
|
||||
print("i18n gate INCONCLUSIVE: negative control failed (an informal te-form sentence matched a formal stem)")
|
||||
return 2
|
||||
|
||||
problems = []
|
||||
used = 0
|
||||
for path in tpls:
|
||||
rel = os.path.relpath(path, CTRL)
|
||||
for i, line in enumerate(io.open(path, encoding="utf-8"), 1):
|
||||
for m in MARKER.finditer(line):
|
||||
used += 1
|
||||
if m.group(1) not in hu:
|
||||
problems.append("%s:%d marker key %r is not in hu.json" % (rel, i, m.group(1)))
|
||||
|
||||
for k in en:
|
||||
base = re.sub(r"\.(one|other)$", "", k)
|
||||
if k not in hu and base not in hu:
|
||||
problems.append("en.json key %r has no Hungarian source" % k)
|
||||
if EN_FORBIDDEN.search(en[k]):
|
||||
problems.append("en.json %r says %r -- plain second person, no pleading" % (k, EN_FORBIDDEN.search(en[k]).group(0)))
|
||||
|
||||
missing = [k for k in hu if k not in en and (k + ".other") not in en]
|
||||
formal = [(k, stem) for k in hu for stem in HU_FORMAL if stem in fold(hu[k])]
|
||||
|
||||
print("i18n: %d markers in %d templates; hu %d keys, en %d keys" % (used, len(tpls), len(hu), len(en)))
|
||||
print("i18n: English missing %d (ceiling %d); Hungarian formal forms %d (ceiling %d, R-516)"
|
||||
% (len(missing), EN_MISSING_CEILING, len(formal), HU_FORMAL_CEILING))
|
||||
for k, stem in formal:
|
||||
print(" formal: %s (%s)" % (k, stem))
|
||||
|
||||
if len(missing) > EN_MISSING_CEILING:
|
||||
problems.append("%d Hungarian keys have no English, ceiling is %d: %s"
|
||||
% (len(missing), EN_MISSING_CEILING, ", ".join(missing[:10])))
|
||||
elif len(missing) < EN_MISSING_CEILING:
|
||||
problems.append("English gap shrank to %d -- lower EN_MISSING_CEILING from %d to match (the ratchet)"
|
||||
% (len(missing), EN_MISSING_CEILING))
|
||||
if len(formal) > HU_FORMAL_CEILING:
|
||||
problems.append("%d formal (on-form) forms in hu.json, ceiling %d -- the product speaks in the te-form" % (len(formal), HU_FORMAL_CEILING))
|
||||
elif len(formal) < HU_FORMAL_CEILING:
|
||||
problems.append("formal forms fell to %d -- lower HU_FORMAL_CEILING from %d to match (the ratchet)"
|
||||
% (len(formal), HU_FORMAL_CEILING))
|
||||
|
||||
if problems:
|
||||
for p in problems:
|
||||
print(" " + p)
|
||||
print("I18N GATE FAILED: %d problem(s)" % len(problems))
|
||||
return 1
|
||||
print("i18n gate OK")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -33,7 +33,7 @@ SIGNATURE = {
|
||||
"ˇ": "ˇ (caron — cp1250 round-trip)",
|
||||
}
|
||||
|
||||
EXTS = (".html", ".css", ".js", ".go")
|
||||
EXTS = (".html", ".css", ".js", ".go", ".json") # .json: the i18n bundles (v0.247.0) are copy
|
||||
|
||||
|
||||
def files():
|
||||
|
||||
@@ -11,6 +11,9 @@ Exit 1 if any native confirm(/prompt(/alert( call remains in the templates.
|
||||
"""
|
||||
import io, os, re, sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import i18n_bundle # noqa: E402
|
||||
|
||||
ROOTS = [
|
||||
os.path.join("internal", "web", "templates"),
|
||||
os.path.join("internal", "setup", "templates"),
|
||||
@@ -41,10 +44,14 @@ def main():
|
||||
for root in ROOTS:
|
||||
for path in _html_files(root):
|
||||
fn = os.path.relpath(path, root)
|
||||
for lineno, line in enumerate(io.open(path, encoding="utf-8"), 1):
|
||||
if NATIVE.search(line):
|
||||
total += 1
|
||||
print("%s:%d %s" % (fn, lineno, line.strip()[:120].encode('ascii', 'backslashreplace').decode()))
|
||||
# v0.247.0: judge each language's expansion -- JS copy lives in the i18n bundle now.
|
||||
seen = set()
|
||||
for lang in i18n_bundle.LANGS:
|
||||
for lineno, line in enumerate(i18n_bundle.read_template(path, lang).split("\n"), 1):
|
||||
if NATIVE.search(line) and (lineno, line) not in seen:
|
||||
seen.add((lineno, line))
|
||||
total += 1
|
||||
print("%s:%d [%s] %s" % (fn, lineno, lang, line.strip()[:120].encode('ascii', 'backslashreplace').decode()))
|
||||
if total:
|
||||
print("NATIVE CONFIRM GATE FAILED: %d native confirm()/prompt()/alert() call(s) remain" % total)
|
||||
sys.exit(1)
|
||||
|
||||
@@ -27,6 +27,9 @@ import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import i18n_bundle # noqa: E402
|
||||
|
||||
_HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
TEMPLATES = os.path.join(_HERE, "..", "internal", "web", "templates")
|
||||
|
||||
@@ -119,7 +122,11 @@ def scan():
|
||||
sources.append((os.path.basename(gp), gp, GO_COMMENT))
|
||||
files = files + [os.path.basename(gp)]
|
||||
for name, path, stripper in sources:
|
||||
text = stripper.sub("", open(path, encoding="utf-8").read())
|
||||
# v0.247.0: a template's copy lives in the i18n bundle; judge the page as it renders in
|
||||
# Hungarian (the stems are Hungarian). English retrieval claims are not scanned yet -- R-row
|
||||
# in 10-localisation.md; the English bundle covers three pages today.
|
||||
raw = open(path, encoding="utf-8").read()
|
||||
text = stripper.sub("", i18n_bundle.expand(raw, "hu") if path.endswith(".html") else raw)
|
||||
for stem in STEMS:
|
||||
for m in re.finditer(re.escape(stem) + r"[a-záéíóöőúüű]*", text):
|
||||
line = text[: m.start()].count("\n") + 1
|
||||
|
||||
@@ -39,6 +39,9 @@ import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import i18n_bundle # noqa: E402
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
CTRL = os.path.dirname(HERE)
|
||||
TPL = os.path.join(CTRL, "internal", "web", "templates")
|
||||
@@ -71,8 +74,18 @@ ALLOWLIST = {
|
||||
|
||||
|
||||
def check(path):
|
||||
# v0.247.0: bundle values carry template actions ({{.RecoveryAbandonDate}}), so a secret-named
|
||||
# expression can arrive through a translation. Judge every language's expansion; dedupe.
|
||||
out = []
|
||||
for lang in i18n_bundle.LANGS:
|
||||
for c in _check_src(i18n_bundle.read_template(path, lang)):
|
||||
if c not in out:
|
||||
out.append(c)
|
||||
return out
|
||||
|
||||
|
||||
def _check_src(src):
|
||||
convictions = []
|
||||
src = open(path, encoding="utf-8").read()
|
||||
for m in ACTION.finditer(src):
|
||||
expr = m.group(1).strip()
|
||||
# A TEMPLATE comment `{{/* ... */}}` is stripped by html/template and never reaches the
|
||||
|
||||
@@ -44,6 +44,7 @@ COVERS = {
|
||||
"debug-routes": "a live dispatcher case commented out - the button survives, the handler dies",
|
||||
"golden-notice": "R-410 in the other direction: an empty dir must not count as a bake",
|
||||
"minagent-header": "the word MinAgent in prose / a code span, not the header line (test_minagent_header_gate.py)",
|
||||
"i18n": "an undefined marker key, a pleading English value, a shrinking/growing gap (v0.247.0)",
|
||||
}
|
||||
|
||||
fails = []
|
||||
@@ -133,6 +134,38 @@ def _comment_out_the_partial(src):
|
||||
swapped("app-row-dedup/comment", "app_row_dedup_gate.py",
|
||||
os.path.join(TPL, "dashboard.html"), _comment_out_the_partial)
|
||||
|
||||
# --- i18n (v0.247.0): copy moved OUT of the templates into the bundles. Two families of decoy: ---
|
||||
# --- the new gate itself, and every copy gate that must still see copy that now lives there. ---
|
||||
LOC = os.path.join(CTRL, "internal", "i18n", "locales")
|
||||
EN_JSON, HU_JSON = os.path.join(LOC, "en.json"), os.path.join(LOC, "hu.json")
|
||||
LAUNCHER = os.path.join(TPL, "launcher.html")
|
||||
|
||||
|
||||
def _one(old, new):
|
||||
def t(src):
|
||||
assert src.count(old) == 1, "decoy anchor %r not found exactly once — the decoy cannot be built" % old
|
||||
return src.replace(old, new)
|
||||
return t
|
||||
|
||||
|
||||
swapped("i18n/undefined-key", "i18n_missing_gate.py", LAUNCHER,
|
||||
_one('{{T "launcher.link_masolasa"}}', '{{T "launcher.no_such_key"}}'))
|
||||
swapped("i18n/pleading", "i18n_missing_gate.py", EN_JSON,
|
||||
_one('"launcher.link_masolasa": "Copy link"', '"launcher.link_masolasa": "Please copy the link"'))
|
||||
swapped("i18n/gap-grew", "i18n_missing_gate.py", EN_JSON,
|
||||
_one(' "launcher.link_masolasa": "Copy link",\n', ''))
|
||||
swapped("i18n/new-formal-form", "i18n_missing_gate.py", HU_JSON,
|
||||
_one('"launcher.link_masolasa": "Link másolása"', '"launcher.link_masolasa": "Kattintson a link másolásához"'))
|
||||
# Scope: the copy gates read the page AS RENDERED, so copy that lives in a bundle is still judged.
|
||||
swapped("emoji/bundle", "emoji_gate.py", EN_JSON,
|
||||
_one('"launcher.link_masolasa": "Copy link"', '"launcher.link_masolasa": "Copy link \U0001F600"'))
|
||||
swapped("native-confirm/bundle", "native_confirm_gate.py", EN_JSON,
|
||||
_one('"launcher.masolva": "Copied"', '"launcher.masolva": "Copied\u0027+confirm(\u0027x\u0027)+\u0027"'))
|
||||
swapped("retrieval-promise/bundle", "retrieval_promise_gate.py", HU_JSON,
|
||||
_one('"launcher.link_masolasa": "Link másolása"', '"launcher.link_masolasa": "A mentésed bármikor visszaállítható."'))
|
||||
swapped("secret-markup/bundle", "secret_in_markup_gate.py", EN_JSON,
|
||||
_one('"launcher.link_masolasa": "Copy link"', '"launcher.link_masolasa": "Copy {{.RetrievalPassword}}"'))
|
||||
|
||||
# --- golden-notice: R-410's decoy, in the other direction. It is ADVISORY, so rc is never the ---
|
||||
# --- question — what it COUNTED is. ---
|
||||
ran += 1
|
||||
|
||||
Reference in New Issue
Block a user