v0.247.0: i18n spike — the dashboard can speak English, Hungarian byte-identical
gates / gates (push) Successful in 19s
gates / gates (push) Successful in 19s
Message bundles (internal/i18n) expanded into templates before parsing, one template set per language. Launcher, /backups, /apps/<slug> and the layout converted; household language setting, POST /settings/language, ?lang= override, report field. Parity test against fixtures captured from unconverted templates; copy gates read templates expanded; new i18n_missing_gate. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,445 @@
|
||||
#!/usr/bin/env python3
|
||||
"""i18n_extract.py -- move the Hungarian copy of a dashboard template into the message bundle.
|
||||
|
||||
python3 scripts/i18n_extract.py internal/web/templates/launcher.html [more.html ...]
|
||||
|
||||
For each file: every Hungarian text run, attribute value and inline-JS string is replaced IN PLACE by
|
||||
a `{{T "<file>.<slug>"}}` marker, and the exact text it replaced is written to
|
||||
internal/i18n/locales/hu.json under that key. Nothing else in the file moves -- surrounding
|
||||
whitespace, tags and template actions stay where they were.
|
||||
|
||||
Why that is safe (and what proves it): internal/i18n expands every marker back to the bundle text
|
||||
BEFORE html/template parses the file (see internal/i18n/i18n.go). The Hungarian template set is
|
||||
therefore parsed from the original bytes, in the original escaping contexts. The parity test
|
||||
(internal/web/i18n_parity_test.go) renders the converted pages and compares them byte-for-byte with
|
||||
fixtures captured from the unconverted templates -- this tool is not trusted, it is checked.
|
||||
|
||||
A "run" is the unit a translator needs: text, value actions ({{.Name}}) and inline tags
|
||||
(<strong>, <a>, <br>, ...) up to the next block tag or control action ({{if}}, {{else}}, {{end}},
|
||||
{{range}}, comments). Word order therefore stays inside one message. English values of JS-context
|
||||
keys must not contain a bare quote, backslash or newline -- pinned by TestI18nJSContextValuesAreSafe.
|
||||
|
||||
Idempotent: a file already carrying markers keeps them; a run that already is a marker is skipped.
|
||||
Source is ASCII-only; Hungarian letters are written as \\u escapes.
|
||||
"""
|
||||
import collections
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
CTRL = os.path.dirname(HERE)
|
||||
HU_JSON = os.path.join(CTRL, "internal", "i18n", "locales", "hu.json")
|
||||
|
||||
HU_LETTERS = set("\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170")
|
||||
# ASCII-only Hungarian: a word with no accented letter is invisible to the letter test. Case-insensitive.
|
||||
# The last line was added after a review of what the first pass left behind (Megszakadt, Elavult,
|
||||
# Csak x86, Pi kompatibilis, Rendszermonitor, Most nem, sikertelen, ismeretlen).
|
||||
ASCII_HU = re.compile(r"\b(Fut|Nincs|Igen|Nem|Hiba|Mentve|Rendben|Adatok|Napi|Heti|Havi|Mindig|Soha|Tegnap|"
|
||||
r"Perc|Letiltva|Bekapcsolva|Kikapcsolva|Folyamatban|Sikertelen|Sikeres|Tartalom|"
|
||||
r"Kapcsolat|Vissza|Megosztva|Kezdeti|Rendszer|Csatlakozva|Kulcs|Megnyit|Elrejt|"
|
||||
r"Ismeretlen|Szabad|Foglalt|Kijelentkezes|Napok|nap|perc|ora|"
|
||||
r"Megszakadt|Elavult|Csak|kompatibilis|Rendszermonitor|Most nem)\b", re.I)
|
||||
|
||||
INLINE = {"strong", "em", "b", "i", "br", "code", "small", "a", "kbd", "u"} # NOT span: it is layout here
|
||||
VOID = {"br"}
|
||||
SKIP_ATTRS = {"class", "id", "href", "src", "style", "name", "type", "for", "action", "method", "rel",
|
||||
"target", "width", "height", "autocomplete", "role", "hidden", "onclick", "onerror",
|
||||
"onchange", "onsubmit", "oninput", "aria-controls", "aria-expanded", "content", "charset",
|
||||
"lang", "value"}
|
||||
CONTROL = re.compile(r"^\{\{-?\s*(/\*|if\b|else\b|end\b|range\b|with\b|define\b|block\b|template\b|break\b|continue\b|\$\w+\s*:?=)")
|
||||
MARKER = re.compile(r'^\{\{\s*T\s+"[^"]+"\s*\}\}$')
|
||||
|
||||
|
||||
def is_hu(s):
|
||||
return any(c in HU_LETTERS for c in s) or bool(ASCII_HU.search(s))
|
||||
|
||||
|
||||
def slugify(text):
|
||||
t = re.sub(r"\{\{.*?\}\}", " ", text)
|
||||
t = re.sub(r"<[^>]*>", " ", t)
|
||||
t = re.sub(r"&[a-z]+;|&#\d+;", " ", t)
|
||||
t = unicodedata.normalize("NFKD", t)
|
||||
t = "".join(c for c in t if not unicodedata.combining(c)).lower()
|
||||
words = re.findall(r"[a-z0-9]+", t)
|
||||
slug = "_".join(words[:5])[:40].strip("_")
|
||||
return slug or "text"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
# tokenizer: (kind, start, end) kind in action | tag | comment | text | script | style
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
|
||||
def find_action_end(src, i):
|
||||
if src.startswith("{{/*", i) or src.startswith("{{- /*", i):
|
||||
j = src.find("*/", i)
|
||||
j = src.find("}}", j)
|
||||
else:
|
||||
j = src.find("}}", i + 2)
|
||||
return len(src) if j < 0 else j + 2
|
||||
|
||||
|
||||
def tokenize(src):
|
||||
toks, i, n = [], 0, len(src)
|
||||
while i < n:
|
||||
if src.startswith("{{", i):
|
||||
j = find_action_end(src, i)
|
||||
toks.append(("action", i, j))
|
||||
i = j
|
||||
elif src.startswith("<!--", i):
|
||||
j = src.find("-->", i)
|
||||
j = n if j < 0 else j + 3
|
||||
toks.append(("comment", i, j))
|
||||
i = j
|
||||
elif src[i] == "<" and i + 1 < n and (src[i + 1].isalpha() or src[i + 1] in "/!"):
|
||||
j, inq = i + 1, False
|
||||
while j < n:
|
||||
if src.startswith("{{", j):
|
||||
j = find_action_end(src, j)
|
||||
continue
|
||||
if src[j] == '"':
|
||||
inq = not inq
|
||||
elif src[j] == ">" and not inq:
|
||||
break
|
||||
j += 1
|
||||
j = min(j + 1, n)
|
||||
toks.append(("tag", i, j))
|
||||
name = tag_name(src[i:j])
|
||||
if name in ("script", "style") and not src[i:j].startswith("</"):
|
||||
close = re.compile(r"</%s\s*>" % name, re.I).search(src, j)
|
||||
k = close.start() if close else n
|
||||
toks.append((name, j, k))
|
||||
i = k
|
||||
else:
|
||||
i = j
|
||||
else:
|
||||
j = i
|
||||
while j < n and not src.startswith("{{", j) and not (
|
||||
src[j] == "<" and j + 1 < n and (src[j + 1].isalpha() or src[j + 1] in "/!")):
|
||||
j += 1
|
||||
toks.append(("text", i, j))
|
||||
i = j
|
||||
return toks
|
||||
|
||||
|
||||
def tag_name(tag):
|
||||
m = re.match(r"</?\s*([a-zA-Z0-9]+)", tag)
|
||||
return m.group(1).lower() if m else ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
# runs
|
||||
# ---------------------------------------------------------------------------------------------------
|
||||
|
||||
class Ctx:
|
||||
def __init__(self, prefix, bundle, stats):
|
||||
self.prefix, self.bundle, self.stats = prefix, bundle, stats
|
||||
|
||||
def key_for(self, text):
|
||||
base = f"{self.prefix}.{slugify(text)}"
|
||||
key, k = base, 2
|
||||
while key in self.bundle and self.bundle[key] != text:
|
||||
key, k = f"{base}_{k}", k + 1
|
||||
self.bundle[key] = text
|
||||
return key
|
||||
|
||||
|
||||
def edits_for_fragment(src, base, ctx, allow_attrs=True):
|
||||
"""Return [(start, end, replacement)] for one HTML fragment (a template, or a JS string's body)."""
|
||||
toks = tokenize(src)
|
||||
edits = []
|
||||
run = []
|
||||
|
||||
def flush():
|
||||
nonlocal run
|
||||
r = run
|
||||
run = []
|
||||
# trim edges: whitespace text, unbalanced/void tags
|
||||
while r:
|
||||
k, s, e = r[0]
|
||||
if k == "text" and not src[s:e].strip():
|
||||
r = r[1:]
|
||||
continue
|
||||
if k == "tag" and not balanced_open(src, r, 0):
|
||||
r = r[1:]
|
||||
continue
|
||||
break
|
||||
while r:
|
||||
k, s, e = r[-1]
|
||||
if k == "text" and not src[s:e].strip():
|
||||
r = r[:-1]
|
||||
continue
|
||||
if k == "tag" and not balanced_close(src, r, len(r) - 1):
|
||||
r = r[:-1]
|
||||
continue
|
||||
break
|
||||
# a wrapper pair around the whole run (<span class="x">...</span>) stays outside the message
|
||||
while len(r) >= 2 and r[0][0] == "tag" and r[-1][0] == "tag" and balanced_open(src, r, 0) \
|
||||
and matching_close(src, r, 0) == len(r) - 1:
|
||||
r = strip_ws(src, r[1:-1])
|
||||
if not r:
|
||||
return
|
||||
segs = [r]
|
||||
if not inline_balanced(src, r):
|
||||
segs = [[t] for t in r if t[0] != "tag"]
|
||||
for seg in segs:
|
||||
s0, e0 = seg[0][1], seg[-1][2]
|
||||
chunk = src[s0:e0]
|
||||
visible = "".join(src[s:e] for k, s, e in seg if k == "text")
|
||||
if not is_hu(visible):
|
||||
continue
|
||||
lead = len(chunk) - len(chunk.lstrip())
|
||||
trail = len(chunk) - len(chunk.rstrip())
|
||||
core = chunk.strip()
|
||||
if MARKER.match(core):
|
||||
continue
|
||||
key = ctx.key_for(core)
|
||||
ctx.stats["run"] += 1
|
||||
edits.append((base + s0 + lead, base + e0 - trail, '{{T "%s"}}' % key))
|
||||
|
||||
for tok in toks:
|
||||
kind, s, e = tok
|
||||
text = src[s:e]
|
||||
if kind == "text":
|
||||
run.append(tok)
|
||||
elif kind == "action":
|
||||
if CONTROL.match(text) or MARKER.match(text):
|
||||
flush()
|
||||
else:
|
||||
run.append(tok)
|
||||
elif kind == "tag" and tag_name(text) in INLINE and not has_hu_attr(text) and not is_button_link(src, tok):
|
||||
run.append(tok)
|
||||
else:
|
||||
flush()
|
||||
if kind == "tag" and allow_attrs:
|
||||
edits.extend(attr_edits(text, base + s, ctx))
|
||||
elif kind == "script":
|
||||
edits.extend(js_edits(text, base + s, ctx))
|
||||
flush()
|
||||
return edits
|
||||
|
||||
|
||||
def is_button_link(src, tok):
|
||||
"""An <a class="btn ..."> is a control of its own, not a word inside a sentence. Its closing </a>
|
||||
is found by walking forward; both the opening and closing tag break the run."""
|
||||
k, s, e = tok
|
||||
t = src[s:e]
|
||||
if tag_name(t) != "a":
|
||||
return False
|
||||
if t.startswith("</"):
|
||||
opener = src.rfind("<a", 0, s)
|
||||
return opener >= 0 and bool(re.search(r'class="[^"]*\bbtn\b', src[opener:src.find(">", opener) + 1]))
|
||||
return bool(re.search(r'class="[^"]*\bbtn\b', t))
|
||||
|
||||
|
||||
def has_hu_attr(tag):
|
||||
return any(is_hu(v) and a.lower() not in SKIP_ATTRS for a, v in attrs(tag))
|
||||
|
||||
|
||||
def attrs(tag):
|
||||
out = []
|
||||
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
|
||||
pos = 0
|
||||
while True:
|
||||
m = pat.search(tag, pos)
|
||||
if not m:
|
||||
break
|
||||
j = m.end()
|
||||
start = j
|
||||
while j < len(tag):
|
||||
if tag.startswith("{{", j):
|
||||
j = find_action_end(tag, j)
|
||||
continue
|
||||
if tag[j] == '"':
|
||||
break
|
||||
j += 1
|
||||
out.append((m.group(1), tag[start:j]))
|
||||
pos = j + 1
|
||||
return out
|
||||
|
||||
|
||||
def attr_edits(tag, base, ctx):
|
||||
edits = []
|
||||
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
|
||||
pos = 0
|
||||
while True:
|
||||
m = pat.search(tag, pos)
|
||||
if not m:
|
||||
break
|
||||
j = start = m.end()
|
||||
while j < len(tag):
|
||||
if tag.startswith("{{", j):
|
||||
j = find_action_end(tag, j)
|
||||
continue
|
||||
if tag[j] == '"':
|
||||
break
|
||||
j += 1
|
||||
name, val = m.group(1).lower(), tag[start:j]
|
||||
pos = j + 1
|
||||
if name in SKIP_ATTRS or not is_hu(re.sub(r"\{\{.*?\}\}", "", val)):
|
||||
continue
|
||||
if re.search(r"\{\{-?\s*(if|else|end|range|with)\b", val) or MARKER.match(val.strip()):
|
||||
continue # conditional attribute copy: left for a hand edit (reported)
|
||||
key = ctx.key_for(val)
|
||||
ctx.stats["attr"] += 1
|
||||
edits.append((base + start, base + j, '{{T "%s"}}' % key))
|
||||
return edits
|
||||
|
||||
|
||||
def js_edits(body, base, ctx):
|
||||
"""String literals inside a <script> body. Comments, regex literals and template actions are
|
||||
skipped so an apostrophe in a comment or a quote inside {{if eq .X "y"}} does not open a string."""
|
||||
edits, i, n = [], 0, len(body)
|
||||
prev = ""
|
||||
while i < n:
|
||||
c = body[i]
|
||||
if body.startswith("{{", i):
|
||||
i = find_action_end(body, i)
|
||||
continue
|
||||
if body.startswith("//", i):
|
||||
j = body.find("\n", i)
|
||||
i = n if j < 0 else j
|
||||
continue
|
||||
if body.startswith("/*", i):
|
||||
j = body.find("*/", i + 2)
|
||||
i = n if j < 0 else j + 2
|
||||
continue
|
||||
if c == "/" and prev in "(,=:[!&|?{};":
|
||||
j = i + 1
|
||||
while j < n and body[j] != "/" and body[j] != "\n":
|
||||
j += 2 if body[j] == "\\" else 1
|
||||
i = j + 1
|
||||
prev = "/"
|
||||
continue
|
||||
if c in "'\"":
|
||||
j = i + 1
|
||||
while j < n and body[j] != c and body[j] != "\n":
|
||||
if body.startswith("{{", j):
|
||||
j = find_action_end(body, j)
|
||||
continue
|
||||
j += 2 if body[j] == "\\" else 1
|
||||
content = body[i + 1:j]
|
||||
if is_hu(re.sub(r"\{\{.*?\}\}", "", content)):
|
||||
# A JS string is often a PIECE of markup: '\')">T\u00f6rl\u00e9s</button>' starts inside a
|
||||
# tag. Everything up to the first '>' that precedes any '<' is that tag's tail, not text.
|
||||
skip = 0
|
||||
gt, lt = content.find(">"), content.find("<")
|
||||
if gt >= 0 and (lt < 0 or gt < lt):
|
||||
skip = gt + 1
|
||||
sub = edits_for_fragment(content[skip:], base + i + 1 + skip, ctx, allow_attrs=False)
|
||||
ctx.stats["js"] += len(sub)
|
||||
edits.extend(sub)
|
||||
i = j + 1
|
||||
prev = "s"
|
||||
continue
|
||||
if not c.isspace():
|
||||
prev = c
|
||||
i += 1
|
||||
return edits
|
||||
|
||||
|
||||
def strip_ws(src, r):
|
||||
while r and r[0][0] == "text" and not src[r[0][1]:r[0][2]].strip():
|
||||
r = r[1:]
|
||||
while r and r[-1][0] == "text" and not src[r[-1][1]:r[-1][2]].strip():
|
||||
r = r[:-1]
|
||||
return r
|
||||
|
||||
|
||||
def matching_close(src, run, idx):
|
||||
name = tag_name(src[run[idx][1]:run[idx][2]])
|
||||
depth = 0
|
||||
for j in range(idx, len(run)):
|
||||
k2, s2, e2 = run[j]
|
||||
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
||||
continue
|
||||
depth += -1 if src[s2:e2].startswith("</") else 1
|
||||
if depth == 0:
|
||||
return j
|
||||
return -1
|
||||
|
||||
|
||||
def balanced_open(src, run, idx):
|
||||
k, s, e = run[idx]
|
||||
t = src[s:e]
|
||||
name = tag_name(t)
|
||||
if name not in INLINE or name in VOID or t.startswith("</") or t.endswith("/>"):
|
||||
return False
|
||||
depth = 0
|
||||
for k2, s2, e2 in run[idx:]:
|
||||
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
||||
continue
|
||||
depth += -1 if src[s2:e2].startswith("</") else 1
|
||||
if depth == 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def balanced_close(src, run, idx):
|
||||
k, s, e = run[idx]
|
||||
t = src[s:e]
|
||||
name = tag_name(t)
|
||||
if name not in INLINE or name in VOID or not t.startswith("</"):
|
||||
return False
|
||||
depth = 0
|
||||
for k2, s2, e2 in reversed(run[:idx + 1]):
|
||||
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
||||
continue
|
||||
depth += 1 if src[s2:e2].startswith("</") else -1
|
||||
if depth == 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def inline_balanced(src, run):
|
||||
depth = collections.Counter()
|
||||
for k, s, e in run:
|
||||
if k != "tag":
|
||||
continue
|
||||
t = src[s:e]
|
||||
name = tag_name(t)
|
||||
if name in VOID or t.endswith("/>"):
|
||||
continue
|
||||
depth[name] += -1 if t.startswith("</") else 1
|
||||
if depth[name] < 0:
|
||||
return False
|
||||
return all(v == 0 for v in depth.values())
|
||||
|
||||
|
||||
def main(argv):
|
||||
if not argv:
|
||||
print(__doc__)
|
||||
return 2
|
||||
bundle = collections.OrderedDict()
|
||||
if os.path.exists(HU_JSON):
|
||||
with open(HU_JSON, encoding="utf-8") as f:
|
||||
bundle.update(json.load(f, object_pairs_hook=collections.OrderedDict))
|
||||
for path in argv:
|
||||
src = open(path, encoding="utf-8").read()
|
||||
prefix = os.path.splitext(os.path.basename(path))[0]
|
||||
stats = collections.Counter()
|
||||
ctx = Ctx(prefix, bundle, stats)
|
||||
edits = sorted(edits_for_fragment(src, 0, ctx), key=lambda e: e[0])
|
||||
out, last = [], 0
|
||||
for s, e, rep in edits:
|
||||
if s < last:
|
||||
raise SystemExit(f"{path}: overlapping edit at {s}")
|
||||
out.append(src[last:s])
|
||||
out.append(rep)
|
||||
last = e
|
||||
out.append(src[last:])
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write("".join(out))
|
||||
print(f"{path}: {stats['run']} runs, {stats['attr']} attributes, {stats['js']} of the runs inside JS strings")
|
||||
os.makedirs(os.path.dirname(HU_JSON), exist_ok=True)
|
||||
with open(HU_JSON, "w", encoding="utf-8") as f:
|
||||
json.dump(bundle, f, ensure_ascii=False, indent=2)
|
||||
f.write("\n")
|
||||
print(f"{HU_JSON}: {len(bundle)} keys")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
Reference in New Issue
Block a user