Files
felhom-controller/controller/scripts/i18n_extract.py
T
admin ff68b0b433
gates / gates (push) Successful in 20s
v0.248.0: i18n slice 1 release A — apps and settings in English, Hungarian byte-identical (R-556)
Ten pages converted against fixtures captured from unconverted templates (6be55a0); marker
coverage test; English page mask narrowed; English retrieval-promise stems; common keys;
title keys and infraMeta from the bundle; old wording tests read the Hungarian expansion.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-17 17:36:08 +02:00

456 lines
17 KiB
Python

#!/usr/bin/env python3
"""i18n_extract.py -- move the Hungarian copy of a dashboard template into the message bundle.
python3 scripts/i18n_extract.py internal/web/templates/launcher.html [more.html ...]
For each file: every Hungarian text run, attribute value and inline-JS string is replaced IN PLACE by
a `{{T "<file>.<slug>"}}` marker, and the exact text it replaced is written to
internal/i18n/locales/hu.json under that key. Nothing else in the file moves -- surrounding
whitespace, tags and template actions stay where they were.
Why that is safe (and what proves it): internal/i18n expands every marker back to the bundle text
BEFORE html/template parses the file (see internal/i18n/i18n.go). The Hungarian template set is
therefore parsed from the original bytes, in the original escaping contexts. The parity test
(internal/web/i18n_parity_test.go) renders the converted pages and compares them byte-for-byte with
fixtures captured from the unconverted templates -- this tool is not trusted, it is checked.
A "run" is the unit a translator needs: text, value actions ({{.Name}}) and inline tags
(<strong>, <a>, <br>, ...) up to the next block tag or control action ({{if}}, {{else}}, {{end}},
{{range}}, comments). Word order therefore stays inside one message. English values of JS-context
keys must not contain a bare quote, backslash or newline -- pinned by TestI18nJSContextValuesAreSafe.
Idempotent: a file already carrying markers keeps them; a run that already is a marker is skipped.
Source is ASCII-only; Hungarian letters are written as \\u escapes.
"""
import collections
import json
import os
import re
import sys
import unicodedata
HERE = os.path.dirname(os.path.abspath(__file__))
CTRL = os.path.dirname(HERE)
HU_JSON = os.path.join(CTRL, "internal", "i18n", "locales", "hu.json")
HU_LETTERS = set("\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170")
# ASCII-only Hungarian: a word with no accented letter is invisible to the letter test. Case-insensitive.
# The last line was added after a review of what the first pass left behind (Megszakadt, Elavult,
# Csak x86, Pi kompatibilis, Rendszermonitor, Most nem, sikertelen, ismeretlen).
ASCII_HU = re.compile(r"\b(Fut|Nincs|Igen|Nem|Hiba|Mentve|Rendben|Adatok|Napi|Heti|Havi|Mindig|Soha|Tegnap|"
r"Perc|Letiltva|Bekapcsolva|Kikapcsolva|Folyamatban|Sikertelen|Sikeres|Tartalom|"
r"Kapcsolat|Vissza|Megosztva|Kezdeti|Rendszer|Csatlakozva|Kulcs|Megnyit|Elrejt|"
r"Ismeretlen|Szabad|Foglalt|Kijelentkezes|Napok|nap|perc|ora|"
r"Megszakadt|Elavult|Csak|kompatibilis|Rendszermonitor|Most nem|"
# slice 1 release A review: left behind on dashboard..settings_notifications
r"Folyamat|Titkos|Processzor|Hamarosan|aldomain|adatai|Figyelem|pelda)\b", re.I)
INLINE = {"strong", "em", "b", "i", "br", "code", "small", "a", "kbd", "u"} # NOT span: it is layout here
VOID = {"br"}
SKIP_ATTRS = {"class", "id", "href", "src", "style", "name", "type", "for", "action", "method", "rel",
"target", "width", "height", "autocomplete", "role", "hidden", "onclick", "onerror",
"onchange", "onsubmit", "oninput", "aria-controls", "aria-expanded", "content", "charset",
"lang", "value"}
CONTROL = re.compile(r"^\{\{-?\s*(/\*|if\b|else\b|end\b|range\b|with\b|define\b|block\b|template\b|break\b|continue\b|\$\w+\s*:?=)")
MARKER = re.compile(r'^\{\{\s*T\s+"[^"]+"\s*\}\}$')
def is_hu(s):
return any(c in HU_LETTERS for c in s) or bool(ASCII_HU.search(s))
def slugify(text):
t = re.sub(r"\{\{.*?\}\}", " ", text)
t = re.sub(r"<[^>]*>", " ", t)
t = re.sub(r"&[a-z]+;|&#\d+;", " ", t)
t = unicodedata.normalize("NFKD", t)
t = "".join(c for c in t if not unicodedata.combining(c)).lower()
words = re.findall(r"[a-z0-9]+", t)
slug = "_".join(words[:5])[:40].strip("_")
return slug or "text"
# ---------------------------------------------------------------------------------------------------
# tokenizer: (kind, start, end) kind in action | tag | comment | text | script | style
# ---------------------------------------------------------------------------------------------------
def find_action_end(src, i):
if src.startswith("{{/*", i) or src.startswith("{{- /*", i):
j = src.find("*/", i)
j = src.find("}}", j)
else:
j = src.find("}}", i + 2)
return len(src) if j < 0 else j + 2
def tokenize(src):
toks, i, n = [], 0, len(src)
while i < n:
if src.startswith("{{", i):
j = find_action_end(src, i)
toks.append(("action", i, j))
i = j
elif src.startswith("<!--", i):
j = src.find("-->", i)
j = n if j < 0 else j + 3
toks.append(("comment", i, j))
i = j
elif src[i] == "<" and i + 1 < n and (src[i + 1].isalpha() or src[i + 1] in "/!"):
j, inq = i + 1, False
while j < n:
if src.startswith("{{", j):
j = find_action_end(src, j)
continue
if src[j] == '"':
inq = not inq
elif src[j] == ">" and not inq:
break
j += 1
j = min(j + 1, n)
toks.append(("tag", i, j))
name = tag_name(src[i:j])
if name in ("script", "style") and not src[i:j].startswith("</"):
close = re.compile(r"</%s\s*>" % name, re.I).search(src, j)
k = close.start() if close else n
toks.append((name, j, k))
i = k
else:
i = j
else:
j = i
while j < n and not src.startswith("{{", j) and not (
src[j] == "<" and j + 1 < n and (src[j + 1].isalpha() or src[j + 1] in "/!")):
j += 1
toks.append(("text", i, j))
i = j
return toks
def tag_name(tag):
m = re.match(r"</?\s*([a-zA-Z0-9]+)", tag)
return m.group(1).lower() if m else ""
# ---------------------------------------------------------------------------------------------------
# runs
# ---------------------------------------------------------------------------------------------------
class Ctx:
def __init__(self, prefix, bundle, stats):
self.prefix, self.bundle, self.stats = prefix, bundle, stats
def key_for(self, text):
base = f"{self.prefix}.{slugify(text)}"
key, k = base, 2
while key in self.bundle and self.bundle[key] != text:
key, k = f"{base}_{k}", k + 1
self.bundle[key] = text
return key
def edits_for_fragment(src, base, ctx, allow_attrs=True):
"""Return [(start, end, replacement)] for one HTML fragment (a template, or a JS string's body)."""
toks = tokenize(src)
edits = []
run = []
def flush():
nonlocal run
r = run
run = []
# trim edges: whitespace text, unbalanced/void tags
while r:
k, s, e = r[0]
if k == "text" and not src[s:e].strip():
r = r[1:]
continue
if k == "tag" and not balanced_open(src, r, 0):
r = r[1:]
continue
break
while r:
k, s, e = r[-1]
if k == "text" and not src[s:e].strip():
r = r[:-1]
continue
if k == "tag" and not balanced_close(src, r, len(r) - 1):
r = r[:-1]
continue
break
# a wrapper pair around the whole run (<span class="x">...</span>) stays outside the message
while len(r) >= 2 and r[0][0] == "tag" and r[-1][0] == "tag" and balanced_open(src, r, 0) \
and matching_close(src, r, 0) == len(r) - 1:
r = strip_ws(src, r[1:-1])
if not r:
return
segs = [r]
if not inline_balanced(src, r):
segs = [[t] for t in r if t[0] != "tag"]
for seg in segs:
s0, e0 = seg[0][1], seg[-1][2]
chunk = src[s0:e0]
visible = "".join(src[s:e] for k, s, e in seg if k == "text")
if not is_hu(visible):
continue
lead = len(chunk) - len(chunk.lstrip())
trail = len(chunk) - len(chunk.rstrip())
core = chunk.strip()
if MARKER.match(core):
continue
key = ctx.key_for(core)
ctx.stats["run"] += 1
edits.append((base + s0 + lead, base + e0 - trail, '{{T "%s"}}' % key))
for tok in toks:
kind, s, e = tok
text = src[s:e]
if kind == "text":
run.append(tok)
elif kind == "action":
if CONTROL.match(text) or MARKER.match(text):
flush()
else:
run.append(tok)
elif kind == "tag" and tag_name(text) in INLINE and not has_hu_attr(text) and not is_button_link(src, tok):
run.append(tok)
else:
flush()
if kind == "tag" and allow_attrs:
edits.extend(attr_edits(text, base + s, ctx))
elif kind == "script":
edits.extend(js_edits(text, base + s, ctx))
flush()
return edits
def is_button_link(src, tok):
"""An <a class="btn ..."> is a control of its own, not a word inside a sentence. Its closing </a>
is found by walking forward; both the opening and closing tag break the run."""
k, s, e = tok
t = src[s:e]
if tag_name(t) != "a":
return False
if t.startswith("</"):
opener = src.rfind("<a", 0, s)
return opener >= 0 and bool(re.search(r'class="[^"]*\bbtn\b', src[opener:src.find(">", opener) + 1]))
return bool(re.search(r'class="[^"]*\bbtn\b', t))
def has_hu_attr(tag):
return any(is_hu(v) and a.lower() not in SKIP_ATTRS for a, v in attrs(tag))
def attrs(tag):
out = []
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
pos = 0
while True:
m = pat.search(tag, pos)
if not m:
break
j = m.end()
start = j
while j < len(tag):
if tag.startswith("{{", j):
j = find_action_end(tag, j)
continue
if tag[j] == '"':
break
j += 1
out.append((m.group(1), tag[start:j]))
pos = j + 1
return out
def attr_edits(tag, base, ctx):
edits = []
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
pos = 0
while True:
m = pat.search(tag, pos)
if not m:
break
j = start = m.end()
while j < len(tag):
if tag.startswith("{{", j):
j = find_action_end(tag, j)
continue
if tag[j] == '"':
break
j += 1
name, val = m.group(1).lower(), tag[start:j]
pos = j + 1
if name in SKIP_ATTRS or not is_hu(re.sub(r"\{\{.*?\}\}", "", val)):
continue
if re.search(r"\{\{-?\s*(if|else|end|range|with)\b", val) or MARKER.match(val.strip()):
continue # conditional attribute copy: left for a hand edit (reported)
key = ctx.key_for(val)
ctx.stats["attr"] += 1
edits.append((base + start, base + j, '{{T "%s"}}' % key))
return edits
def js_edits(body, base, ctx):
"""String literals inside a <script> body. Comments, regex literals and template actions are
skipped so an apostrophe in a comment or a quote inside {{if eq .X "y"}} does not open a string."""
edits, i, n = [], 0, len(body)
prev = ""
while i < n:
c = body[i]
if body.startswith("{{", i):
i = find_action_end(body, i)
continue
if body.startswith("//", i):
j = body.find("\n", i)
i = n if j < 0 else j
continue
if body.startswith("/*", i):
j = body.find("*/", i + 2)
i = n if j < 0 else j + 2
continue
if c == "/" and prev in "(,=:[!&|?{};":
j = i + 1
while j < n and body[j] != "/" and body[j] != "\n":
j += 2 if body[j] == "\\" else 1
i = j + 1
prev = "/"
continue
if c in "'\"":
j = i + 1
while j < n and body[j] != c and body[j] != "\n":
if body.startswith("{{", j):
j = find_action_end(body, j)
continue
j += 2 if body[j] == "\\" else 1
content = body[i + 1:j]
if is_hu(re.sub(r"\{\{.*?\}\}", "", content)):
# A JS string is often a PIECE of markup: '\')">T\u00f6rl\u00e9s</button>' starts inside a
# tag. Everything up to the first '>' that precedes any '<' is that tag's tail, not text.
skip = 0
gt, lt = content.find(">"), content.find("<")
if gt >= 0 and (lt < 0 or gt < lt):
skip = gt + 1
# ...or it starts inside an attribute value: '%;opacity:.5" title="Rendszer: ' — only the
# text after the last attribute opening is copy (slice 1 release A, monitoring.html).
elif lt < 0:
last = None
for am in re.finditer(r'=\s*"', content):
last = am
if last is not None:
skip = last.end()
sub = edits_for_fragment(content[skip:], base + i + 1 + skip, ctx, allow_attrs=False)
ctx.stats["js"] += len(sub)
edits.extend(sub)
i = j + 1
prev = "s"
continue
if not c.isspace():
prev = c
i += 1
return edits
def strip_ws(src, r):
while r and r[0][0] == "text" and not src[r[0][1]:r[0][2]].strip():
r = r[1:]
while r and r[-1][0] == "text" and not src[r[-1][1]:r[-1][2]].strip():
r = r[:-1]
return r
def matching_close(src, run, idx):
name = tag_name(src[run[idx][1]:run[idx][2]])
depth = 0
for j in range(idx, len(run)):
k2, s2, e2 = run[j]
if k2 != "tag" or tag_name(src[s2:e2]) != name:
continue
depth += -1 if src[s2:e2].startswith("</") else 1
if depth == 0:
return j
return -1
def balanced_open(src, run, idx):
k, s, e = run[idx]
t = src[s:e]
name = tag_name(t)
if name not in INLINE or name in VOID or t.startswith("</") or t.endswith("/>"):
return False
depth = 0
for k2, s2, e2 in run[idx:]:
if k2 != "tag" or tag_name(src[s2:e2]) != name:
continue
depth += -1 if src[s2:e2].startswith("</") else 1
if depth == 0:
return True
return False
def balanced_close(src, run, idx):
k, s, e = run[idx]
t = src[s:e]
name = tag_name(t)
if name not in INLINE or name in VOID or not t.startswith("</"):
return False
depth = 0
for k2, s2, e2 in reversed(run[:idx + 1]):
if k2 != "tag" or tag_name(src[s2:e2]) != name:
continue
depth += 1 if src[s2:e2].startswith("</") else -1
if depth == 0:
return True
return False
def inline_balanced(src, run):
depth = collections.Counter()
for k, s, e in run:
if k != "tag":
continue
t = src[s:e]
name = tag_name(t)
if name in VOID or t.endswith("/>"):
continue
depth[name] += -1 if t.startswith("</") else 1
if depth[name] < 0:
return False
return all(v == 0 for v in depth.values())
def main(argv):
if not argv:
print(__doc__)
return 2
bundle = collections.OrderedDict()
if os.path.exists(HU_JSON):
with open(HU_JSON, encoding="utf-8") as f:
bundle.update(json.load(f, object_pairs_hook=collections.OrderedDict))
for path in argv:
src = open(path, encoding="utf-8").read()
prefix = os.path.splitext(os.path.basename(path))[0]
stats = collections.Counter()
ctx = Ctx(prefix, bundle, stats)
edits = sorted(edits_for_fragment(src, 0, ctx), key=lambda e: e[0])
out, last = [], 0
for s, e, rep in edits:
if s < last:
raise SystemExit(f"{path}: overlapping edit at {s}")
out.append(src[last:s])
out.append(rep)
last = e
out.append(src[last:])
with open(path, "w", encoding="utf-8") as f:
f.write("".join(out))
print(f"{path}: {stats['run']} runs, {stats['attr']} attributes, {stats['js']} of the runs inside JS strings")
os.makedirs(os.path.dirname(HU_JSON), exist_ok=True)
with open(HU_JSON, "w", encoding="utf-8") as f:
json.dump(bundle, f, ensure_ascii=False, indent=2)
f.write("\n")
print(f"{HU_JSON}: {len(bundle)} keys")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))