ff68b0b433
gates / gates (push) Successful in 20s
Ten pages converted against fixtures captured from unconverted templates (6be55a0); marker
coverage test; English page mask narrowed; English retrieval-promise stems; common keys;
title keys and infraMeta from the bundle; old wording tests read the Hungarian expansion.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
456 lines
17 KiB
Python
456 lines
17 KiB
Python
#!/usr/bin/env python3
|
|
"""i18n_extract.py -- move the Hungarian copy of a dashboard template into the message bundle.
|
|
|
|
python3 scripts/i18n_extract.py internal/web/templates/launcher.html [more.html ...]
|
|
|
|
For each file: every Hungarian text run, attribute value and inline-JS string is replaced IN PLACE by
|
|
a `{{T "<file>.<slug>"}}` marker, and the exact text it replaced is written to
|
|
internal/i18n/locales/hu.json under that key. Nothing else in the file moves -- surrounding
|
|
whitespace, tags and template actions stay where they were.
|
|
|
|
Why that is safe (and what proves it): internal/i18n expands every marker back to the bundle text
|
|
BEFORE html/template parses the file (see internal/i18n/i18n.go). The Hungarian template set is
|
|
therefore parsed from the original bytes, in the original escaping contexts. The parity test
|
|
(internal/web/i18n_parity_test.go) renders the converted pages and compares them byte-for-byte with
|
|
fixtures captured from the unconverted templates -- this tool is not trusted, it is checked.
|
|
|
|
A "run" is the unit a translator needs: text, value actions ({{.Name}}) and inline tags
|
|
(<strong>, <a>, <br>, ...) up to the next block tag or control action ({{if}}, {{else}}, {{end}},
|
|
{{range}}, comments). Word order therefore stays inside one message. English values of JS-context
|
|
keys must not contain a bare quote, backslash or newline -- pinned by TestI18nJSContextValuesAreSafe.
|
|
|
|
Idempotent: a file already carrying markers keeps them; a run that already is a marker is skipped.
|
|
Source is ASCII-only; Hungarian letters are written as \\u escapes.
|
|
"""
|
|
import collections
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import unicodedata
|
|
|
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
CTRL = os.path.dirname(HERE)
|
|
HU_JSON = os.path.join(CTRL, "internal", "i18n", "locales", "hu.json")
|
|
|
|
HU_LETTERS = set("\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170")
|
|
# ASCII-only Hungarian: a word with no accented letter is invisible to the letter test. Case-insensitive.
|
|
# The last line was added after a review of what the first pass left behind (Megszakadt, Elavult,
|
|
# Csak x86, Pi kompatibilis, Rendszermonitor, Most nem, sikertelen, ismeretlen).
|
|
ASCII_HU = re.compile(r"\b(Fut|Nincs|Igen|Nem|Hiba|Mentve|Rendben|Adatok|Napi|Heti|Havi|Mindig|Soha|Tegnap|"
|
|
r"Perc|Letiltva|Bekapcsolva|Kikapcsolva|Folyamatban|Sikertelen|Sikeres|Tartalom|"
|
|
r"Kapcsolat|Vissza|Megosztva|Kezdeti|Rendszer|Csatlakozva|Kulcs|Megnyit|Elrejt|"
|
|
r"Ismeretlen|Szabad|Foglalt|Kijelentkezes|Napok|nap|perc|ora|"
|
|
r"Megszakadt|Elavult|Csak|kompatibilis|Rendszermonitor|Most nem|"
|
|
# slice 1 release A review: left behind on dashboard..settings_notifications
|
|
r"Folyamat|Titkos|Processzor|Hamarosan|aldomain|adatai|Figyelem|pelda)\b", re.I)
|
|
|
|
INLINE = {"strong", "em", "b", "i", "br", "code", "small", "a", "kbd", "u"} # NOT span: it is layout here
|
|
VOID = {"br"}
|
|
SKIP_ATTRS = {"class", "id", "href", "src", "style", "name", "type", "for", "action", "method", "rel",
|
|
"target", "width", "height", "autocomplete", "role", "hidden", "onclick", "onerror",
|
|
"onchange", "onsubmit", "oninput", "aria-controls", "aria-expanded", "content", "charset",
|
|
"lang", "value"}
|
|
CONTROL = re.compile(r"^\{\{-?\s*(/\*|if\b|else\b|end\b|range\b|with\b|define\b|block\b|template\b|break\b|continue\b|\$\w+\s*:?=)")
|
|
MARKER = re.compile(r'^\{\{\s*T\s+"[^"]+"\s*\}\}$')
|
|
|
|
|
|
def is_hu(s):
|
|
return any(c in HU_LETTERS for c in s) or bool(ASCII_HU.search(s))
|
|
|
|
|
|
def slugify(text):
|
|
t = re.sub(r"\{\{.*?\}\}", " ", text)
|
|
t = re.sub(r"<[^>]*>", " ", t)
|
|
t = re.sub(r"&[a-z]+;|&#\d+;", " ", t)
|
|
t = unicodedata.normalize("NFKD", t)
|
|
t = "".join(c for c in t if not unicodedata.combining(c)).lower()
|
|
words = re.findall(r"[a-z0-9]+", t)
|
|
slug = "_".join(words[:5])[:40].strip("_")
|
|
return slug or "text"
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------------
|
|
# tokenizer: (kind, start, end) kind in action | tag | comment | text | script | style
|
|
# ---------------------------------------------------------------------------------------------------
|
|
|
|
def find_action_end(src, i):
|
|
if src.startswith("{{/*", i) or src.startswith("{{- /*", i):
|
|
j = src.find("*/", i)
|
|
j = src.find("}}", j)
|
|
else:
|
|
j = src.find("}}", i + 2)
|
|
return len(src) if j < 0 else j + 2
|
|
|
|
|
|
def tokenize(src):
|
|
toks, i, n = [], 0, len(src)
|
|
while i < n:
|
|
if src.startswith("{{", i):
|
|
j = find_action_end(src, i)
|
|
toks.append(("action", i, j))
|
|
i = j
|
|
elif src.startswith("<!--", i):
|
|
j = src.find("-->", i)
|
|
j = n if j < 0 else j + 3
|
|
toks.append(("comment", i, j))
|
|
i = j
|
|
elif src[i] == "<" and i + 1 < n and (src[i + 1].isalpha() or src[i + 1] in "/!"):
|
|
j, inq = i + 1, False
|
|
while j < n:
|
|
if src.startswith("{{", j):
|
|
j = find_action_end(src, j)
|
|
continue
|
|
if src[j] == '"':
|
|
inq = not inq
|
|
elif src[j] == ">" and not inq:
|
|
break
|
|
j += 1
|
|
j = min(j + 1, n)
|
|
toks.append(("tag", i, j))
|
|
name = tag_name(src[i:j])
|
|
if name in ("script", "style") and not src[i:j].startswith("</"):
|
|
close = re.compile(r"</%s\s*>" % name, re.I).search(src, j)
|
|
k = close.start() if close else n
|
|
toks.append((name, j, k))
|
|
i = k
|
|
else:
|
|
i = j
|
|
else:
|
|
j = i
|
|
while j < n and not src.startswith("{{", j) and not (
|
|
src[j] == "<" and j + 1 < n and (src[j + 1].isalpha() or src[j + 1] in "/!")):
|
|
j += 1
|
|
toks.append(("text", i, j))
|
|
i = j
|
|
return toks
|
|
|
|
|
|
def tag_name(tag):
|
|
m = re.match(r"</?\s*([a-zA-Z0-9]+)", tag)
|
|
return m.group(1).lower() if m else ""
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------------
|
|
# runs
|
|
# ---------------------------------------------------------------------------------------------------
|
|
|
|
class Ctx:
|
|
def __init__(self, prefix, bundle, stats):
|
|
self.prefix, self.bundle, self.stats = prefix, bundle, stats
|
|
|
|
def key_for(self, text):
|
|
base = f"{self.prefix}.{slugify(text)}"
|
|
key, k = base, 2
|
|
while key in self.bundle and self.bundle[key] != text:
|
|
key, k = f"{base}_{k}", k + 1
|
|
self.bundle[key] = text
|
|
return key
|
|
|
|
|
|
def edits_for_fragment(src, base, ctx, allow_attrs=True):
|
|
"""Return [(start, end, replacement)] for one HTML fragment (a template, or a JS string's body)."""
|
|
toks = tokenize(src)
|
|
edits = []
|
|
run = []
|
|
|
|
def flush():
|
|
nonlocal run
|
|
r = run
|
|
run = []
|
|
# trim edges: whitespace text, unbalanced/void tags
|
|
while r:
|
|
k, s, e = r[0]
|
|
if k == "text" and not src[s:e].strip():
|
|
r = r[1:]
|
|
continue
|
|
if k == "tag" and not balanced_open(src, r, 0):
|
|
r = r[1:]
|
|
continue
|
|
break
|
|
while r:
|
|
k, s, e = r[-1]
|
|
if k == "text" and not src[s:e].strip():
|
|
r = r[:-1]
|
|
continue
|
|
if k == "tag" and not balanced_close(src, r, len(r) - 1):
|
|
r = r[:-1]
|
|
continue
|
|
break
|
|
# a wrapper pair around the whole run (<span class="x">...</span>) stays outside the message
|
|
while len(r) >= 2 and r[0][0] == "tag" and r[-1][0] == "tag" and balanced_open(src, r, 0) \
|
|
and matching_close(src, r, 0) == len(r) - 1:
|
|
r = strip_ws(src, r[1:-1])
|
|
if not r:
|
|
return
|
|
segs = [r]
|
|
if not inline_balanced(src, r):
|
|
segs = [[t] for t in r if t[0] != "tag"]
|
|
for seg in segs:
|
|
s0, e0 = seg[0][1], seg[-1][2]
|
|
chunk = src[s0:e0]
|
|
visible = "".join(src[s:e] for k, s, e in seg if k == "text")
|
|
if not is_hu(visible):
|
|
continue
|
|
lead = len(chunk) - len(chunk.lstrip())
|
|
trail = len(chunk) - len(chunk.rstrip())
|
|
core = chunk.strip()
|
|
if MARKER.match(core):
|
|
continue
|
|
key = ctx.key_for(core)
|
|
ctx.stats["run"] += 1
|
|
edits.append((base + s0 + lead, base + e0 - trail, '{{T "%s"}}' % key))
|
|
|
|
for tok in toks:
|
|
kind, s, e = tok
|
|
text = src[s:e]
|
|
if kind == "text":
|
|
run.append(tok)
|
|
elif kind == "action":
|
|
if CONTROL.match(text) or MARKER.match(text):
|
|
flush()
|
|
else:
|
|
run.append(tok)
|
|
elif kind == "tag" and tag_name(text) in INLINE and not has_hu_attr(text) and not is_button_link(src, tok):
|
|
run.append(tok)
|
|
else:
|
|
flush()
|
|
if kind == "tag" and allow_attrs:
|
|
edits.extend(attr_edits(text, base + s, ctx))
|
|
elif kind == "script":
|
|
edits.extend(js_edits(text, base + s, ctx))
|
|
flush()
|
|
return edits
|
|
|
|
|
|
def is_button_link(src, tok):
|
|
"""An <a class="btn ..."> is a control of its own, not a word inside a sentence. Its closing </a>
|
|
is found by walking forward; both the opening and closing tag break the run."""
|
|
k, s, e = tok
|
|
t = src[s:e]
|
|
if tag_name(t) != "a":
|
|
return False
|
|
if t.startswith("</"):
|
|
opener = src.rfind("<a", 0, s)
|
|
return opener >= 0 and bool(re.search(r'class="[^"]*\bbtn\b', src[opener:src.find(">", opener) + 1]))
|
|
return bool(re.search(r'class="[^"]*\bbtn\b', t))
|
|
|
|
|
|
def has_hu_attr(tag):
|
|
return any(is_hu(v) and a.lower() not in SKIP_ATTRS for a, v in attrs(tag))
|
|
|
|
|
|
def attrs(tag):
|
|
out = []
|
|
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
|
|
pos = 0
|
|
while True:
|
|
m = pat.search(tag, pos)
|
|
if not m:
|
|
break
|
|
j = m.end()
|
|
start = j
|
|
while j < len(tag):
|
|
if tag.startswith("{{", j):
|
|
j = find_action_end(tag, j)
|
|
continue
|
|
if tag[j] == '"':
|
|
break
|
|
j += 1
|
|
out.append((m.group(1), tag[start:j]))
|
|
pos = j + 1
|
|
return out
|
|
|
|
|
|
def attr_edits(tag, base, ctx):
|
|
edits = []
|
|
pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"')
|
|
pos = 0
|
|
while True:
|
|
m = pat.search(tag, pos)
|
|
if not m:
|
|
break
|
|
j = start = m.end()
|
|
while j < len(tag):
|
|
if tag.startswith("{{", j):
|
|
j = find_action_end(tag, j)
|
|
continue
|
|
if tag[j] == '"':
|
|
break
|
|
j += 1
|
|
name, val = m.group(1).lower(), tag[start:j]
|
|
pos = j + 1
|
|
if name in SKIP_ATTRS or not is_hu(re.sub(r"\{\{.*?\}\}", "", val)):
|
|
continue
|
|
if re.search(r"\{\{-?\s*(if|else|end|range|with)\b", val) or MARKER.match(val.strip()):
|
|
continue # conditional attribute copy: left for a hand edit (reported)
|
|
key = ctx.key_for(val)
|
|
ctx.stats["attr"] += 1
|
|
edits.append((base + start, base + j, '{{T "%s"}}' % key))
|
|
return edits
|
|
|
|
|
|
def js_edits(body, base, ctx):
|
|
"""String literals inside a <script> body. Comments, regex literals and template actions are
|
|
skipped so an apostrophe in a comment or a quote inside {{if eq .X "y"}} does not open a string."""
|
|
edits, i, n = [], 0, len(body)
|
|
prev = ""
|
|
while i < n:
|
|
c = body[i]
|
|
if body.startswith("{{", i):
|
|
i = find_action_end(body, i)
|
|
continue
|
|
if body.startswith("//", i):
|
|
j = body.find("\n", i)
|
|
i = n if j < 0 else j
|
|
continue
|
|
if body.startswith("/*", i):
|
|
j = body.find("*/", i + 2)
|
|
i = n if j < 0 else j + 2
|
|
continue
|
|
if c == "/" and prev in "(,=:[!&|?{};":
|
|
j = i + 1
|
|
while j < n and body[j] != "/" and body[j] != "\n":
|
|
j += 2 if body[j] == "\\" else 1
|
|
i = j + 1
|
|
prev = "/"
|
|
continue
|
|
if c in "'\"":
|
|
j = i + 1
|
|
while j < n and body[j] != c and body[j] != "\n":
|
|
if body.startswith("{{", j):
|
|
j = find_action_end(body, j)
|
|
continue
|
|
j += 2 if body[j] == "\\" else 1
|
|
content = body[i + 1:j]
|
|
if is_hu(re.sub(r"\{\{.*?\}\}", "", content)):
|
|
# A JS string is often a PIECE of markup: '\')">T\u00f6rl\u00e9s</button>' starts inside a
|
|
# tag. Everything up to the first '>' that precedes any '<' is that tag's tail, not text.
|
|
skip = 0
|
|
gt, lt = content.find(">"), content.find("<")
|
|
if gt >= 0 and (lt < 0 or gt < lt):
|
|
skip = gt + 1
|
|
# ...or it starts inside an attribute value: '%;opacity:.5" title="Rendszer: ' — only the
|
|
# text after the last attribute opening is copy (slice 1 release A, monitoring.html).
|
|
elif lt < 0:
|
|
last = None
|
|
for am in re.finditer(r'=\s*"', content):
|
|
last = am
|
|
if last is not None:
|
|
skip = last.end()
|
|
sub = edits_for_fragment(content[skip:], base + i + 1 + skip, ctx, allow_attrs=False)
|
|
ctx.stats["js"] += len(sub)
|
|
edits.extend(sub)
|
|
i = j + 1
|
|
prev = "s"
|
|
continue
|
|
if not c.isspace():
|
|
prev = c
|
|
i += 1
|
|
return edits
|
|
|
|
|
|
def strip_ws(src, r):
|
|
while r and r[0][0] == "text" and not src[r[0][1]:r[0][2]].strip():
|
|
r = r[1:]
|
|
while r and r[-1][0] == "text" and not src[r[-1][1]:r[-1][2]].strip():
|
|
r = r[:-1]
|
|
return r
|
|
|
|
|
|
def matching_close(src, run, idx):
|
|
name = tag_name(src[run[idx][1]:run[idx][2]])
|
|
depth = 0
|
|
for j in range(idx, len(run)):
|
|
k2, s2, e2 = run[j]
|
|
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
|
continue
|
|
depth += -1 if src[s2:e2].startswith("</") else 1
|
|
if depth == 0:
|
|
return j
|
|
return -1
|
|
|
|
|
|
def balanced_open(src, run, idx):
|
|
k, s, e = run[idx]
|
|
t = src[s:e]
|
|
name = tag_name(t)
|
|
if name not in INLINE or name in VOID or t.startswith("</") or t.endswith("/>"):
|
|
return False
|
|
depth = 0
|
|
for k2, s2, e2 in run[idx:]:
|
|
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
|
continue
|
|
depth += -1 if src[s2:e2].startswith("</") else 1
|
|
if depth == 0:
|
|
return True
|
|
return False
|
|
|
|
|
|
def balanced_close(src, run, idx):
|
|
k, s, e = run[idx]
|
|
t = src[s:e]
|
|
name = tag_name(t)
|
|
if name not in INLINE or name in VOID or not t.startswith("</"):
|
|
return False
|
|
depth = 0
|
|
for k2, s2, e2 in reversed(run[:idx + 1]):
|
|
if k2 != "tag" or tag_name(src[s2:e2]) != name:
|
|
continue
|
|
depth += 1 if src[s2:e2].startswith("</") else -1
|
|
if depth == 0:
|
|
return True
|
|
return False
|
|
|
|
|
|
def inline_balanced(src, run):
|
|
depth = collections.Counter()
|
|
for k, s, e in run:
|
|
if k != "tag":
|
|
continue
|
|
t = src[s:e]
|
|
name = tag_name(t)
|
|
if name in VOID or t.endswith("/>"):
|
|
continue
|
|
depth[name] += -1 if t.startswith("</") else 1
|
|
if depth[name] < 0:
|
|
return False
|
|
return all(v == 0 for v in depth.values())
|
|
|
|
|
|
def main(argv):
|
|
if not argv:
|
|
print(__doc__)
|
|
return 2
|
|
bundle = collections.OrderedDict()
|
|
if os.path.exists(HU_JSON):
|
|
with open(HU_JSON, encoding="utf-8") as f:
|
|
bundle.update(json.load(f, object_pairs_hook=collections.OrderedDict))
|
|
for path in argv:
|
|
src = open(path, encoding="utf-8").read()
|
|
prefix = os.path.splitext(os.path.basename(path))[0]
|
|
stats = collections.Counter()
|
|
ctx = Ctx(prefix, bundle, stats)
|
|
edits = sorted(edits_for_fragment(src, 0, ctx), key=lambda e: e[0])
|
|
out, last = [], 0
|
|
for s, e, rep in edits:
|
|
if s < last:
|
|
raise SystemExit(f"{path}: overlapping edit at {s}")
|
|
out.append(src[last:s])
|
|
out.append(rep)
|
|
last = e
|
|
out.append(src[last:])
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
f.write("".join(out))
|
|
print(f"{path}: {stats['run']} runs, {stats['attr']} attributes, {stats['js']} of the runs inside JS strings")
|
|
os.makedirs(os.path.dirname(HU_JSON), exist_ok=True)
|
|
with open(HU_JSON, "w", encoding="utf-8") as f:
|
|
json.dump(bundle, f, ensure_ascii=False, indent=2)
|
|
f.write("\n")
|
|
print(f"{HU_JSON}: {len(bundle)} keys")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main(sys.argv[1:]))
|