#!/usr/bin/env python3 """i18n_extract.py -- move the Hungarian copy of a dashboard template into the message bundle. python3 scripts/i18n_extract.py internal/web/templates/launcher.html [more.html ...] For each file: every Hungarian text run, attribute value and inline-JS string is replaced IN PLACE by a `{{T "."}}` marker, and the exact text it replaced is written to internal/i18n/locales/hu.json under that key. Nothing else in the file moves -- surrounding whitespace, tags and template actions stay where they were. Why that is safe (and what proves it): internal/i18n expands every marker back to the bundle text BEFORE html/template parses the file (see internal/i18n/i18n.go). The Hungarian template set is therefore parsed from the original bytes, in the original escaping contexts. The parity test (internal/web/i18n_parity_test.go) renders the converted pages and compares them byte-for-byte with fixtures captured from the unconverted templates -- this tool is not trusted, it is checked. A "run" is the unit a translator needs: text, value actions ({{.Name}}) and inline tags (, ,
, ...) up to the next block tag or control action ({{if}}, {{else}}, {{end}}, {{range}}, comments). Word order therefore stays inside one message. English values of JS-context keys must not contain a bare quote, backslash or newline -- pinned by TestI18nJSContextValuesAreSafe. Idempotent: a file already carrying markers keeps them; a run that already is a marker is skipped. Source is ASCII-only; Hungarian letters are written as \\u escapes. """ import collections import json import os import re import sys import unicodedata HERE = os.path.dirname(os.path.abspath(__file__)) CTRL = os.path.dirname(HERE) HU_JSON = os.path.join(CTRL, "internal", "i18n", "locales", "hu.json") HU_LETTERS = set("\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170") # ASCII-only Hungarian: a word with no accented letter is invisible to the letter test. Case-insensitive. # The last line was added after a review of what the first pass left behind (Megszakadt, Elavult, # Csak x86, Pi kompatibilis, Rendszermonitor, Most nem, sikertelen, ismeretlen). ASCII_HU = re.compile(r"\b(Fut|Nincs|Igen|Nem|Hiba|Mentve|Rendben|Adatok|Napi|Heti|Havi|Mindig|Soha|Tegnap|" r"Perc|Letiltva|Bekapcsolva|Kikapcsolva|Folyamatban|Sikertelen|Sikeres|Tartalom|" r"Kapcsolat|Vissza|Megosztva|Kezdeti|Rendszer|Csatlakozva|Kulcs|Megnyit|Elrejt|" r"Ismeretlen|Szabad|Foglalt|Kijelentkezes|Napok|nap|perc|ora|" r"Megszakadt|Elavult|Csak|kompatibilis|Rendszermonitor|Most nem)\b", re.I) INLINE = {"strong", "em", "b", "i", "br", "code", "small", "a", "kbd", "u"} # NOT span: it is layout here VOID = {"br"} SKIP_ATTRS = {"class", "id", "href", "src", "style", "name", "type", "for", "action", "method", "rel", "target", "width", "height", "autocomplete", "role", "hidden", "onclick", "onerror", "onchange", "onsubmit", "oninput", "aria-controls", "aria-expanded", "content", "charset", "lang", "value"} CONTROL = re.compile(r"^\{\{-?\s*(/\*|if\b|else\b|end\b|range\b|with\b|define\b|block\b|template\b|break\b|continue\b|\$\w+\s*:?=)") MARKER = re.compile(r'^\{\{\s*T\s+"[^"]+"\s*\}\}$') def is_hu(s): return any(c in HU_LETTERS for c in s) or bool(ASCII_HU.search(s)) def slugify(text): t = re.sub(r"\{\{.*?\}\}", " ", text) t = re.sub(r"<[^>]*>", " ", t) t = re.sub(r"&[a-z]+;|&#\d+;", " ", t) t = unicodedata.normalize("NFKD", t) t = "".join(c for c in t if not unicodedata.combining(c)).lower() words = re.findall(r"[a-z0-9]+", t) slug = "_".join(words[:5])[:40].strip("_") return slug or "text" # --------------------------------------------------------------------------------------------------- # tokenizer: (kind, start, end) kind in action | tag | comment | text | script | style # --------------------------------------------------------------------------------------------------- def find_action_end(src, i): if src.startswith("{{/*", i) or src.startswith("{{- /*", i): j = src.find("*/", i) j = src.find("}}", j) else: j = src.find("}}", i + 2) return len(src) if j < 0 else j + 2 def tokenize(src): toks, i, n = [], 0, len(src) while i < n: if src.startswith("{{", i): j = find_action_end(src, i) toks.append(("action", i, j)) i = j elif src.startswith("", i) j = n if j < 0 else j + 3 toks.append(("comment", i, j)) i = j elif src[i] == "<" and i + 1 < n and (src[i + 1].isalpha() or src[i + 1] in "/!"): j, inq = i + 1, False while j < n: if src.startswith("{{", j): j = find_action_end(src, j) continue if src[j] == '"': inq = not inq elif src[j] == ">" and not inq: break j += 1 j = min(j + 1, n) toks.append(("tag", i, j)) name = tag_name(src[i:j]) if name in ("script", "style") and not src[i:j].startswith("" % name, re.I).search(src, j) k = close.start() if close else n toks.append((name, j, k)) i = k else: i = j else: j = i while j < n and not src.startswith("{{", j) and not ( src[j] == "<" and j + 1 < n and (src[j + 1].isalpha() or src[j + 1] in "/!")): j += 1 toks.append(("text", i, j)) i = j return toks def tag_name(tag): m = re.match(r"...) stays outside the message while len(r) >= 2 and r[0][0] == "tag" and r[-1][0] == "tag" and balanced_open(src, r, 0) \ and matching_close(src, r, 0) == len(r) - 1: r = strip_ws(src, r[1:-1]) if not r: return segs = [r] if not inline_balanced(src, r): segs = [[t] for t in r if t[0] != "tag"] for seg in segs: s0, e0 = seg[0][1], seg[-1][2] chunk = src[s0:e0] visible = "".join(src[s:e] for k, s, e in seg if k == "text") if not is_hu(visible): continue lead = len(chunk) - len(chunk.lstrip()) trail = len(chunk) - len(chunk.rstrip()) core = chunk.strip() if MARKER.match(core): continue key = ctx.key_for(core) ctx.stats["run"] += 1 edits.append((base + s0 + lead, base + e0 - trail, '{{T "%s"}}' % key)) for tok in toks: kind, s, e = tok text = src[s:e] if kind == "text": run.append(tok) elif kind == "action": if CONTROL.match(text) or MARKER.match(text): flush() else: run.append(tok) elif kind == "tag" and tag_name(text) in INLINE and not has_hu_attr(text) and not is_button_link(src, tok): run.append(tok) else: flush() if kind == "tag" and allow_attrs: edits.extend(attr_edits(text, base + s, ctx)) elif kind == "script": edits.extend(js_edits(text, base + s, ctx)) flush() return edits def is_button_link(src, tok): """An
is a control of its own, not a word inside a sentence. Its closing is found by walking forward; both the opening and closing tag break the run.""" k, s, e = tok t = src[s:e] if tag_name(t) != "a": return False if t.startswith("= 0 and bool(re.search(r'class="[^"]*\bbtn\b', src[opener:src.find(">", opener) + 1])) return bool(re.search(r'class="[^"]*\bbtn\b', t)) def has_hu_attr(tag): return any(is_hu(v) and a.lower() not in SKIP_ATTRS for a, v in attrs(tag)) def attrs(tag): out = [] pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"') pos = 0 while True: m = pat.search(tag, pos) if not m: break j = m.end() start = j while j < len(tag): if tag.startswith("{{", j): j = find_action_end(tag, j) continue if tag[j] == '"': break j += 1 out.append((m.group(1), tag[start:j])) pos = j + 1 return out def attr_edits(tag, base, ctx): edits = [] pat = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"') pos = 0 while True: m = pat.search(tag, pos) if not m: break j = start = m.end() while j < len(tag): if tag.startswith("{{", j): j = find_action_end(tag, j) continue if tag[j] == '"': break j += 1 name, val = m.group(1).lower(), tag[start:j] pos = j + 1 if name in SKIP_ATTRS or not is_hu(re.sub(r"\{\{.*?\}\}", "", val)): continue if re.search(r"\{\{-?\s*(if|else|end|range|with)\b", val) or MARKER.match(val.strip()): continue # conditional attribute copy: left for a hand edit (reported) key = ctx.key_for(val) ctx.stats["attr"] += 1 edits.append((base + start, base + j, '{{T "%s"}}' % key)) return edits def js_edits(body, base, ctx): """String literals inside a