5b7b2b22f1
gates / gates (push) Successful in 23s
Phase 0 of the localisation starter: i18n_inventory.py counts every customer-visible Hungarian string; the audit names six further claims in the prompt that live source disproved. Rows for the compare-not-show sites, wizard deletion, the wire-contract comment blind spot, and localisation slices 1-6. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
532 lines
25 KiB
Python
532 lines
25 KiB
Python
#!/usr/bin/env python3
|
|
"""i18n_inventory.py -- count every customer-visible Hungarian string across the Felhom repos.
|
|
|
|
Produces the numbers behind documentation/audits/I18N-INVENTORY-2026-09-17.md: one table per
|
|
surface (page templates, controller Go, inline JS, hub customer mail, installer, catalog, guide,
|
|
first-boot wizard), the ten most-repeated strings per surface, strings built with word order
|
|
(Sprintf/concatenation), numeric-parameter strings (English needs plural forms), date layouts and
|
|
size helpers, and every place a Hungarian string is COMPARED rather than shown.
|
|
|
|
It is a survey, not a gate: it never fails a build. It reads files as UTF-8 in Python, so the
|
|
accented-grep traps (locale, kubectl exec) do not apply -- but the workspace rule for Hungarian
|
|
search still holds, and each surface prints a POSITIVE control (an ASCII fragment of a Hungarian
|
|
word known to be there, which the letter matcher must also flag) and a NEGATIVE control (a fragment
|
|
known to be absent, and an ASCII-only line the matcher must NOT flag). A surface whose positive
|
|
control reads 0 is reported as BROKEN, not as "no Hungarian".
|
|
|
|
This source is ASCII-only on purpose: the Hungarian letters are written as \\u escapes.
|
|
|
|
Usage:
|
|
python3 scripts/i18n_inventory.py [--root /mnt/5_hdd/felhom.eu/git] [--json OUT.json]
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import collections
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import yaml
|
|
|
|
HU_LETTERS = set(
|
|
"\u00e1\u00e9\u00ed\u00f3\u00f6\u0151\u00fa\u00fc\u0171"
|
|
"\u00c1\u00c9\u00cd\u00d3\u00d6\u0150\u00da\u00dc\u0170"
|
|
)
|
|
|
|
# ASCII-only Hungarian words: a string made only of these carries no accented letter and would be
|
|
# missed by a letter search. Chosen to avoid English homographs ("Most", "Ma", "Online" are left out).
|
|
ASCII_HU_SEED = [
|
|
"Fut", "Nincs", "Igen", "Nem", "Hiba", "Mentve", "Rendben", "Adatok", "Napi", "Heti",
|
|
"Havi", "Mindig", "Soha", "Tegnap", "Perc", "Letiltva", "Bekapcsolva", "Kikapcsolva",
|
|
"Folyamatban", "Sikertelen", "Sikeres", "Tartalom", "Kapcsolat", "Vissza", "Megosztva",
|
|
"Kezdeti", "Rendszer", "Csatlakozva", "Kulcs", "Szerkeszt", "Keres", "Megnyit", "Elrejt",
|
|
"Mutasd", "Mentsd", "Kattints", "Nyisd", "Ismeretlen", "Szabad", "Foglalt",
|
|
# added 2026-09-17 after the spike's by-eye review of what a letter search left behind
|
|
"Megszakadt", "Elavult", "Csak", "kompatibilis", "Rendszermonitor", "Most nem",
|
|
]
|
|
SEED_RE = re.compile(r"\b(" + "|".join(ASCII_HU_SEED) + r")\b", re.I)
|
|
|
|
NEG_FRAGMENT = "qzxvkjq" # appears in no Felhom file
|
|
NEG_ASCII_LINE = "Backup completed at 03:00, next run tomorrow."
|
|
|
|
|
|
def has_hu(s: str) -> bool:
|
|
return any(c in HU_LETTERS for c in s)
|
|
|
|
|
|
def is_hu_text(s: str) -> bool:
|
|
"""Hungarian by letter, or an ASCII-only string made of a seed word."""
|
|
if has_hu(s):
|
|
return True
|
|
return bool(SEED_RE.search(s))
|
|
|
|
|
|
def read(p: Path) -> str:
|
|
return p.read_text(encoding="utf-8", errors="replace")
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------------
|
|
# Go string-literal lexer (comments skipped, raw strings kept, line numbers tracked)
|
|
# ---------------------------------------------------------------------------------------------------
|
|
|
|
def go_literals(src: str):
|
|
"""Yield (line, literal_body, raw) for every string literal outside comments."""
|
|
i, n, line = 0, len(src), 1
|
|
while i < n:
|
|
c = src[i]
|
|
if c == "\n":
|
|
line += 1
|
|
i += 1
|
|
elif src.startswith("//", i):
|
|
j = src.find("\n", i)
|
|
i = n if j < 0 else j
|
|
elif src.startswith("/*", i):
|
|
j = src.find("*/", i + 2)
|
|
j = n if j < 0 else j + 2
|
|
line += src.count("\n", i, j)
|
|
i = j
|
|
elif c == "'":
|
|
j = i + 1
|
|
while j < n and src[j] != "'":
|
|
j += 2 if src[j] == "\\" else 1
|
|
i = j + 1
|
|
elif c == '"':
|
|
j = i + 1
|
|
while j < n and src[j] != '"' and src[j] != "\n":
|
|
j += 2 if src[j] == "\\" else 1
|
|
yield line, src[i + 1:j], False
|
|
i = j + 1
|
|
elif c == "`":
|
|
j = src.find("`", i + 1)
|
|
j = n if j < 0 else j
|
|
yield line, src[i + 1:j], True
|
|
line += src.count("\n", i, j)
|
|
i = j + 1
|
|
else:
|
|
i += 1
|
|
|
|
|
|
LOG_RE = re.compile(r"(\blog(ger)?\.|\.logger\.|\blg\.|Logf|Printf|Println|debugf|Debugf|infof|warnf)")
|
|
ERR_RE = re.compile(r"(fmt\.Errorf|errors\.New)\(")
|
|
CMP_RE = re.compile(r'(==\s*"|!=\s*"|\bcase\s+"|strings\.(Contains|HasPrefix|HasSuffix|EqualFold|Index)\()')
|
|
FMT_VERB_RE = re.compile(r"%[-+# 0-9.]*[sdvqfx]")
|
|
NUM_VERB_RE = re.compile(r"%[-+# 0-9.]*d")
|
|
|
|
|
|
def classify_go(line_text: str, body: str) -> str:
|
|
if body.lstrip().startswith(("[DEBUG]", "[INFO]", "[WARN]", "[ERROR]")) or LOG_RE.search(line_text):
|
|
return "log"
|
|
if CMP_RE.search(line_text):
|
|
return "compare"
|
|
if ERR_RE.search(line_text):
|
|
return "error"
|
|
return "shown"
|
|
|
|
|
|
def go_package_consts(src: str):
|
|
"""Yield (name, value) for PACKAGE-LEVEL const/var string declarations only -- a local `msg := ...`
|
|
reused as a name in another package is not the same value and must not be matched."""
|
|
in_block = False
|
|
for l in src.split("\n"):
|
|
if re.match(r"^(const|var)\s*\($", l):
|
|
in_block = True
|
|
continue
|
|
if in_block and l.startswith(")"):
|
|
in_block = False
|
|
continue
|
|
m = (re.match(r'^\t([A-Za-z_]\w*)\s*(?:string\s*)?=\s*"((?:[^"\\]|\\.)*)"', l) if in_block
|
|
else re.match(r'^(?:const|var)\s+([A-Za-z_]\w*)\s*(?:string\s*)?=\s*"((?:[^"\\]|\\.)*)"', l))
|
|
if m:
|
|
yield m.group(1), m.group(2)
|
|
|
|
|
|
def go_indirect_compares(files, root: Path):
|
|
"""Hungarian held in a package-level constant/variable, then COMPARED through that name in the
|
|
same package -- the shape a same-line literal search cannot see (handlers.go
|
|
offboxStaleWarningMarker is the case that prompted it)."""
|
|
bydir = collections.defaultdict(list)
|
|
for f in files:
|
|
bydir[f.parent].append(f)
|
|
out = []
|
|
for d, fs in bydir.items():
|
|
srcs = {f: read(f) for f in fs}
|
|
names = {}
|
|
for f, src in srcs.items():
|
|
for n, v in go_package_consts(src):
|
|
if is_hu_text(v):
|
|
names[n] = (str(f.relative_to(root)), v)
|
|
if not names:
|
|
continue
|
|
alt = "|".join(map(re.escape, names))
|
|
use_re = re.compile(r"(strings\.(?:Contains|HasPrefix|HasSuffix|EqualFold|Index)\([^)]*\b(%s)\b|[=!]=\s*(%s)\b|\b(%s)\s*[=!]=)" % (alt, alt, alt))
|
|
for f, src in srcs.items():
|
|
for i, l in enumerate(src.split("\n"), 1):
|
|
if l.lstrip().startswith("//"):
|
|
continue
|
|
m = use_re.search(l)
|
|
if m:
|
|
n = m.group(2) or m.group(3) or m.group(4)
|
|
out.append({"file": str(f.relative_to(root)), "line": i, "name": n,
|
|
"defined": names[n][0], "text": names[n][1][:120]})
|
|
return out
|
|
|
|
|
|
def scan_go_files(files, root: Path):
|
|
rows = []
|
|
for f in files:
|
|
src = read(f)
|
|
lines = src.split("\n")
|
|
for ln, body, raw in go_literals(src):
|
|
if not is_hu_text(body):
|
|
continue
|
|
# A raw literal that is itself an embedded HTML page/template: count it, flag it.
|
|
lt = lines[ln - 1] if ln - 1 < len(lines) else ""
|
|
kind = classify_go(lt, body)
|
|
rows.append({
|
|
"file": str(f.relative_to(root)), "line": ln, "text": body[:160], "kind": kind,
|
|
"raw": raw, "multiline": "\n" in body,
|
|
"format": bool(FMT_VERB_RE.search(body)),
|
|
"numeric": bool(NUM_VERB_RE.search(body)),
|
|
"concat": bool(re.search(r'"\s*\+|\+\s*"', lt)) and not raw,
|
|
})
|
|
return rows
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------------
|
|
# HTML template scanner
|
|
# ---------------------------------------------------------------------------------------------------
|
|
|
|
ACTION_RE = re.compile(r"\{\{.*?\}\}", re.S)
|
|
CONTROL_RE = re.compile(r"\{\{-?\s*(if|else|end|range|with|define|block|template)\b.*?\}\}", re.S)
|
|
SCRIPT_RE = re.compile(r"<script\b[^>]*>(.*?)</script>", re.S | re.I)
|
|
STYLE_RE = re.compile(r"<style\b[^>]*>(.*?)</style>", re.S | re.I)
|
|
TAG_RE = re.compile(r"<[a-zA-Z!/][^>]*>", re.S)
|
|
ATTR_RE = re.compile(r'\s([a-zA-Z][a-zA-Z0-9:_-]*)\s*=\s*"([^"]*)"', re.S)
|
|
JS_STR_RE = re.compile(r"'(?:[^'\\\n]|\\.)*'|\"(?:[^\"\\\n]|\\.)*\"")
|
|
NUM_PARAM_RE = re.compile(r"\{\{\s*(len\b|\.\w*(Count|Num|Total|Days|Hours|Minutes|Size|N)\b)")
|
|
|
|
|
|
def line_of(src: str, pos: int) -> int:
|
|
return src.count("\n", 0, pos) + 1
|
|
|
|
|
|
def scan_template(f: Path, root: Path):
|
|
src = read(f)
|
|
rel = str(f.relative_to(root))
|
|
out = {"file": rel, "hu_lines": sum(1 for l in src.split("\n") if has_hu(l)),
|
|
"text": [], "attr": [], "js": [], "compare": [], "numeric": []}
|
|
|
|
# Mask script/style bodies (keep newlines so line numbers survive).
|
|
masked = src
|
|
for m in list(SCRIPT_RE.finditer(src)):
|
|
for jm in JS_STR_RE.finditer(m.group(1)):
|
|
s = jm.group(0)[1:-1]
|
|
if is_hu_text(ACTION_RE.sub("", s)):
|
|
out["js"].append({"line": line_of(src, m.start(1) + jm.start()), "text": s[:160]})
|
|
def blank(m):
|
|
return re.sub(r"[^\n]", " ", m.group(0))
|
|
masked = SCRIPT_RE.sub(blank, masked)
|
|
masked = STYLE_RE.sub(blank, masked)
|
|
|
|
# A compare is an eq/ne action whose OWN string arguments are Hungarian -- parsed inside the one
|
|
# action, never across the text between two actions.
|
|
for m in ACTION_RE.finditer(masked):
|
|
act = m.group(0)
|
|
if not re.search(r"\b(eq|ne)\b", act):
|
|
continue
|
|
for lit in re.findall(r'"((?:[^"\\]|\\.)*)"', act):
|
|
if is_hu_text(lit):
|
|
out["compare"].append({"line": line_of(src, m.start()), "text": lit})
|
|
|
|
for tm in TAG_RE.finditer(masked):
|
|
for am in ATTR_RE.finditer(tm.group(0)):
|
|
name, val = am.group(1).lower(), am.group(2)
|
|
if name in ("class", "id", "href", "src", "style", "name", "type", "for", "action", "method"):
|
|
continue
|
|
if is_hu_text(ACTION_RE.sub("", val)):
|
|
out["attr"].append({"line": line_of(src, tm.start()), "attr": name, "text": val[:160]})
|
|
|
|
# Text nodes: what lies between tags, split at control actions (if/range/...), with the
|
|
# remaining value actions kept in the message (they are its parameters).
|
|
text_only = TAG_RE.sub(lambda m: "\x00" + re.sub(r"[^\n]", " ", m.group(0))[1:], masked)
|
|
pos = 0
|
|
for chunk in text_only.split("\x00"):
|
|
start = pos
|
|
pos += len(chunk) + 1
|
|
for piece in CONTROL_RE.split(chunk):
|
|
if piece is None:
|
|
continue
|
|
p = piece.strip()
|
|
if not p or p in ("if", "else", "end", "range", "with", "define", "block", "template"):
|
|
continue
|
|
visible = ACTION_RE.sub("", p).strip()
|
|
if visible and is_hu_text(visible):
|
|
ln = line_of(src, start + max(chunk.find(p[:20]), 0))
|
|
out["text"].append({"line": ln, "text": " ".join(p.split())[:160]})
|
|
if NUM_PARAM_RE.search(p):
|
|
out["numeric"].append({"line": ln, "text": " ".join(p.split())[:160]})
|
|
return out
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------------
|
|
# Other surfaces
|
|
# ---------------------------------------------------------------------------------------------------
|
|
|
|
def walk_yaml(node, path=""):
|
|
if isinstance(node, dict):
|
|
for k, v in node.items():
|
|
yield from walk_yaml(v, f"{path}.{k}" if path else str(k))
|
|
elif isinstance(node, list):
|
|
for v in node:
|
|
yield from walk_yaml(v, path + "[]")
|
|
elif isinstance(node, str):
|
|
yield path, node
|
|
|
|
|
|
def controls(name: str, blobs: list[str], pos_fragment: str):
|
|
"""Positive: pos_fragment (ASCII) occurs, and some line holding it is flagged Hungarian.
|
|
Negative: NEG_FRAGMENT occurs nowhere, and an ASCII English line is not flagged."""
|
|
pos_lines = [l for b in blobs for l in b.split("\n") if pos_fragment in l]
|
|
pos_flagged = sum(1 for l in pos_lines if is_hu_text(l))
|
|
neg = sum(b.count(NEG_FRAGMENT) for b in blobs)
|
|
ok = len(pos_lines) > 0 and pos_flagged > 0 and neg == 0 and not is_hu_text(NEG_ASCII_LINE)
|
|
return {"surface": name, "positive_fragment": pos_fragment, "positive_lines": len(pos_lines),
|
|
"positive_flagged": pos_flagged, "negative_fragment": NEG_FRAGMENT, "negative_hits": neg,
|
|
"negative_ascii_line_flagged": is_hu_text(NEG_ASCII_LINE), "ok": ok}
|
|
|
|
|
|
def top10(strings):
|
|
c = collections.Counter(" ".join(s.split()) for s in strings)
|
|
return [(t, n) for t, n in c.most_common(10) if n > 1]
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--root", default="/mnt/5_hdd/felhom.eu/git")
|
|
ap.add_argument("--json", default=None)
|
|
args = ap.parse_args()
|
|
root = Path(args.root)
|
|
ctl = root / "felhom-controller" / "controller"
|
|
hub = root / "felhom.eu" / "hub"
|
|
report = {"script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), "surfaces": {}}
|
|
ctrls = []
|
|
|
|
# 1-3. Controller templates (+ inline JS) -------------------------------------------------------
|
|
tdir = ctl / "internal" / "web" / "templates"
|
|
tfiles = sorted(tdir.glob("*.html"))
|
|
tres = [scan_template(f, root) for f in tfiles]
|
|
report["surfaces"]["templates"] = tres
|
|
ctrls.append(controls("controller templates", [read(f) for f in tfiles], "pult"))
|
|
|
|
# 2. Controller Go (non-test), excluding the first-boot wizard ------------------------------------
|
|
gofiles = sorted(p for p in ctl.rglob("*.go")
|
|
if not p.name.endswith("_test.go") and "/internal/setup/" not in str(p))
|
|
gores = scan_go_files(gofiles, root)
|
|
report["surfaces"]["controller_go"] = gores
|
|
indirect = go_indirect_compares(gofiles, root)
|
|
report["surfaces"]["controller_go_indirect_compares"] = indirect
|
|
ctrls.append(controls("controller Go", [read(f) for f in gofiles], "ment"))
|
|
|
|
# 4. Hub customer surfaces ------------------------------------------------------------------------
|
|
hub_customer = sorted(list((hub / "internal" / "notify").glob("*.go"))
|
|
+ list((hub / "internal" / "web").glob("selfbind*.go"))
|
|
+ [hub / "internal" / "web" / "customer_reset.go", hub / "internal" / "claim" / "engine.go"])
|
|
hub_customer = [p for p in hub_customer if p.exists() and not p.name.endswith("_test.go")]
|
|
hubres = scan_go_files(hub_customer, root)
|
|
report["surfaces"]["hub_customer_go"] = hubres
|
|
hub_all = sorted(p for p in hub.rglob("*.go") if not p.name.endswith("_test.go"))
|
|
hub_all_rows = scan_go_files(hub_all, root)
|
|
hub_tpl = [scan_template(f, root) for f in sorted((hub / "internal" / "web" / "templates").glob("*.html"))]
|
|
report["surfaces"]["hub_operator_templates"] = hub_tpl
|
|
ctrls.append(controls("hub customer Go", [read(f) for f in hub_customer], "ment"))
|
|
|
|
# 5. Installer ------------------------------------------------------------------------------------
|
|
banner = root / "felhom.eu" / "scripts" / "iso" / "felhom-bootstrap.sh"
|
|
btext = read(banner)
|
|
banner_lines = [{"line": i + 1, "text": l.strip()[:160]} for i, l in enumerate(btext.split("\n")) if is_hu_text(l)]
|
|
dl = root / "felhom.eu" / "website" / "letoltes.html"
|
|
dlres = scan_template(dl, root) if dl.exists() else None
|
|
report["surfaces"]["installer_banner"] = banner_lines
|
|
report["surfaces"]["download_page"] = dlres
|
|
ctrls.append(controls("installer banner", [btext], "Felhom"))
|
|
|
|
# 6. Catalog --------------------------------------------------------------------------------------
|
|
cat = []
|
|
catfiles = sorted((root / "app-catalog-felhom.eu" / "templates").glob("*/.felhom.yml"))
|
|
for f in catfiles:
|
|
try:
|
|
doc = yaml.safe_load(read(f)) or {}
|
|
except yaml.YAMLError as e:
|
|
cat.append({"file": str(f.relative_to(root)), "path": "<YAML ERROR>", "text": str(e)[:120]})
|
|
continue
|
|
for p, v in walk_yaml(doc):
|
|
if is_hu_text(v):
|
|
cat.append({"file": str(f.relative_to(root)), "path": re.sub(r"\[\]", "[]", p), "text": v[:160],
|
|
"words": len(v.split())})
|
|
report["surfaces"]["catalog"] = cat
|
|
ctrls.append(controls("catalog", [read(f) for f in catfiles], "Alkalmaz"))
|
|
|
|
# 7. Guide ----------------------------------------------------------------------------------------
|
|
guide = root / "felhom.eu" / "documentation" / "runbooks" / "VOLUNTEER-first-hour.md"
|
|
gtext = read(guide)
|
|
paras = [p for p in re.split(r"\n\s*\n", gtext) if p.strip()]
|
|
report["surfaces"]["guide"] = {"file": str(guide.relative_to(root)), "paragraphs": len(paras),
|
|
"hu_paragraphs": sum(1 for p in paras if is_hu_text(p)),
|
|
"words": len(gtext.split())}
|
|
ctrls.append(controls("guide", [gtext], "Felhom"))
|
|
|
|
# 8. First-boot wizard ----------------------------------------------------------------------------
|
|
sdir = ctl / "internal" / "setup"
|
|
stpl = [scan_template(f, root) for f in sorted((sdir / "templates").glob("*.html"))]
|
|
sgo = scan_go_files(sorted(p for p in sdir.rglob("*.go") if not p.name.endswith("_test.go")), root)
|
|
report["surfaces"]["setup_wizard"] = {"templates": stpl, "go": sgo}
|
|
|
|
# Dates and sizes ---------------------------------------------------------------------------------
|
|
fmt_sites = []
|
|
for f in sorted((ctl / "internal" / "web").glob("*.go")):
|
|
if f.name.endswith("_test.go"):
|
|
continue
|
|
for i, l in enumerate(read(f).split("\n"), 1):
|
|
for m in re.finditer(r'\.Format\("([^"]*)"\)', l):
|
|
fmt_sites.append({"file": str(f.relative_to(root)), "line": i, "layout": m.group(1)})
|
|
for f in tfiles:
|
|
for i, l in enumerate(read(f).split("\n"), 1):
|
|
for m in re.finditer(r'\.Format\s+"([^"]*)"|Format\s+"([^"]*)"', l):
|
|
fmt_sites.append({"file": str(f.relative_to(root)), "line": i, "layout": m.group(1) or m.group(2)})
|
|
fmtmb = []
|
|
for f in tfiles + sorted((ctl / "internal" / "web").glob("*.go")):
|
|
if f.name.endswith("_test.go"):
|
|
continue
|
|
for i, l in enumerate(read(f).split("\n"), 1):
|
|
if "fmtMB" in l:
|
|
fmtmb.append({"file": str(f.relative_to(root)), "line": i})
|
|
ctl_fmt_elsewhere = 0
|
|
for f in ctl.rglob("*.go"):
|
|
if f.name.endswith("_test.go") or "/internal/web/" in str(f):
|
|
continue
|
|
ctl_fmt_elsewhere += len(re.findall(r'\.Format\("', read(f)))
|
|
report["dates"] = fmt_sites
|
|
report["fmtMB"] = fmtmb
|
|
report["format_calls_outside_web"] = ctl_fmt_elsewhere
|
|
report["controls"] = ctrls
|
|
|
|
# ---------------------------------------------------------------------------------------------
|
|
# Markdown summary
|
|
# ---------------------------------------------------------------------------------------------
|
|
P = print
|
|
P(f"script sha256: {report['script_sha256']}\n")
|
|
P("## Controls (ASCII fragment, positive + negative)\n")
|
|
P("| surface | positive fragment | lines holding it | of those flagged | negative fragment hits | ASCII English line flagged | verdict |")
|
|
P("|---|---|---|---|---|---|---|")
|
|
for c in ctrls:
|
|
P(f"| {c['surface']} | `{c['positive_fragment']}` | {c['positive_lines']} | {c['positive_flagged']} | "
|
|
f"{c['negative_hits']} | {c['negative_ascii_line_flagged']} | {'ok' if c['ok'] else 'BROKEN'} |")
|
|
|
|
def tsum(rs):
|
|
return (sum(r["hu_lines"] for r in rs), sum(len(r["text"]) for r in rs), sum(len(r["attr"]) for r in rs),
|
|
sum(len(r["js"]) for r in rs), sum(len(r["compare"]) for r in rs), sum(len(r["numeric"]) for r in rs),
|
|
sum(1 for r in rs if r["hu_lines"] or r["text"] or r["attr"] or r["js"]))
|
|
|
|
P("\n## Controller page templates\n")
|
|
hl, tx, at, js, cm, nu, withhu = tsum(tres)
|
|
P(f"{len(tres)} files; {withhu} carry Hungarian. Lines with an accented letter: **{hl}**. "
|
|
f"Strings: **{tx} text runs + {at} attribute values + {js} inline-JS literals = {tx + at + js}**. "
|
|
f"Template compares against Hungarian: {cm}. Text runs with a numeric parameter: {nu}.\n")
|
|
P("| template | HU lines | text | attr | JS | compare |")
|
|
P("|---|---|---|---|---|---|")
|
|
for r in sorted(tres, key=lambda r: -(len(r["text"]) + len(r["attr"]) + len(r["js"]))):
|
|
P(f"| {Path(r['file']).name} | {r['hu_lines']} | {len(r['text'])} | {len(r['attr'])} | {len(r['js'])} | {len(r['compare'])} |")
|
|
P("\nMost repeated (text + attr):")
|
|
for t, n in top10([x["text"] for r in tres for x in r["text"] + r["attr"]]):
|
|
P(f"- {n}x `{t}`")
|
|
|
|
def gosum(rows, label):
|
|
k = collections.Counter(r["kind"] for r in rows)
|
|
files = len({r["file"] for r in rows})
|
|
P(f"\n## {label}\n")
|
|
P(f"**{len(rows)} literals in {files} files.** shown: {k['shown']}, error (may reach a page): {k['error']}, "
|
|
f"log only: {k['log']}, compared: {k['compare']}. Format strings with parameters (non-log): "
|
|
f"{sum(1 for r in rows if r['format'] and r['kind'] != 'log')}; numeric %d (non-log, plural risk): "
|
|
f"{sum(1 for r in rows if r['numeric'] and r['kind'] != 'log')}; concatenated (non-log): "
|
|
f"{sum(1 for r in rows if r['concat'] and r['kind'] != 'log')}; raw multi-line blocks: "
|
|
f"{sum(1 for r in rows if r['multiline'])}.\n")
|
|
byfile = collections.Counter(r["file"] for r in rows if r["kind"] != "log")
|
|
P("| file (top 15, non-log) | literals |")
|
|
P("|---|---|")
|
|
for fname, n in byfile.most_common(15):
|
|
P(f"| {fname} | {n} |")
|
|
P("\nMost repeated (non-log):")
|
|
for t, n in top10([r["text"] for r in rows if r["kind"] != "log"]):
|
|
P(f"- {n}x `{t}`")
|
|
return k
|
|
|
|
gosum(gores, "Controller Go (non-test, first-boot wizard excluded)")
|
|
P("\n### Compared, not shown (controller Go)\n")
|
|
for r in gores:
|
|
if r["kind"] == "compare":
|
|
P(f"- {r['file']}:{r['line']} `{r['text']}`")
|
|
for r in indirect:
|
|
P(f"- {r['file']}:{r['line']} via `{r['name']}` (defined {r['defined']}) `{r['text']}`")
|
|
P("\n### Compared, not shown (templates)\n")
|
|
for r in tres:
|
|
for x in r["compare"]:
|
|
P(f"- {r['file']}:{x['line']} `{x['text']}`")
|
|
|
|
gosum(hubres, "Hub customer surfaces (notify, self-bind, reset, claim)")
|
|
hk = collections.Counter(r["kind"] for r in hub_all_rows)
|
|
hl2, tx2, at2, js2, _, _, _ = tsum(hub_tpl)
|
|
P(f"\nFor scale, NOT in scope: all hub Go non-test = {len(hub_all_rows)} literals "
|
|
f"(shown {hk['shown']}, error {hk['error']}, log {hk['log']}, compare {hk['compare']}); "
|
|
f"hub operator templates = {tx2 + at2 + js2} strings, {hl2} HU lines.")
|
|
|
|
P("\n## Installer\n")
|
|
P(f"Console banner `{banner.relative_to(root)}`: **{len(banner_lines)}** Hungarian lines.")
|
|
if dlres:
|
|
P(f"Download page `{dl.relative_to(root)}`: {len(dlres['text'])} text + {len(dlres['attr'])} attr + "
|
|
f"{len(dlres['js'])} JS = **{len(dlres['text']) + len(dlres['attr']) + len(dlres['js'])}** strings, "
|
|
f"{dlres['hu_lines']} HU lines.")
|
|
|
|
P("\n## Catalog\n")
|
|
byfield = collections.Counter(re.sub(r"^.*?\b(app_info\.\w+|deploy_fields\[\]\.\w+|\w+)$", r"\1", r["path"]) for r in cat)
|
|
apps = len({r["file"] for r in cat})
|
|
words = sum(r.get("words", 0) for r in cat)
|
|
P(f"{len(catfiles)} `.felhom.yml`; {apps} carry Hungarian; **{len(cat)} strings, ~{words} words**.\n")
|
|
P("| field | strings |")
|
|
P("|---|---|")
|
|
for fld, n in byfield.most_common(20):
|
|
P(f"| {fld} | {n} |")
|
|
|
|
g = report["surfaces"]["guide"]
|
|
P(f"\n## Guide\n\n`{g['file']}`: {g['paragraphs']} paragraphs ({g['hu_paragraphs']} Hungarian), ~{g['words']} words.")
|
|
|
|
P("\n## First-boot wizard (reported separately -- never merged into the dashboard count)\n")
|
|
shl, stx, sat, sjs, scm, _, _ = tsum(stpl)
|
|
sk = collections.Counter(r["kind"] for r in sgo)
|
|
P(f"{len(stpl)} templates: {shl} HU lines; {stx} text + {sat} attr + {sjs} JS = **{stx + sat + sjs}** strings. "
|
|
f"Go: {len(sgo)} literals (shown {sk['shown']}, error {sk['error']}, log {sk['log']}, compare {sk['compare']}).")
|
|
|
|
P("\n## Dates and sizes\n")
|
|
lay = collections.Counter((Path(x["file"]).suffix, x["layout"]) for x in fmt_sites)
|
|
P("| where | layout | sites |")
|
|
P("|---|---|---|")
|
|
for (suf, l), n in sorted(lay.items(), key=lambda kv: -kv[1]):
|
|
P(f"| {'template' if suf == '.html' else 'internal/web Go'} | `{l}` | {n} |")
|
|
P(f"\n`fmtMB` sites: {sum(1 for x in fmtmb if x['file'].endswith('.html'))} in templates, "
|
|
f"{sum(1 for x in fmtmb if x['file'].endswith('.go'))} in Go. `.Format(\"` calls in the controller outside "
|
|
f"internal/web: {ctl_fmt_elsewhere}.")
|
|
|
|
if args.json:
|
|
Path(args.json).write_text(json.dumps(report, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
bad = [c for c in ctrls if not c["ok"]]
|
|
if bad:
|
|
P("\nCONTROLS BROKEN: " + ", ".join(c["surface"] for c in bad), file=sys.stderr)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|