Files
felhom.eu/scripts/site_gates.py
T
admin 67b3fd23d1
gates / gates (push) Failing after 13m1s
website refresh: true numbers (56 apps), claims checked against the capability map, ten missing apps, an English twin of every public page (public, choice B), 404 page; site gates 13-18 + 11 decoys; R-902 opened
Part A: claim table documentation/audits/website-refresh-2026-10-08/claims.md; "100% open source" corrected
from the licence read; unbacked claims cut (firewall, RAID, snapshots, new-machine restore, self-managed
mode, household VPN, "never lost"). Part B: dawarich, docmost, grimmory, homebox, karakeep, mealie, metube,
radicale, recipe-importer, sparkyfitness cards; plant-it and wger cut (not offered). Part C: /en/ twins,
nav language switch, hreflang both ways, sitemap with xhtml:link, og-image-en.png. Part D: site_gates.py
twins/lang/same-apps/no-Hungarian/FAQ-JSON-LD/tail. Operator ruling 9 + choice B in 10-localisation.md §11,
§10.7. Register: R-902 (contact mailer source in no repo); R-813, R-784, R-559 annotated.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_012qRErfCoiTkvDK9N5XHbzb
2026-10-08 08:29:17 +02:00

402 lines
20 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""TASK-D3 site gates — mechanical checks against the static site's known failure mode:
silent per-page drift. Run from the repo root: python scripts/site_gates.py
Gates (all must pass; non-zero exit on any failure):
1. BOM — every website/*.html begins with EF BB BF (byte-checked)
2. emoji — zero emoji/pictographs in website/*.html (codepoint ranges; NEVER grep —
Windows grep false-negatives multibyte emoji, proven in D0)
3. nav — the <nav>…</nav> and <footer>…</footer> blocks are identical PER LANGUAGE SET
(Hungarian pages against index.html, English pages against en/index.html) after
stripping the active-link marker and the language switch's target
4. analytics — the umami snippet is present on every public page (nonpublic draft exempt)
5. no-CDN — zero fonts.googleapis.com / fonts.gstatic.com references
6. banned — zero legacy hexes / 999px radius / box-shadow in pages + site.css
7. style — zero embedded <style> blocks in pages (everything lives in site.css)
8. cachebust — every site.css / icons.svg reference carries ?v=<N>
12. download — the two download pages name the same installer file and checksum (R-559)
13. twins — every public Hungarian page has its English twin and back; each pair carries the same
hu/en/x-default alternates; each page's language switch points at its own twin
14. lang — <html lang="hu"> on the Hungarian set, lang="en" on the English set
15. same apps — the app logos on /alkalmazasok and /en/apps are the same set
16. no HU — no Hungarian in the English pages: accent scan PLUS an ASCII-folded, word-bounded stem
scan (an accent-only scan passes ASCII-only Hungarian — 10-localisation.md §7), over
visible text, attributes, JSON-LD and the inline scripts' string literals. Positive and
negative controls are printed on every run. Also: no "please"/"kindly", and no English
retrieval promise beyond what the Hungarian twin makes
17. faq-ld — each FAQ page's FAQPage JSON-LD carries exactly its visible questions and answers
18. tail — nothing after </html> (a stray file-dump summary sat visible below technologiak.html)
"""
import html as _html, io, json, os, re, sys, unicodedata
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
W = os.path.join(ROOT, "website")
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from customer_copy_vocab import RETRIEVAL_STEMS as _RET_HU, RETRIEVAL_STEMS_EN as _RET_EN
# The public pairs: Hungarian page -> its English twin (2026-10-08, operator ruling 9 in 10-localisation.md §11:
# the whole public site gets an English twin; the Hungarian stays x-default). A new public page needs its twin
# in the same commit — gate 13 refuses one without.
TWINS = {
"index.html": os.path.join("en", "index.html"),
"alkalmazasok.html": os.path.join("en", "apps.html"),
"technologiak.html": os.path.join("en", "technology.html"),
"biztonsagimentes.html": os.path.join("en", "backups.html"),
"gyik.html": os.path.join("en", "faq.html"),
"kapcsolat.html": os.path.join("en", "contact.html"),
"letoltes.html": os.path.join("en", "download.html"), # R-559 (2026-09-18) — the first twin
}
URL = {"index.html": "/", "alkalmazasok.html": "/alkalmazasok", "technologiak.html": "/technologiak",
"biztonsagimentes.html": "/biztonsagimentes", "gyik.html": "/gyik", "kapcsolat.html": "/kapcsolat",
"letoltes.html": "/letoltes", os.path.join("en", "index.html"): "/en/", os.path.join("en", "apps.html"): "/en/apps",
os.path.join("en", "technology.html"): "/en/technology", os.path.join("en", "backups.html"): "/en/backups",
os.path.join("en", "faq.html"): "/en/faq", os.path.join("en", "contact.html"): "/en/contact",
os.path.join("en", "download.html"): "/en/download"}
# Pages with no twin, BY NAME: the hidden prices page stays Hungarian only (brief 2026-10-08), and the 404 page
# is served at whatever URL was wrong — it carries one English line instead of a twin.
NO_TWIN = {"szolgaltatasok-nonpublic.html", "404.html"}
PAGES = list(TWINS) + list(TWINS.values()) + sorted(NO_TWIN)
EN_PAGES = set(TWINS.values())
HU_PAGES = set(PAGES) - EN_PAGES
ANALYTICS_EXEMPT = {"szolgaltatasok-nonpublic.html"}
ANALYTICS_MARK = "https://stats.felhom.eu/script.js"
SITE = "https://felhom.eu"
BANNED = ["fonts.googleapis.com", "fonts.gstatic.com",
"#0d1117", "#161b22", "#1c2128", "#30363d", "#238636", "#da3633", "#d29922",
"border-radius: 999px", "box-shadow"]
fails = []
def fail(msg):
fails.append(msg)
print("FAIL:", msg)
def is_emoji(ch):
o = ord(ch)
return any(a <= o <= b for a, b in [
(0x1F000, 0x1FAFF), # pictographs, emoticons, transport, symbols-ext, cards
(0x2600, 0x27BF), # misc symbols + dingbats (incl. ✓ ✗ ★ ⚠ — sprite icons instead)
(0x2300, 0x23FF), # misc technical (⏱ ⏳ ⌛ …)
(0x2B00, 0x2BFF), # ⭐ etc.
(0xFE00, 0xFE0F), # variation selectors
(0x1F1E6, 0x1F1FF), # regional indicators
])
SWITCH_RE = re.compile(r'<a href="([^"]*)" lang="(hu|en)" hreflang="(hu|en)" class="lang-switch">([^<]*)</a>')
def norm_block(b):
b = b.replace(' class="active"', "")
b = re.sub(r'\s*aria-current="[^"]*"', "", b)
# the language switch is the one nav item whose target differs per page (it points at the page's own
# twin) — gate 13 checks that target; here only its presence and wording must match
b = SWITCH_RE.sub(lambda m: '<a lang-switch="%s">%s</a>' % (m.group(2), m.group(4)), b)
return b
# R-423 (2026-10-05): PAGES is still the list the checks run on, but it may no longer be SHORTER than the site. Every
# *.html under website/ (a walk, any depth) must be in it — a page added without being added here used to go
# unscanned, which is how a gate checks seven files of nine and prints OK.
_on_disk = set()
for _dp, _dns, _fns in os.walk(W):
_dns[:] = [d for d in _dns if not d.startswith(".")]
for _f in _fns:
if _f.endswith(".html"):
_on_disk.add(os.path.relpath(os.path.join(_dp, _f), W))
for _missing in sorted(_on_disk - set(PAGES)):
fail("%s: an HTML page the site gates do not scan — add it to TWINS (with its twin) or NO_TWIN in "
"scripts/site_gates.py (R-423)" % _missing)
pages = {}
for p in PAGES:
path = os.path.join(W, p)
if not os.path.exists(path):
fail("%s: listed in the site gates but missing from website/" % p)
continue
raw = io.open(path, "rb").read()
# gate 1: BOM
if raw[:3] != b"\xef\xbb\xbf":
fail("%s: missing UTF-8 BOM (first bytes: %s)" % (p, raw[:3].hex()))
pages[p] = raw.decode("utf-8-sig")
# gate 2: emoji (pages + the shared stylesheet)
total_emoji = 0
emoji_targets = dict(pages)
_css = os.path.join(W, "assets", "site.css")
if os.path.exists(_css):
emoji_targets["assets/site.css"] = io.open(_css, encoding="utf-8").read()
for p, s in emoji_targets.items():
hits = [(i, ch) for i, ch in enumerate(s) if is_emoji(ch)]
if hits:
total_emoji += len(hits)
sample = " ".join(ch for _, ch in hits[:10])
fail("%s: %d emoji (%s ...)" % (p, len(hits), sample))
if total_emoji:
print(" emoji total: %d" % total_emoji)
# gate 3: nav + footer consistency, per language set
for lang, members, ref in (("hu", HU_PAGES, "index.html"), ("en", EN_PAGES, os.path.join("en", "index.html"))):
if ref not in pages:
continue
blocks = {}
for p in sorted(members, key=lambda x: (x != ref, x)):
s = pages.get(p)
if s is None:
continue
nm = re.search(r"<nav>.*?</nav>", s, re.DOTALL)
fm = re.search(r"<footer>.*?</footer>", s, re.DOTALL)
if not nm or not fm:
fail("%s: missing <nav> or <footer>" % p)
continue
blocks[p] = (norm_block(nm.group(0)), norm_block(fm.group(0)))
rnav, rfoot = blocks.get(ref, (None, None))
for p, (nav, foot) in blocks.items():
if nav != rnav:
fail("%s: <nav> differs from %s (after active-marker normalization)" % (p, ref))
if foot != rfoot:
fail("%s: <footer> differs from %s" % (p, ref))
if os.path.join("en", "index.html") in pages:
_f = re.search(r"<footer>.*?</footer>", pages[os.path.join("en", "index.html")], re.DOTALL)
if not _f or "households in Hungary" not in _f.group(0):
fail("en/index.html: the English footer no longer says that Felhom serves households in Hungary "
"(the brief's rule: every English page says it once — the shared footer is where it lives)")
# gate 4: analytics presence
for p, s in pages.items():
if p in ANALYTICS_EXEMPT:
continue
if ANALYTICS_MARK not in s:
fail("%s: analytics snippet missing (%s)" % (p, ANALYTICS_MARK))
# gate 5+6: banned strings in pages + site.css
targets = dict(pages)
css_path = os.path.join(W, "assets", "site.css")
if os.path.exists(css_path):
targets["assets/site.css"] = io.open(css_path, encoding="utf-8").read()
for name, s in targets.items():
low = s.lower()
for pat in BANNED:
c = low.count(pat.lower())
if c:
fail("%s: banned %r ×%d" % (name, pat, c))
# gate 7: no embedded style blocks
for p, s in pages.items():
c = len(re.findall(r"<style[\s>]", s))
if c:
fail("%s: %d embedded <style> block(s)" % (p, c))
# gate 12 (R-559): the two download pages name the SAME installer file and the SAME checksum.
#
# THE FAILURE THIS EXISTS FOR IS A HALF-DONE RELEASE. Publishing a new ISO means editing a filename,
# a size and a 64-character hash on TWO pages now. Update one and forget the other and an English
# reader downloads yesterday's image, or — worse — checks today's image against yesterday's hash,
# fails the comparison, and is told by the page itself not to use the file.
#
# It compares what the pages SAY, not what is published: a gate cannot reach the bucket, and a
# published-file check belongs to the ISO release gate (G11), which does exactly that.
_HU_DL, _EN_DL = "letoltes.html", os.path.join("en", "download.html")
if _HU_DL in pages and _EN_DL in pages:
_iso_re = re.compile(r"felhom-installer-[0-9.]+-pve[0-9.\-]+\.iso")
_sha_re = re.compile(r"\b[0-9a-f]{64}\b")
hu_iso = sorted(set(_iso_re.findall(pages[_HU_DL])))
en_iso = sorted(set(_iso_re.findall(pages[_EN_DL])))
hu_sha = sorted(set(_sha_re.findall(pages[_HU_DL])))
en_sha = sorted(set(_sha_re.findall(pages[_EN_DL])))
if not hu_iso:
fail("%s: names no installer file at all" % _HU_DL)
if not hu_sha:
fail("%s: names no SHA-256 at all" % _HU_DL)
if hu_iso != en_iso:
fail("the download pages name DIFFERENT installer files: hu=%s en=%s" % (hu_iso, en_iso))
if hu_sha != en_sha:
fail("the download pages name DIFFERENT checksums: hu=%s en=%s" % (hu_sha, en_sha))
else:
print(" download pages agree: %s, sha %s…" % (", ".join(hu_iso), (hu_sha or ["-"])[0][:12]))
# gate 8: cache-busting on shared assets
for p, s in pages.items():
for m in re.finditer(r"/assets/(site\.css|icons\.svg)([^\"'#\s>]*)", s):
if not m.group(2).startswith("?v="):
fail("%s: %s referenced without ?v= (got %r)" % (p, m.group(1), m.group(0)))
# gate 13: twins — both ways, matching alternates, the switch at the twin
ALT_RE = re.compile(r'<link rel="alternate" hreflang="([^"]+)" href="([^"]+)">')
for hu, en in TWINS.items():
if hu not in pages or en not in pages:
fail("twin pair %s <-> %s: a page of the pair is missing" % (hu, en))
continue
want = {"hu": SITE + URL[hu], "en": SITE + URL[en], "x-default": SITE + URL[hu]}
for p, other, lang in ((hu, en, "en"), (en, hu, "hu")):
got = dict(ALT_RE.findall(pages[p]))
if got != want:
fail("%s: hreflang alternates %s — the pair %s <-> %s needs exactly %s" % (p, got, hu, en, want))
sw = SWITCH_RE.findall(pages[p])
if len(sw) != 1:
fail("%s: %d language switch links in the page — exactly one, in the nav" % (p, len(sw)))
elif sw[0][0] != URL[other] or sw[0][1] != lang or sw[0][2] != lang:
fail("%s: the language switch points at %r (lang=%s) — its twin is %r (lang=%s)"
% (p, sw[0][0], sw[0][1], URL[other], lang))
canon = re.search(r'<link rel="canonical" href="([^"]+)">', pages[p])
if not canon or canon.group(1) != SITE + URL[p]:
fail("%s: canonical %r — it must point at the page itself (%s)" % (p, canon and canon.group(1), SITE + URL[p]))
print(" twins: %d pairs checked both ways (hreflang, switch, canonical)" % len(TWINS))
# gate 14: the lang attribute
for p, s in pages.items():
m = re.search(r'<html lang="([^"]+)"', s)
want = "en" if p in EN_PAGES else "hu"
if not m or m.group(1) != want:
fail("%s: <html lang=%r> — the %s set needs lang=%r" % (p, m and m.group(1), "English" if want == "en" else "Hungarian", want))
# gate 15: the same apps both ways
def app_logos(s):
return set(re.findall(r'<img src="/assets/([^"]+)" alt="[^"]*" class="app-logo"', s))
_A_HU, _A_EN = "alkalmazasok.html", os.path.join("en", "apps.html")
if _A_HU in pages and _A_EN in pages:
lh, le = app_logos(pages[_A_HU]), app_logos(pages[_A_EN])
cards_hu = len(re.findall(r'<div class="app-card">', pages[_A_HU]))
cards_en = len(re.findall(r'<div class="app-card">', pages[_A_EN]))
if lh != le or cards_hu != cards_en:
fail("the apps pages show DIFFERENT apps: only hu %s, only en %s, cards hu=%d en=%d"
% (sorted(lh - le), sorted(le - lh), cards_hu, cards_en))
else:
print(" apps pages agree: %d cards, %d logos" % (cards_hu, len(lh)))
# gate 16: no Hungarian on the English pages
def fold(s):
s = unicodedata.normalize("NFKD", s)
return "".join(c for c in s if not unicodedata.combining(c)).lower()
HU_LETTER = re.compile(u"[áéíóöőúüűÁÉÍÓÖŐÚÜŰ]")
# ASCII-folded stems. Stems of five letters or more match the START of a word ("jelentkez" convicts
# "Jelentkezz"); short ones, and the two language names, match WHOLE words only — "angol" must not convict
# "Angola", nor "az" convict "Amazon". Reused approach: app-catalog-felhom.eu/scripts/check-copy-i18n.py.
HU_STEMS = ["aldomain", "jelentkez", "oszd meg", "kattints", "valaszd", "magyar", "angol", "felhasznalo", "jelszo",
"nyelv", "beallitas", "alkalmazas", "szerver", "fajl", "mappa", "megosztas", "mentes", "doboz",
"vezerlopult", "kapcsolat", "otthon", "szolgaltatas", "biztonsag", "telepit", "haztartas", "kerdes",
"valasz", "letoltes", "tovabb", "ingyenes", "nyilt", "forraskod", "tarhely", "frissites", "uzemeltet",
"technologia", "hamarosan", "kerlek", "koszon", "uzenet", "targy",
"es", "egy", "vagy", "nem", "igen", "az", "hogy", "mint", "csak", "itt", "ahol", "ezt", "ami", "mert",
"nev", "gyik"]
_WHOLE = {"angol", "magyar"}
HU_RX = [(st, re.compile(r"(?<![a-z0-9])%s" % re.escape(st) + (r"(?![a-z0-9])" if len(st) <= 4 or st in _WHOLE else "")))
for st in HU_STEMS]
BEG = re.compile(r"\b(please|kindly)\b", re.I)
def visible_strings(s):
"""Everything a reader can meet: text, attribute values, JSON-LD, and the string literals of inline scripts
(status messages are shown to the visitor). Not scanned: comments, URLs, and anything marked lang="hu"
(the switch back to Magyar)."""
js = " ".join(re.findall(r"<script(?![^>]*ld\+json)[^>]*>(.*?)</script>", s, flags=re.S))
js = re.sub(r"(?m)^\s*//.*$", " ", js)
lits = [a or b or c for a, b, c in re.findall(r"'([^'\n]*)'|`([^`]*)`|\"([^\"\n]*)\"", js)]
s = re.sub(r"<script(?![^>]*ld\+json)[^>]*>.*?</script>", " ", s, flags=re.S)
s = re.sub(r"<!--.*?-->", " ", s, flags=re.S)
s = re.sub(r'<(?!html\b)([a-z0-9]+)\b[^>]*\blang="hu"[^>]*>.*?</\1>', " ", s, flags=re.S)
attrs = re.findall(r'\b(?:alt|title|aria-label|placeholder|content)="([^"]*)"', s)
s = re.sub(r'(https?://|/)[^\s"<>]*', " ", s)
text = re.sub(r"<[^>]+>", "\n", s)
return [_html.unescape(t).strip() for t in text.split("\n") + attrs + lits if t.strip()]
def hungarian_hits(s):
out = []
for t in visible_strings(s):
t2 = re.sub(r"\b[\w.-]+\.hu\b", " ", t) # Hungarian site names (mindmegette.hu …) are names
if HU_LETTER.search(t2):
out.append(("an accented Hungarian letter", t))
continue
f = fold(t2)
for st, rx in HU_RX:
if rx.search(f):
out.append(("the Hungarian stem %r" % st, t))
break
return out
# controls, printed on every run (workspace rule: a zero from a Hungarian search is suspect without them)
_pos = hungarian_hits(u"<p>Jelentkezz be: admin</p>")
_pos2 = hungarian_hits(u"<p>Kapcsolat</p>")
_neg = hungarian_hits(u"<p>The third best option is in Angola. These documents are kept on the server. "
u"Amazon is not an apps page.</p>")
if not _pos or not _pos2 or _neg:
fail("gate 16 INCONCLUSIVE: matcher controls failed (positive %r/%r must convict, negative %r must not)"
% (_pos, _pos2, _neg))
else:
print(" no-HU matcher controls OK (positive 'Jelentkezz be'/'Kapcsolat' convict, negative 'Angola…' does not)")
for p in sorted(EN_PAGES):
s = pages.get(p)
if s is None:
continue
for why, t in hungarian_hits(s)[:5]:
fail("%s: Hungarian on an English page (%s): %r" % (p, why, t[:120]))
for t in visible_strings(s):
if BEG.search(t):
fail("%s: the site does not beg — no \"please\"/\"kindly\": %r" % (p, t[:120]))
# the retrieval promise: the English page may make it no more often than its Hungarian twin does
hu = next((h for h, e in TWINS.items() if e == p), None)
if hu and hu in pages:
en_n = sum(len(re.findall(rx, t, re.I)) for t in visible_strings(s) for rx in _RET_EN)
hu_n = sum(fold(t).count(fold(st)) for t in visible_strings(pages[hu]) for st in _RET_HU)
if en_n > hu_n:
fail("%s: %d English retrieval promise(s) ('can be restored', 'recoverable' …) against %d in %s — "
"the English promises nothing the Hungarian does not" % (p, en_n, hu_n, hu))
# gate 17: FAQ JSON-LD == visible questions and answers
def faq_pairs(s):
out = []
for q, a in re.findall(r'<button class="faq-question">\s*(.*?)\s*<svg class="faq-chevron".*?'
r'<div class="faq-answer-inner">(.*?)</div></div>', s, re.S):
parts = re.findall(r"<(?:p|li)[^>]*>(.*?)</(?:p|li)>", a, re.S)
out.append((_html.unescape(q.strip()),
" ".join(re.sub(r"\s+", " ", _html.unescape(re.sub(r"<[^>]+>", "", x))).strip() for x in parts)))
return out
for p in ("gyik.html", os.path.join("en", "faq.html")):
s = pages.get(p)
if s is None:
continue
m = re.search(r'<script type="application/ld\+json">(.*?)</script>', s, re.S)
try:
ld = json.loads(m.group(1)) if m else {}
except ValueError as e:
fail("%s: the FAQPage JSON-LD does not parse: %s" % (p, e))
continue
got = [(e.get("name"), (e.get("acceptedAnswer") or {}).get("text")) for e in ld.get("mainEntity", [])]
vis = faq_pairs(s)
if not vis:
fail("%s: found no visible FAQ items — the gate's pattern no longer matches the page" % p)
elif got != vis:
diff = [(i, a, b) for i, (a, b) in enumerate(zip(got, vis)) if a != b][:2]
fail("%s: the FAQPage JSON-LD (%d items) differs from the visible questions/answers (%d): first differences %r"
% (p, len(got), len(vis), diff))
else:
print(" %s: JSON-LD equals the %d visible questions and answers" % (p, len(vis)))
# gate 18: nothing after </html>
for p, s in pages.items():
i = s.rfind("</html>")
if i < 0 or s[i + len("</html>"):].strip():
fail("%s: text after </html> (%r) — it renders on the page" % (p, s[i + 7:i + 87] if i >= 0 else "no </html>"))
if fails:
print("\nSITE GATES FAILED: %d problem(s)" % len(fails))
sys.exit(1)
print("site gates OK — BOM, emoji=0, nav/footer per language, analytics, no CDN, no legacy tokens, no <style>, "
"cache-busted assets, twins + hreflang + switch, lang, same apps, no Hungarian in English, FAQ JSON-LD, clean tail")