803af4798d
gates / gates (push) Successful in 3m54s
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_012qRErfCoiTkvDK9N5XHbzb
447 lines
23 KiB
Python
447 lines
23 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""TASK-D3 site gates — mechanical checks against the static site's known failure mode:
|
||
silent per-page drift. Run from the repo root: python scripts/site_gates.py
|
||
|
||
Gates (all must pass; non-zero exit on any failure):
|
||
1. BOM — every website/*.html begins with EF BB BF (byte-checked)
|
||
2. emoji — zero emoji/pictographs in website/*.html (codepoint ranges; NEVER grep —
|
||
Windows grep false-negatives multibyte emoji, proven in D0)
|
||
3. nav — the <nav>…</nav> and <footer>…</footer> blocks are identical PER LANGUAGE SET
|
||
(Hungarian pages against index.html, English pages against en/index.html) after
|
||
stripping the active-link marker and the language globe (gate 13 checks it)
|
||
4. analytics — the umami snippet is present on every public page (nonpublic draft exempt)
|
||
5. no-CDN — zero fonts.googleapis.com / fonts.gstatic.com references
|
||
6. banned — zero legacy hexes / 999px radius / box-shadow in pages + site.css
|
||
7. style — zero embedded <style> blocks in pages (everything lives in site.css)
|
||
8. cachebust — every site.css / icons.svg reference carries ?v=<N>
|
||
12. download — the two download pages name the same installer file and checksum (R-559)
|
||
13. twins — every public Hungarian page has its English twin and back; each pair carries the same
|
||
hu/en/x-default alternates; each page's language GLOBE (one per page, every page) lists LANGS in
|
||
order, marks only the page's own language current, and points the other items at the twins;
|
||
no leftover text switch
|
||
14. lang — <html lang="hu"> on the Hungarian set, lang="en" on the English set
|
||
15. same apps — the app logos on /alkalmazasok and /en/apps are the same set
|
||
16. no HU — no Hungarian in the English pages: accent scan PLUS an ASCII-folded, word-bounded stem
|
||
scan (an accent-only scan passes ASCII-only Hungarian — 10-localisation.md §7), over
|
||
visible text, attributes, JSON-LD and the inline scripts' string literals. Positive and
|
||
negative controls are printed on every run. Also: no "please"/"kindly", and no English
|
||
retrieval promise beyond what the Hungarian twin makes
|
||
17. faq-ld — each FAQ page's FAQPage JSON-LD carries exactly its visible questions and answers
|
||
18. tail — nothing after </html> (a stray file-dump summary sat visible below technologiak.html)
|
||
"""
|
||
import html as _html, io, json, os, re, sys, unicodedata
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
W = os.path.join(ROOT, "website")
|
||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||
from customer_copy_vocab import RETRIEVAL_STEMS as _RET_HU, RETRIEVAL_STEMS_EN as _RET_EN
|
||
|
||
# The public pairs: Hungarian page -> its English twin (2026-10-08, operator ruling 9 in 10-localisation.md §11:
|
||
# the whole public site gets an English twin; the Hungarian stays x-default). A new public page needs its twin
|
||
# in the same commit — gate 13 refuses one without.
|
||
TWINS = {
|
||
"index.html": os.path.join("en", "index.html"),
|
||
"alkalmazasok.html": os.path.join("en", "apps.html"),
|
||
"technologiak.html": os.path.join("en", "technology.html"),
|
||
"biztonsagimentes.html": os.path.join("en", "backups.html"),
|
||
"gyik.html": os.path.join("en", "faq.html"),
|
||
"kapcsolat.html": os.path.join("en", "contact.html"),
|
||
"letoltes.html": os.path.join("en", "download.html"), # R-559 (2026-09-18) — the first twin
|
||
}
|
||
URL = {"index.html": "/", "alkalmazasok.html": "/alkalmazasok", "technologiak.html": "/technologiak",
|
||
"biztonsagimentes.html": "/biztonsagimentes", "gyik.html": "/gyik", "kapcsolat.html": "/kapcsolat",
|
||
"letoltes.html": "/letoltes", os.path.join("en", "index.html"): "/en/", os.path.join("en", "apps.html"): "/en/apps",
|
||
os.path.join("en", "technology.html"): "/en/technology", os.path.join("en", "backups.html"): "/en/backups",
|
||
os.path.join("en", "faq.html"): "/en/faq", os.path.join("en", "contact.html"): "/en/contact",
|
||
os.path.join("en", "download.html"): "/en/download"}
|
||
# Pages with no twin, BY NAME: the hidden prices page stays Hungarian only (brief 2026-10-08), and the 404 page
|
||
# is served at whatever URL was wrong — it carries one English line instead of a twin.
|
||
NO_TWIN = {"szolgaltatasok-nonpublic.html", "404.html"}
|
||
|
||
PAGES = list(TWINS) + list(TWINS.values()) + sorted(NO_TWIN)
|
||
EN_PAGES = set(TWINS.values())
|
||
HU_PAGES = set(PAGES) - EN_PAGES
|
||
ANALYTICS_EXEMPT = {"szolgaltatasok-nonpublic.html"}
|
||
ANALYTICS_MARK = "https://stats.felhom.eu/script.js"
|
||
SITE = "https://felhom.eu"
|
||
|
||
BANNED = ["fonts.googleapis.com", "fonts.gstatic.com",
|
||
"#0d1117", "#161b22", "#1c2128", "#30363d", "#238636", "#da3633", "#d29922",
|
||
"border-radius: 999px", "box-shadow"]
|
||
|
||
fails = []
|
||
|
||
|
||
def fail(msg):
|
||
fails.append(msg)
|
||
print("FAIL:", msg)
|
||
|
||
|
||
def is_emoji(ch):
|
||
o = ord(ch)
|
||
return any(a <= o <= b for a, b in [
|
||
(0x1F000, 0x1FAFF), # pictographs, emoticons, transport, symbols-ext, cards
|
||
(0x2600, 0x27BF), # misc symbols + dingbats (incl. ✓ ✗ ★ ⚠ — sprite icons instead)
|
||
(0x2300, 0x23FF), # misc technical (⏱ ⏳ ⌛ …)
|
||
(0x2B00, 0x2BFF), # ⭐ etc.
|
||
(0xFE00, 0xFE0F), # variation selectors
|
||
(0x1F1E6, 0x1F1FF), # regional indicators
|
||
])
|
||
|
||
|
||
# The language globe (2026-10-08 ruling, 10-localisation.md §10.7): one <details> per page, its list built from THIS
|
||
# list — a third language is one more entry here plus its pages, and nothing else in the switch.
|
||
LANGS = [("hu", "Magyar"), ("en", "English")]
|
||
GLOBE_LABEL = {"hu": "Nyelv", "en": "Language"}
|
||
GLOBE_RE = re.compile(r'<details class="lang-globe">.*?</details>', re.S)
|
||
GLOBE_ITEM_RE = re.compile(r'<a href="([^"]*)" lang="([^"]*)" hreflang="([^"]*)" class="lang-globe-item( current)?"'
|
||
r'( aria-current="true")?>([^<]*)</a>')
|
||
|
||
|
||
def norm_block(b):
|
||
b = b.replace(' class="active"', "")
|
||
b = re.sub(r'\s*aria-current="[^"]*"', "", b)
|
||
# the globe is the one nav item whose content differs per page (its targets, its current mark, its label) —
|
||
# gate 13 checks all of that; here only its presence must match
|
||
b = GLOBE_RE.sub('<details class="lang-globe"/>', b)
|
||
return b
|
||
|
||
|
||
# R-423 (2026-10-05): PAGES is still the list the checks run on, but it may no longer be SHORTER than the site. Every
|
||
# *.html under website/ (a walk, any depth) must be in it — a page added without being added here used to go
|
||
# unscanned, which is how a gate checks seven files of nine and prints OK.
|
||
_on_disk = set()
|
||
for _dp, _dns, _fns in os.walk(W):
|
||
_dns[:] = [d for d in _dns if not d.startswith(".")]
|
||
for _f in _fns:
|
||
if _f.endswith(".html"):
|
||
_on_disk.add(os.path.relpath(os.path.join(_dp, _f), W))
|
||
for _missing in sorted(_on_disk - set(PAGES)):
|
||
fail("%s: an HTML page the site gates do not scan — add it to TWINS (with its twin) or NO_TWIN in "
|
||
"scripts/site_gates.py (R-423)" % _missing)
|
||
|
||
pages = {}
|
||
for p in PAGES:
|
||
path = os.path.join(W, p)
|
||
if not os.path.exists(path):
|
||
fail("%s: listed in the site gates but missing from website/" % p)
|
||
continue
|
||
raw = io.open(path, "rb").read()
|
||
# gate 1: BOM
|
||
if raw[:3] != b"\xef\xbb\xbf":
|
||
fail("%s: missing UTF-8 BOM (first bytes: %s)" % (p, raw[:3].hex()))
|
||
pages[p] = raw.decode("utf-8-sig")
|
||
|
||
# gate 2: emoji (pages + the shared stylesheet)
|
||
total_emoji = 0
|
||
emoji_targets = dict(pages)
|
||
_css = os.path.join(W, "assets", "site.css")
|
||
if os.path.exists(_css):
|
||
emoji_targets["assets/site.css"] = io.open(_css, encoding="utf-8").read()
|
||
for p, s in emoji_targets.items():
|
||
hits = [(i, ch) for i, ch in enumerate(s) if is_emoji(ch)]
|
||
if hits:
|
||
total_emoji += len(hits)
|
||
sample = " ".join(ch for _, ch in hits[:10])
|
||
fail("%s: %d emoji (%s ...)" % (p, len(hits), sample))
|
||
if total_emoji:
|
||
print(" emoji total: %d" % total_emoji)
|
||
|
||
# gate 3: nav + footer consistency, per language set
|
||
for lang, members, ref in (("hu", HU_PAGES, "index.html"), ("en", EN_PAGES, os.path.join("en", "index.html"))):
|
||
if ref not in pages:
|
||
continue
|
||
blocks = {}
|
||
for p in sorted(members, key=lambda x: (x != ref, x)):
|
||
s = pages.get(p)
|
||
if s is None:
|
||
continue
|
||
nm = re.search(r"<nav>.*?</nav>", s, re.DOTALL)
|
||
fm = re.search(r"<footer>.*?</footer>", s, re.DOTALL)
|
||
if not nm or not fm:
|
||
fail("%s: missing <nav> or <footer>" % p)
|
||
continue
|
||
blocks[p] = (norm_block(nm.group(0)), norm_block(fm.group(0)))
|
||
rnav, rfoot = blocks.get(ref, (None, None))
|
||
for p, (nav, foot) in blocks.items():
|
||
if nav != rnav:
|
||
fail("%s: <nav> differs from %s (after active-marker normalization)" % (p, ref))
|
||
if foot != rfoot:
|
||
fail("%s: <footer> differs from %s" % (p, ref))
|
||
if os.path.join("en", "index.html") in pages:
|
||
_f = re.search(r"<footer>.*?</footer>", pages[os.path.join("en", "index.html")], re.DOTALL)
|
||
if not _f or "households in Hungary" not in _f.group(0):
|
||
fail("en/index.html: the English footer no longer says that Felhom serves households in Hungary "
|
||
"(the brief's rule: every English page says it once — the shared footer is where it lives)")
|
||
|
||
# gate 4: analytics presence
|
||
for p, s in pages.items():
|
||
if p in ANALYTICS_EXEMPT:
|
||
continue
|
||
if ANALYTICS_MARK not in s:
|
||
fail("%s: analytics snippet missing (%s)" % (p, ANALYTICS_MARK))
|
||
|
||
# gate 5+6: banned strings in pages + site.css
|
||
targets = dict(pages)
|
||
css_path = os.path.join(W, "assets", "site.css")
|
||
if os.path.exists(css_path):
|
||
targets["assets/site.css"] = io.open(css_path, encoding="utf-8").read()
|
||
for name, s in targets.items():
|
||
low = s.lower()
|
||
for pat in BANNED:
|
||
c = low.count(pat.lower())
|
||
if c:
|
||
fail("%s: banned %r ×%d" % (name, pat, c))
|
||
|
||
# gate 7: no embedded style blocks
|
||
for p, s in pages.items():
|
||
c = len(re.findall(r"<style[\s>]", s))
|
||
if c:
|
||
fail("%s: %d embedded <style> block(s)" % (p, c))
|
||
|
||
# gate 12 (R-559): the two download pages name the SAME installer file and the SAME checksum.
|
||
#
|
||
# THE FAILURE THIS EXISTS FOR IS A HALF-DONE RELEASE. Publishing a new ISO means editing a filename,
|
||
# a size and a 64-character hash on TWO pages now. Update one and forget the other and an English
|
||
# reader downloads yesterday's image, or — worse — checks today's image against yesterday's hash,
|
||
# fails the comparison, and is told by the page itself not to use the file.
|
||
#
|
||
# It compares what the pages SAY, not what is published: a gate cannot reach the bucket, and a
|
||
# published-file check belongs to the ISO release gate (G11), which does exactly that.
|
||
_HU_DL, _EN_DL = "letoltes.html", os.path.join("en", "download.html")
|
||
if _HU_DL in pages and _EN_DL in pages:
|
||
_iso_re = re.compile(r"felhom-installer-[0-9.]+-pve[0-9.\-]+\.iso")
|
||
_sha_re = re.compile(r"\b[0-9a-f]{64}\b")
|
||
hu_iso = sorted(set(_iso_re.findall(pages[_HU_DL])))
|
||
en_iso = sorted(set(_iso_re.findall(pages[_EN_DL])))
|
||
hu_sha = sorted(set(_sha_re.findall(pages[_HU_DL])))
|
||
en_sha = sorted(set(_sha_re.findall(pages[_EN_DL])))
|
||
if not hu_iso:
|
||
fail("%s: names no installer file at all" % _HU_DL)
|
||
if not hu_sha:
|
||
fail("%s: names no SHA-256 at all" % _HU_DL)
|
||
if hu_iso != en_iso:
|
||
fail("the download pages name DIFFERENT installer files: hu=%s en=%s" % (hu_iso, en_iso))
|
||
if hu_sha != en_sha:
|
||
fail("the download pages name DIFFERENT checksums: hu=%s en=%s" % (hu_sha, en_sha))
|
||
else:
|
||
print(" download pages agree: %s, sha %s…" % (", ".join(hu_iso), (hu_sha or ["-"])[0][:12]))
|
||
|
||
# gate 8: cache-busting on shared assets
|
||
for p, s in pages.items():
|
||
for m in re.finditer(r"/assets/(site\.css|icons\.svg)([^\"'#\s>]*)", s):
|
||
if not m.group(2).startswith("?v="):
|
||
fail("%s: %s referenced without ?v= (got %r)" % (p, m.group(1), m.group(0)))
|
||
|
||
# gate 13: twins — both ways, matching alternates, the globe pointing at the twins
|
||
|
||
|
||
def lang_of(p):
|
||
return "en" if p in EN_PAGES else "hu"
|
||
|
||
|
||
def check_globe(p, lang, urls=None):
|
||
"""Exactly one globe; its label in the page's language; one item per LANGS entry, in order, lang/hreflang equal to
|
||
the entry; the page's own language is the only current/aria-current item; with `urls`, each item's target."""
|
||
s = pages[p]
|
||
g = GLOBE_RE.findall(s)
|
||
if len(g) != 1:
|
||
fail("%s: %d language globes in the page — exactly one, in the nav" % (p, len(g)))
|
||
return
|
||
lab = re.search(r'<summary class="lang-globe-btn" aria-label="([^"]*)" title="([^"]*)">', g[0])
|
||
if not lab or lab.group(1) != GLOBE_LABEL[lang] or lab.group(2) != GLOBE_LABEL[lang]:
|
||
fail("%s: the globe's label is %r — on a %s page it is %r" % (p, lab and lab.groups(), lang, GLOBE_LABEL[lang]))
|
||
items = GLOBE_ITEM_RE.findall(g[0])
|
||
if len(re.findall(r"<a ", g[0])) != len(items):
|
||
fail("%s: the globe holds a link the gate cannot read — every item is <a href lang hreflang class=lang-globe-item>" % p)
|
||
got = [(it[1], it[5]) for it in items]
|
||
if got != LANGS:
|
||
fail("%s: the globe lists %r — every page lists exactly %r, in that order" % (p, got, LANGS))
|
||
return
|
||
for href, code, hreflang, cur, aria, name in items:
|
||
if hreflang != code:
|
||
fail("%s: the globe's %s item has hreflang=%r" % (p, code, hreflang))
|
||
is_cur = code == lang
|
||
if bool(cur) != is_cur or bool(aria) != is_cur:
|
||
fail("%s: the globe marks %r as current (class current=%s, aria-current=%s) — only the page's own language %r is"
|
||
% (p, code, bool(cur), bool(aria), lang))
|
||
if urls is not None and href != urls[code]:
|
||
fail("%s: the globe's %s item points at %r — %s" % (p, code, href,
|
||
"the page itself is %r" % urls[code] if is_cur else "its twin is %r" % urls[code]))
|
||
|
||
|
||
ALT_RE = re.compile(r'<link rel="alternate" hreflang="([^"]+)" href="([^"]+)">')
|
||
for hu, en in TWINS.items():
|
||
if hu not in pages or en not in pages:
|
||
fail("twin pair %s <-> %s: a page of the pair is missing" % (hu, en))
|
||
continue
|
||
want = {"hu": SITE + URL[hu], "en": SITE + URL[en], "x-default": SITE + URL[hu]}
|
||
for p, other, lang in ((hu, en, "en"), (en, hu, "hu")):
|
||
got = dict(ALT_RE.findall(pages[p]))
|
||
if got != want:
|
||
fail("%s: hreflang alternates %s — the pair %s <-> %s needs exactly %s" % (p, got, hu, en, want))
|
||
check_globe(p, lang_of(p), {"hu": URL[hu], "en": URL[en]})
|
||
canon = re.search(r'<link rel="canonical" href="([^"]+)">', pages[p])
|
||
if not canon or canon.group(1) != SITE + URL[p]:
|
||
fail("%s: canonical %r — it must point at the page itself (%s)" % (p, canon and canon.group(1), SITE + URL[p]))
|
||
print(" twins: %d pairs checked both ways (hreflang, globe, canonical)" % len(TWINS))
|
||
for p in sorted(NO_TWIN): # no twin: the globe's structure, current mark and label still hold
|
||
if p in pages:
|
||
check_globe(p, lang_of(p))
|
||
for p, s in pages.items():
|
||
if 'class="lang-switch"' in s:
|
||
fail("%s: a leftover text language switch (class=\"lang-switch\") — the globe replaced it" % p)
|
||
|
||
# gate 14: the lang attribute
|
||
for p, s in pages.items():
|
||
m = re.search(r'<html lang="([^"]+)"', s)
|
||
want = "en" if p in EN_PAGES else "hu"
|
||
if not m or m.group(1) != want:
|
||
fail("%s: <html lang=%r> — the %s set needs lang=%r" % (p, m and m.group(1), "English" if want == "en" else "Hungarian", want))
|
||
|
||
|
||
# gate 15: the same apps both ways
|
||
def app_logos(s):
|
||
return set(re.findall(r'<img src="/assets/([^"]+)" alt="[^"]*" class="app-logo"', s))
|
||
|
||
|
||
_A_HU, _A_EN = "alkalmazasok.html", os.path.join("en", "apps.html")
|
||
if _A_HU in pages and _A_EN in pages:
|
||
lh, le = app_logos(pages[_A_HU]), app_logos(pages[_A_EN])
|
||
cards_hu = len(re.findall(r'<div class="app-card">', pages[_A_HU]))
|
||
cards_en = len(re.findall(r'<div class="app-card">', pages[_A_EN]))
|
||
if lh != le or cards_hu != cards_en:
|
||
fail("the apps pages show DIFFERENT apps: only hu %s, only en %s, cards hu=%d en=%d"
|
||
% (sorted(lh - le), sorted(le - lh), cards_hu, cards_en))
|
||
else:
|
||
print(" apps pages agree: %d cards, %d logos" % (cards_hu, len(lh)))
|
||
|
||
|
||
# gate 16: no Hungarian on the English pages
|
||
def fold(s):
|
||
s = unicodedata.normalize("NFKD", s)
|
||
return "".join(c for c in s if not unicodedata.combining(c)).lower()
|
||
|
||
|
||
HU_LETTER = re.compile(u"[áéíóöőúüűÁÉÍÓÖŐÚÜŰ]")
|
||
# ASCII-folded stems. Stems of five letters or more match the START of a word ("jelentkez" convicts
|
||
# "Jelentkezz"); short ones, and the two language names, match WHOLE words only — "angol" must not convict
|
||
# "Angola", nor "az" convict "Amazon". Reused approach: app-catalog-felhom.eu/scripts/check-copy-i18n.py.
|
||
HU_STEMS = ["aldomain", "jelentkez", "oszd meg", "kattints", "valaszd", "magyar", "angol", "felhasznalo", "jelszo",
|
||
"nyelv", "beallitas", "alkalmazas", "szerver", "fajl", "mappa", "megosztas", "mentes", "doboz",
|
||
"vezerlopult", "kapcsolat", "otthon", "szolgaltatas", "biztonsag", "telepit", "haztartas", "kerdes",
|
||
"valasz", "letoltes", "tovabb", "ingyenes", "nyilt", "forraskod", "tarhely", "frissites", "uzemeltet",
|
||
"technologia", "hamarosan", "kerlek", "koszon", "uzenet", "targy",
|
||
"es", "egy", "vagy", "nem", "igen", "az", "hogy", "mint", "csak", "itt", "ahol", "ezt", "ami", "mert",
|
||
"nev", "gyik"]
|
||
_WHOLE = {"angol", "magyar"}
|
||
HU_RX = [(st, re.compile(r"(?<![a-z0-9])%s" % re.escape(st) + (r"(?![a-z0-9])" if len(st) <= 4 or st in _WHOLE else "")))
|
||
for st in HU_STEMS]
|
||
BEG = re.compile(r"\b(please|kindly)\b", re.I)
|
||
|
||
|
||
def visible_strings(s):
|
||
"""Everything a reader can meet: text, attribute values, JSON-LD, and the string literals of inline scripts
|
||
(status messages are shown to the visitor). Not scanned: comments, URLs, and anything marked lang="hu"
|
||
(the switch back to Magyar)."""
|
||
js = " ".join(re.findall(r"<script(?![^>]*ld\+json)[^>]*>(.*?)</script>", s, flags=re.S))
|
||
js = re.sub(r"(?m)^\s*//.*$", " ", js)
|
||
lits = [a or b or c for a, b, c in re.findall(r"'([^'\n]*)'|`([^`]*)`|\"([^\"\n]*)\"", js)]
|
||
s = re.sub(r"<script(?![^>]*ld\+json)[^>]*>.*?</script>", " ", s, flags=re.S)
|
||
s = re.sub(r"<!--.*?-->", " ", s, flags=re.S)
|
||
s = re.sub(r'<(?!html\b)([a-z0-9]+)\b[^>]*\blang="hu"[^>]*>.*?</\1>', " ", s, flags=re.S)
|
||
attrs = re.findall(r'\b(?:alt|title|aria-label|placeholder|content)="([^"]*)"', s)
|
||
s = re.sub(r'(https?://|/)[^\s"<>]*', " ", s)
|
||
text = re.sub(r"<[^>]+>", "\n", s)
|
||
return [_html.unescape(t).strip() for t in text.split("\n") + attrs + lits if t.strip()]
|
||
|
||
|
||
def hungarian_hits(s):
|
||
out = []
|
||
for t in visible_strings(s):
|
||
t2 = re.sub(r"\b[\w.-]+\.hu\b", " ", t) # Hungarian site names (mindmegette.hu …) are names
|
||
if HU_LETTER.search(t2):
|
||
out.append(("an accented Hungarian letter", t))
|
||
continue
|
||
f = fold(t2)
|
||
for st, rx in HU_RX:
|
||
if rx.search(f):
|
||
out.append(("the Hungarian stem %r" % st, t))
|
||
break
|
||
return out
|
||
|
||
|
||
# controls, printed on every run (workspace rule: a zero from a Hungarian search is suspect without them)
|
||
_pos = hungarian_hits(u"<p>Jelentkezz be: admin</p>")
|
||
_pos2 = hungarian_hits(u"<p>Kapcsolat</p>")
|
||
_neg = hungarian_hits(u"<p>The third best option is in Angola. These documents are kept on the server. "
|
||
u"Amazon is not an apps page.</p>")
|
||
if not _pos or not _pos2 or _neg:
|
||
fail("gate 16 INCONCLUSIVE: matcher controls failed (positive %r/%r must convict, negative %r must not)"
|
||
% (_pos, _pos2, _neg))
|
||
else:
|
||
print(" no-HU matcher controls OK (positive 'Jelentkezz be'/'Kapcsolat' convict, negative 'Angola…' does not)")
|
||
|
||
for p in sorted(EN_PAGES):
|
||
s = pages.get(p)
|
||
if s is None:
|
||
continue
|
||
for why, t in hungarian_hits(s)[:5]:
|
||
fail("%s: Hungarian on an English page (%s): %r" % (p, why, t[:120]))
|
||
for t in visible_strings(s):
|
||
if BEG.search(t):
|
||
fail("%s: the site does not beg — no \"please\"/\"kindly\": %r" % (p, t[:120]))
|
||
# the retrieval promise: the English page may make it no more often than its Hungarian twin does
|
||
hu = next((h for h, e in TWINS.items() if e == p), None)
|
||
if hu and hu in pages:
|
||
en_n = sum(len(re.findall(rx, t, re.I)) for t in visible_strings(s) for rx in _RET_EN)
|
||
hu_n = sum(fold(t).count(fold(st)) for t in visible_strings(pages[hu]) for st in _RET_HU)
|
||
if en_n > hu_n:
|
||
fail("%s: %d English retrieval promise(s) ('can be restored', 'recoverable' …) against %d in %s — "
|
||
"the English promises nothing the Hungarian does not" % (p, en_n, hu_n, hu))
|
||
|
||
|
||
# gate 17: FAQ JSON-LD == visible questions and answers
|
||
def faq_pairs(s):
|
||
out = []
|
||
for q, a in re.findall(r'<button class="faq-question">\s*(.*?)\s*<svg class="faq-chevron".*?'
|
||
r'<div class="faq-answer-inner">(.*?)</div></div>', s, re.S):
|
||
parts = re.findall(r"<(?:p|li)[^>]*>(.*?)</(?:p|li)>", a, re.S)
|
||
out.append((_html.unescape(q.strip()),
|
||
" ".join(re.sub(r"\s+", " ", _html.unescape(re.sub(r"<[^>]+>", "", x))).strip() for x in parts)))
|
||
return out
|
||
|
||
|
||
for p in ("gyik.html", os.path.join("en", "faq.html")):
|
||
s = pages.get(p)
|
||
if s is None:
|
||
continue
|
||
m = re.search(r'<script type="application/ld\+json">(.*?)</script>', s, re.S)
|
||
try:
|
||
ld = json.loads(m.group(1)) if m else {}
|
||
except ValueError as e:
|
||
fail("%s: the FAQPage JSON-LD does not parse: %s" % (p, e))
|
||
continue
|
||
got = [(e.get("name"), (e.get("acceptedAnswer") or {}).get("text")) for e in ld.get("mainEntity", [])]
|
||
vis = faq_pairs(s)
|
||
if not vis:
|
||
fail("%s: found no visible FAQ items — the gate's pattern no longer matches the page" % p)
|
||
elif got != vis:
|
||
diff = [(i, a, b) for i, (a, b) in enumerate(zip(got, vis)) if a != b][:2]
|
||
fail("%s: the FAQPage JSON-LD (%d items) differs from the visible questions/answers (%d): first differences %r"
|
||
% (p, len(got), len(vis), diff))
|
||
else:
|
||
print(" %s: JSON-LD equals the %d visible questions and answers" % (p, len(vis)))
|
||
|
||
# gate 18: nothing after </html>
|
||
for p, s in pages.items():
|
||
i = s.rfind("</html>")
|
||
if i < 0 or s[i + len("</html>"):].strip():
|
||
fail("%s: text after </html> (%r) — it renders on the page" % (p, s[i + 7:i + 87] if i >= 0 else "no </html>"))
|
||
|
||
if fails:
|
||
print("\nSITE GATES FAILED: %d problem(s)" % len(fails))
|
||
sys.exit(1)
|
||
print("site gates OK — BOM, emoji=0, nav/footer per language, analytics, no CDN, no legacy tokens, no <style>, "
|
||
"cache-busted assets, twins + hreflang + globe, lang, same apps, no Hungarian in English, FAQ JSON-LD, clean tail")
|