Files
felhom.eu/scripts/site_gates.py
T

186 lines
8.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""TASK-D3 site gates — mechanical checks against the static site's known failure mode:
silent per-page drift. Run from the repo root: python scripts/site_gates.py
Gates (all must pass; non-zero exit on any failure):
1. BOM — every website/*.html begins with EF BB BF (byte-checked)
2. emoji — zero emoji/pictographs in website/*.html (codepoint ranges; NEVER grep —
Windows grep false-negatives multibyte emoji, proven in D0)
3. nav — the <nav>…</nav> and <footer>…</footer> blocks of all pages are identical
after stripping the active-link marker
4. analytics — the umami snippet is present on every public page (nonpublic draft exempt)
5. no-CDN — zero fonts.googleapis.com / fonts.gstatic.com references
6. banned — zero legacy hexes / 999px radius / box-shadow in pages + site.css
7. style — zero embedded <style> blocks in pages (everything lives in site.css)
8. cachebust — every site.css / icons.svg reference carries ?v=<N>
"""
import io, os, re, sys
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
W = os.path.join(ROOT, "website")
PAGES = ["index.html", "kapcsolat.html", "alkalmazasok.html", "technologiak.html",
"biztonsagimentes.html", "gyik.html", "szolgaltatasok-nonpublic.html",
"letoltes.html",
# R-559 (2026-09-18): the English download page. The marketing site stays Hungarian by
# operator ruling 1b; this one page is its English twin because a volunteer who reads no
# Hungarian still has to fetch the installer.
os.path.join("en", "download.html")]
# EN_PAGES carry their OWN nav and footer, in English, and must not be compared with the Hungarian
# set. Everything else — BOM, emoji, banned tokens, analytics, no <style>, cache-busted assets —
# applies to them exactly as to any other page. Stated as a SET rather than a filename test so a
# second English page cannot join by accident.
EN_PAGES = {os.path.join("en", "download.html")}
ANALYTICS_EXEMPT = {"szolgaltatasok-nonpublic.html"}
ANALYTICS_MARK = "https://stats.felhom.eu/script.js"
BANNED = ["fonts.googleapis.com", "fonts.gstatic.com",
"#0d1117", "#161b22", "#1c2128", "#30363d", "#238636", "#da3633", "#d29922",
"border-radius: 999px", "box-shadow"]
fails = []
def fail(msg):
fails.append(msg)
print("FAIL:", msg)
def is_emoji(ch):
o = ord(ch)
return any(a <= o <= b for a, b in [
(0x1F000, 0x1FAFF), # pictographs, emoticons, transport, symbols-ext, cards
(0x2600, 0x27BF), # misc symbols + dingbats (incl. ✓ ✗ ★ ⚠ — sprite icons instead)
(0x2300, 0x23FF), # misc technical (⏱ ⏳ ⌛ …)
(0x2B00, 0x2BFF), # ⭐ etc.
(0xFE00, 0xFE0F), # variation selectors
(0x1F1E6, 0x1F1FF), # regional indicators
])
def norm_block(b):
b = b.replace(' class="active"', "")
b = re.sub(r'\s*aria-current="[^"]*"', "", b)
return b
# R-423 (2026-10-05): PAGES is still the list the checks run on, but it may no longer be SHORTER than the site. Every
# *.html under website/ (a walk, any depth) must be in it — a page added without being added here used to go
# unscanned, which is how a gate checks seven files of nine and prints OK.
_on_disk = set()
for _dp, _dns, _fns in os.walk(W):
_dns[:] = [d for d in _dns if not d.startswith(".")]
for _f in _fns:
if _f.endswith(".html"):
_on_disk.add(os.path.relpath(os.path.join(_dp, _f), W))
for _missing in sorted(_on_disk - set(PAGES)):
fail("%s: an HTML page the site gates do not scan — add it to PAGES in scripts/site_gates.py (R-423)" % _missing)
pages = {}
for p in PAGES:
path = os.path.join(W, p)
raw = io.open(path, "rb").read()
# gate 1: BOM
if raw[:3] != b"\xef\xbb\xbf":
fail("%s: missing UTF-8 BOM (first bytes: %s)" % (p, raw[:3].hex()))
pages[p] = raw.decode("utf-8-sig")
# gate 2: emoji (pages + the shared stylesheet)
total_emoji = 0
emoji_targets = dict(pages)
_css = os.path.join(W, "assets", "site.css")
if os.path.exists(_css):
emoji_targets["assets/site.css"] = io.open(_css, encoding="utf-8").read()
for p, s in emoji_targets.items():
hits = [(i, ch) for i, ch in enumerate(s) if is_emoji(ch)]
if hits:
total_emoji += len(hits)
sample = " ".join(ch for _, ch in hits[:10])
fail("%s: %d emoji (%s ...)" % (p, len(hits), sample))
if total_emoji:
print(" emoji total: %d" % total_emoji)
# gate 3: nav + footer consistency (Hungarian set only — see EN_PAGES)
ref_nav = ref_footer = None
for p, s in pages.items():
if p in EN_PAGES:
continue
nm = re.search(r"<nav>.*?</nav>", s, re.DOTALL)
fm = re.search(r"<footer>.*?</footer>", s, re.DOTALL)
if not nm or not fm:
fail("%s: missing <nav> or <footer>" % p)
continue
nav, foot = norm_block(nm.group(0)), norm_block(fm.group(0))
if ref_nav is None:
ref_nav, ref_footer, ref_page = nav, foot, p
else:
if nav != ref_nav:
fail("%s: <nav> differs from %s (after active-marker normalization)" % (p, ref_page))
if foot != ref_footer:
fail("%s: <footer> differs from %s" % (p, ref_page))
# gate 4: analytics presence
for p, s in pages.items():
if p in ANALYTICS_EXEMPT:
continue
if ANALYTICS_MARK not in s:
fail("%s: analytics snippet missing (%s)" % (p, ANALYTICS_MARK))
# gate 5+6: banned strings in pages + site.css
targets = dict(pages)
css_path = os.path.join(W, "assets", "site.css")
if os.path.exists(css_path):
targets["assets/site.css"] = io.open(css_path, encoding="utf-8").read()
for name, s in targets.items():
low = s.lower()
for pat in BANNED:
c = low.count(pat.lower())
if c:
fail("%s: banned %r ×%d" % (name, pat, c))
# gate 7: no embedded style blocks
for p, s in pages.items():
c = len(re.findall(r"<style[\s>]", s))
if c:
fail("%s: %d embedded <style> block(s)" % (p, c))
# gate 8: cache-busting on shared assets
# gate 12 (R-559): the two download pages name the SAME installer file and the SAME checksum.
#
# THE FAILURE THIS EXISTS FOR IS A HALF-DONE RELEASE. Publishing a new ISO means editing a filename,
# a size and a 64-character hash on TWO pages now. Update one and forget the other and an English
# reader downloads yesterday's image, or — worse — checks today's image against yesterday's hash,
# fails the comparison, and is told by the page itself not to use the file.
#
# It compares what the pages SAY, not what is published: a gate cannot reach the bucket, and a
# published-file check belongs to the ISO release gate (G11), which does exactly that.
_HU_DL, _EN_DL = "letoltes.html", os.path.join("en", "download.html")
if _HU_DL in pages and _EN_DL in pages:
_iso_re = re.compile(r"felhom-installer-[0-9.]+-pve[0-9.\-]+\.iso")
_sha_re = re.compile(r"\b[0-9a-f]{64}\b")
hu_iso = sorted(set(_iso_re.findall(pages[_HU_DL])))
en_iso = sorted(set(_iso_re.findall(pages[_EN_DL])))
hu_sha = sorted(set(_sha_re.findall(pages[_HU_DL])))
en_sha = sorted(set(_sha_re.findall(pages[_EN_DL])))
if not hu_iso:
fail("%s: names no installer file at all" % _HU_DL)
if not hu_sha:
fail("%s: names no SHA-256 at all" % _HU_DL)
if hu_iso != en_iso:
fail("the download pages name DIFFERENT installer files: hu=%s en=%s" % (hu_iso, en_iso))
if hu_sha != en_sha:
fail("the download pages name DIFFERENT checksums: hu=%s en=%s" % (hu_sha, en_sha))
else:
print(" download pages agree: %s, sha %s…" % (", ".join(hu_iso), (hu_sha or ["-"])[0][:12]))
for p, s in pages.items():
for m in re.finditer(r"/assets/(site\.css|icons\.svg)([^\"'#\s>]*)", s):
if not m.group(2).startswith("?v="):
fail("%s: %s referenced without ?v= (got %r)" % (p, m.group(1), m.group(0)))
if fails:
print("\nSITE GATES FAILED: %d problem(s)" % len(fails))
sys.exit(1)
print("site gates OK — BOM, emoji=0, nav/footer consistent, analytics present, no CDN, no legacy tokens, no <style>, cache-busted assets")