#!/usr/bin/env python3 # -*- coding: utf-8 -*- """fb_probe.py — the Facebook Page access probe (spike 2026-10-08). Stdlib only. Usage: python3 scripts/facebook/fb_probe.py [-v] [--version v26.0] [--evidence DIR] read python3 scripts/facebook/fb_probe.py [-v] [--version v26.0] [--evidence DIR] write-test `read` Scenarios A-C: the system-user token debugs itself, reaches the Felhom.eu Page, derives the Page token (memory only), reads the Page, its feed and a few insights metrics. `write-test` Scenarios D-E: ONE scheduled text post and ONE scheduled photo, a week ahead, each read back (UTF-8 hex compare) and DELETED; removal is proven by a failed GET, never by DELETE's own success. Refuses unless A and B pass and the Page grants CREATE_CONTENT. There is deliberately NO "post for real" sub-command: real posting belongs to the skill written from this spike's findings (audits/SPIKE-facebook-page-api-2026-10-08.md). SECRETS. The token comes ONLY from ~/.config/credentials (key FACEBOOK_API); no flag takes one. It travels as `Authorization: Bearer`, never in a logged URL (the one query-string carrier, debug_token's input_token, is never printed: -v prints the path without its query). Every saved response passes redact(): every `access_token` key is dropped and any string carrying a held token or an `EAA…` token shape is replaced. The Page token never touches disk. Exit codes: 0 ok · 2 key refused · 3 Scenario A failed · 4 Scenario B failed · 5 write gate refused · 6 a write was refused (stop, finding) · 7 a test object could not be proven removed · 8 a test object read back PUBLISHED (deleted at once — read the report first) """ import argparse import json import os import re import sys import time import urllib.error import urllib.parse import urllib.request import uuid HERE = os.path.dirname(os.path.abspath(__file__)) ROOT = os.path.dirname(os.path.dirname(HERE)) sys.path.insert(0, os.path.dirname(HERE)) from read_credential import CredentialError, unwrap # noqa: E402 (R-453: the one quote-stripper) GRAPH = "https://graph.facebook.com" APP_ID = "2273465403490709" PAGE_NAME = "Felhom.eu" KEY = "FACEBOOK_API" DEFAULT_EVIDENCE = os.path.join(ROOT, "documentation", "audits", "facebook-page-api-2026-10-08") LOGO = os.path.join(ROOT, "website", "assets", "logo.png") WEEK_S = 7 * 24 * 3600 # "Felhom teszt – árvíztűrő tükörfúrógép. Ez a bejegyzés törlődik." — built from escapes, never typed # through a shell (brief §9.8). TEST_TEXT = ("Felhom teszt – árvíztűrő tükörfúrógép. " "Ez a bejegyzés törlődik.") HEADERS_KEPT = ("facebook-api-version", "x-business-use-case-usage", "x-app-usage", "x-page-usage", "x-fb-trace-id", "x-fb-rev") TOKEN_SHAPE = re.compile(r"EAA[A-Za-z0-9]{10,}") VERBOSE = False SECRETS = [] # every token value held this run; redact() scrubs each def log(msg): print(msg, flush=True) def vlog(msg): if VERBOSE: print(" " + msg, flush=True) # ---------------------------------------------------------------- the key (R-453) def load_key(path, key=KEY): """Return the value of `key`. Accepts an optional `export ` prefix; ONE matching quote pair is stripped by read_credential.unwrap. Then asserts: no quote, no whitespace, starts with EAA.""" with open(path, encoding="utf-8") as fh: for line in fh: s = line.strip() if s.startswith("export "): s = s[len("export "):].lstrip() if s.startswith(key + "="): value = unwrap(s[len(key) + 1:].strip()) if any(q in value for q in ("'", '"')): raise CredentialError("value still contains a quote character") if any(c.isspace() for c in value): raise CredentialError("value contains whitespace") if not value.startswith("EAA"): raise CredentialError("value does not start with EAA (starts %r)" % value[:3]) return value raise CredentialError("key %r not present in %s" % (key, path)) # ---------------------------------------------------------------- redaction def redact(obj, secrets=None): """Deep copy of `obj` with every access_token key removed and every token-bearing string replaced.""" secrets = SECRETS if secrets is None else secrets if isinstance(obj, dict): return {k: redact(v, secrets) for k, v in obj.items() if k != "access_token"} if isinstance(obj, list): return [redact(v, secrets) for v in obj] if isinstance(obj, str): if any(s and s in obj for s in secrets) or TOKEN_SHAPE.search(obj): return "[REDACTED]" return obj # ---------------------------------------------------------------- HTTP class Call: def __init__(self, step, method, path, status, body, headers, err): self.step, self.method, self.path = step, method, path self.status, self.body, self.headers, self.err = status, body, headers, err @property def ok(self): return self.err is None class Graph: def __init__(self, version, evidence): self.version, self.evidence = version, evidence self.headers_seen = {} os.makedirs(evidence, exist_ok=True) def call(self, step, method, path, token, query=None, form=None, files=None): """One Graph call. `path` is printed and saved; `query` is sent but NEVER printed or saved when it carries a secret (it is saved only as the list of its keys).""" url = "%s/%s/%s" % (GRAPH, self.version, path.lstrip("/")) if query: url += "?" + urllib.parse.urlencode(query) data, ctype = None, None if files: data, ctype = multipart(form or {}, files) elif form is not None: data, ctype = urllib.parse.urlencode(form).encode("utf-8"), "application/x-www-form-urlencoded" req = urllib.request.Request(url, data=data, method=method) req.add_header("Authorization", "Bearer " + token) if ctype: req.add_header("Content-Type", ctype) status, raw, hdrs = None, b"", {} try: with urllib.request.urlopen(req, timeout=60) as r: status, raw, hdrs = r.status, r.read(), r.headers except urllib.error.HTTPError as e: status, raw, hdrs = e.code, e.read(), e.headers except urllib.error.URLError as e: status, raw = None, json.dumps({"transport_error": str(e.reason)}).encode() try: body = json.loads(raw.decode("utf-8")) if raw else {} except ValueError: body = {"non_json_body": raw[:500].decode("utf-8", "replace")} kept = {h: hdrs.get(h) for h in HEADERS_KEPT if hdrs and hdrs.get(h) is not None} for h, v in kept.items(): self.headers_seen.setdefault(h, v) err = None if status is None or not (200 <= status < 300): err = body.get("error", body) if isinstance(body, dict) else body elif isinstance(body, dict) and "error" in body: # Meta can say 200 and mean no (§9.3) err = body["error"] c = Call(step, method, path, status, body, kept, err) vlog("%s %s -> HTTP %s%s" % (method, path, status, "" if err is None else " ERROR " + json.dumps(redact(err), ensure_ascii=False))) self.save(c, query, form, files) return c def save(self, c, query, form, files): rec = { "step": c.step, "method": c.method, "path": c.path, "api_version": self.version, "query_keys": sorted(query) if query else [], "form": redact({k: v for k, v in (form or {}).items()}), "files": {k: os.path.relpath(v, ROOT) for k, v in (files or {}).items()}, "http_status": c.status, "headers": c.headers, "ok": c.ok, "error": c.err, "response": c.body, "at_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), } path = os.path.join(self.evidence, c.step + ".json") with open(path, "w", encoding="utf-8") as fh: json.dump(redact(rec), fh, ensure_ascii=False, indent=2, sort_keys=True) fh.write("\n") def multipart(fields, files): boundary = "----felhomprobe" + uuid.uuid4().hex out = [] for k, v in fields.items(): out += [b"--" + boundary.encode(), ('Content-Disposition: form-data; name="%s"' % k).encode(), b"", str(v).encode("utf-8")] for k, p in files.items(): with open(p, "rb") as fh: blob = fh.read() out += [b"--" + boundary.encode(), ('Content-Disposition: form-data; name="%s"; filename="%s"' % (k, os.path.basename(p))).encode(), b"Content-Type: image/png", b"", blob] out += [b"--" + boundary.encode() + b"--", b""] return b"\r\n".join(out), "multipart/form-data; boundary=" + boundary def note(g, step, data): """A non-call record (a verdict, a comparison) saved beside the calls.""" with open(os.path.join(g.evidence, step + ".json"), "w", encoding="utf-8") as fh: json.dump(redact(data), fh, ensure_ascii=False, indent=2, sort_keys=True) fh.write("\n") # ---------------------------------------------------------------- scenarios def scenario_a(g, token): c = g.call("A1-debug-token", "GET", "debug_token", token, query={"input_token": token}) d = (c.body or {}).get("data", {}) if c.ok else {} checks = {"call_ok": c.ok, "is_valid": d.get("is_valid") is True, "app_id": d.get("app_id") == APP_ID} log("A debug_token: ok=%s is_valid=%s type=%s app_id=%s expires_at=%s data_access_expires_at=%s" % (c.ok, d.get("is_valid"), d.get("type"), d.get("app_id"), d.get("expires_at"), d.get("data_access_expires_at"))) log(" scopes=%s" % d.get("scopes")) return all(checks.values()), d def scenario_b(g, token): me = g.call("B1-me", "GET", "me", token, query={"fields": "id,name"}) log("B /me: ok=%s id=%s name=%s" % (me.ok, me.body.get("id"), me.body.get("name"))) acc = g.call("B2-me-accounts", "GET", "me/accounts", token, query={"fields": "id,name,tasks,access_token"}) if not (me.ok and acc.ok): return None pages = acc.body.get("data", []) log(" /me/accounts: %d page(s): %s" % (len(pages), [p.get("name") for p in pages])) mine = [p for p in pages if p.get("name") == PAGE_NAME] if len(mine) != 1: log(" STOP: expected exactly one page named %s, found %d" % (PAGE_NAME, len(mine))) return None p = mine[0] ptok = p.get("access_token") if not ptok: log(" STOP: the page entry carries no access_token") return None SECRETS.append(ptok) log(" page id=%s tasks=%s (page token held in memory, %d chars)" % (p["id"], p.get("tasks"), len(ptok))) dbg = g.call("B3-debug-page-token", "GET", "debug_token", token, query={"input_token": ptok}) dd = dbg.body.get("data", {}) if dbg.ok else {} log(" page token: ok=%s type=%s is_valid=%s expires_at=%s data_access_expires_at=%s scopes=%s" % (dbg.ok, dd.get("type"), dd.get("is_valid"), dd.get("expires_at"), dd.get("data_access_expires_at"), dd.get("scopes"))) return {"id": p["id"], "tasks": p.get("tasks") or [], "token": ptok} INSIGHT_METRICS = ("page_post_engagements", "page_follows", "page_media_view") def scenario_c(g, page): pid, ptok = page["id"], page["token"] c1 = g.call("C1-page", "GET", pid, ptok, query={"fields": "id,name,link,category,about,website,followers_count,fan_count"}) log("C page fields: ok=%s %s" % (c1.ok, {k: v for k, v in c1.body.items() if k != "error"} if c1.ok else c1.err)) c2 = g.call("C2-feed", "GET", pid + "/feed", ptok, query={"limit": "5"}) log(" feed: ok=%s posts=%s" % (c2.ok, len(c2.body.get("data", [])) if c2.ok else c2.err)) for m in INSIGHT_METRICS: # one call per metric: one renamed metric must not hide the others c = g.call("C3-insights-" + m, "GET", pid + "/insights", ptok, query={"metric": m, "period": "day"}) log(" insights %s: ok=%s %s" % (m, c.ok, "values=%d" % len(c.body.get("data", [])) if c.ok else json.dumps(c.err, ensure_ascii=False))) def hexs(s): return (s or "").encode("utf-8").hex() def gone_error(err): """True when a GET's error says the object does not exist. MEASURED 2026-10-08 (v26.0, Page token): a deleted scheduled post answers code 10 „(#10) Object does not exist, cannot be loaded due to missing permission…", not code 100. Code 10 also means a missing permission, so callers count it only after the SAME token read the object successfully before the DELETE (the readback is the control).""" if not isinstance(err, dict): return False msg = err.get("message") or "" return err.get("code") == 100 or (err.get("code") == 10 and "Object does not exist" in msg) def prove_removed(g, step, obj_id, ptok): c = g.call(step, "GET", obj_id, ptok, query={"fields": "id"}) gone = (not c.ok) and gone_error(c.err) log(" GET after DELETE %s: HTTP %s, removed=%s, error=%s" % (obj_id, c.status, gone, json.dumps(c.err, ensure_ascii=False))) return gone def delete(g, step, obj_id, ptok): c = g.call(step, "DELETE", obj_id, ptok) if not (c.ok and c.body.get("success") is True): log(" DELETE %s failed (%s) — once more after 30 s" % (obj_id, c.err)) time.sleep(30) c = g.call(step + "-retry", "DELETE", obj_id, ptok) log(" DELETE %s: ok=%s body=%s" % (obj_id, c.ok, c.body)) return c.ok and c.body.get("success") is True def check_readback(label, rb, sent_hex, field, when): got = rb.body.get(field) if rb.ok else None res = { "readback_ok": rb.ok, "is_published": rb.body.get("is_published") if rb.ok else None, "scheduled_publish_time_sent": when, "scheduled_publish_time_read": rb.body.get("scheduled_publish_time") if rb.ok else None, "text_field": field, "sent_hex": sent_hex, "read_hex": hexs(got) if got is not None else None, } res["hex_equal"] = res["read_hex"] == sent_hex log(" %s read back: is_published=%s scheduled=%s (sent %s) hex_equal=%s" % (label, res["is_published"], res["scheduled_publish_time_read"], when, res["hex_equal"])) return res def scenario_d(g, page, rc): pid, ptok = page["id"], page["token"] when = int(time.time()) + WEEK_S c = g.call("D1-create-feed", "POST", pid + "/feed", ptok, form={"message": TEST_TEXT, "published": "false", "scheduled_publish_time": str(when)}) if not c.ok: log("D REFUSED: %s — stopping all writes" % json.dumps(c.err, ensure_ascii=False)) return 6, None post_id = c.body.get("id") log("D created %s" % post_id) rb = g.call("D2-readback", "GET", post_id, ptok, query={"fields": "message,is_published,scheduled_publish_time,created_time"}) res = check_readback("D", rb, hexs(TEST_TEXT), "message", when) res["post_id"] = post_id published = res["is_published"] is True deleted = delete(g, "D3-delete", post_id, ptok) res["delete_ok"] = deleted res["removed"] = prove_removed(g, "D4-get-after-delete", post_id, ptok) note(g, "D9-verdict", res) if published: log("D !!! the post read back PUBLISHED — deleted; removed=%s" % res["removed"]) return 8, res if not res["removed"]: log("D !!! REMOVAL NOT PROVEN for %s — remove it by hand in Meta Business Suite > Planner" % post_id) return 7, res return rc, res def scenario_e(g, page, rc): pid, ptok = page["id"], page["token"] when = int(time.time()) + WEEK_S c = g.call("E1-create-photo", "POST", pid + "/photos", ptok, form={"caption": TEST_TEXT, "published": "false", "scheduled_publish_time": str(when)}, files={"source": LOGO}) if not c.ok: log("E REFUSED: %s — stopping all writes" % json.dumps(c.err, ensure_ascii=False)) return 6, None photo_id, post_id = c.body.get("id"), c.body.get("post_id") log("E created photo id=%s post_id=%s (keys returned: %s)" % (photo_id, post_id, sorted(c.body))) res = {"photo_id": photo_id, "post_id": post_id, "create_keys": sorted(c.body), "logo": os.path.relpath(LOGO, ROOT)} if post_id: rb = g.call("E2-readback-post", "GET", post_id, ptok, query={"fields": "message,is_published,scheduled_publish_time,created_time"}) res["post"] = check_readback("E post", rb, hexs(TEST_TEXT), "message", when) if photo_id: rp = g.call("E3-readback-photo", "GET", photo_id, ptok, query={"fields": "id,name,created_time,link"}) got = rp.body.get("name") if rp.ok else None res["photo"] = {"readback_ok": rp.ok, "name_hex": hexs(got) if got is not None else None} res["photo"]["hex_equal"] = res["photo"]["name_hex"] == hexs(TEST_TEXT) log(" E photo read back: ok=%s caption hex_equal=%s" % (rp.ok, res["photo"]["hex_equal"])) published = bool(res.get("post", {}).get("is_published")) # Delete the post first, then look at both ids: which object DELETE needs is a finding (§7 E). first = post_id or photo_id res["delete_first_target"] = "post_id" if post_id else "photo_id" res["delete_first_ok"] = delete(g, "E4-delete-" + res["delete_first_target"], first, ptok) res["post_removed"] = prove_removed(g, "E5-get-post-after-delete", post_id, ptok) if post_id else None res["photo_removed"] = prove_removed(g, "E6-get-photo-after-delete", photo_id, ptok) if photo_id else None if photo_id and post_id and not res["photo_removed"]: log(" the photo outlived its post's DELETE — deleting the photo id too") res["delete_photo_ok"] = delete(g, "E7-delete-photo_id", photo_id, ptok) res["photo_removed"] = prove_removed(g, "E8-get-photo-after-delete", photo_id, ptok) note(g, "E9-verdict", res) removed = res["photo_removed"] is not False and res["post_removed"] is not False if published: log("E !!! the photo post read back PUBLISHED — deleted; removed=%s" % removed) return 8, res if not removed: log("E !!! REMOVAL NOT PROVEN (photo %s, post %s) — remove by hand in Meta Business Suite > Planner" % (photo_id, post_id)) return 7, res return rc, res # ---------------------------------------------------------------- main def main(argv=None): global VERBOSE ap = argparse.ArgumentParser(description=__doc__.splitlines()[0]) ap.add_argument("-v", action="store_true", help="print each call: method, path (no query), status, error") ap.add_argument("--version", default="v26.0", help="Graph API version (default v26.0)") ap.add_argument("--evidence", default=DEFAULT_EVIDENCE) ap.add_argument("--credentials", default=os.path.expanduser("~/.config/credentials")) ap.add_argument("cmd", choices=("read", "write-test")) a = ap.parse_args(argv) VERBOSE = a.v try: token = load_key(a.credentials) except (CredentialError, OSError) as e: log("KEY REFUSED [%s]: %s" % (KEY, e)) return 2 SECRETS.append(token) log("key %s: %d chars, starts %s" % (KEY, len(token), token[:3])) # write-test re-derives A and B into its own sub-directory, so the read run's files stay as they were g = Graph(a.version, a.evidence if a.cmd == "read" else os.path.join(a.evidence, "write-test")) try: return run(a.cmd, g, token) finally: note(g, "F1-headers", g.headers_seen) # §7 F: recorded once each, on every exit path log("F headers: %s" % json.dumps(g.headers_seen)) def run(cmd, g, token): ok_a, _ = scenario_a(g, token) if not ok_a: log("A FAILED") return 3 page = scenario_b(g, token) if page is None: log("B FAILED") return 4 if cmd == "read": scenario_c(g, page) log("read: done") return 0 if "CREATE_CONTENT" not in page["tasks"]: log("WRITE GATE: the page's tasks lack CREATE_CONTENT (%s) — no write" % page["tasks"]) return 5 rc, _ = scenario_d(g, page, 0) if rc != 0: return rc rc, _ = scenario_e(g, page, 0) log("write-test: done rc=%d" % rc) return rc if __name__ == "__main__": sys.exit(main())