From 6b5804c63cbb3484d44cb5f58ff10e512650c32b Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Tue, 6 Oct 2026 11:18:14 +0200 Subject: [PATCH] R-99 runbook (pbs-phantom-cleanup) + read-only ep0 listing: no phantom today, nothing deleted; R-444 trim measured on demo-hp; R-618 closed by ruling (150 -> 149) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- .../ten-answers-2026-10-06/r444-measure.txt | 16 +++++ .../r99-ep0-listing.txt | 16 +++++ documentation/backlog/CLOSED-ITEMS.md | 8 +++ documentation/backlog/OPEN-ITEMS.md | 3 +- documentation/runbooks/pbs-phantom-cleanup.md | 62 +++++++++++++++++++ documentation/runbooks/pbs-phantom-list.py | 23 +++++++ 6 files changed, 126 insertions(+), 2 deletions(-) create mode 100644 documentation/audits/ten-answers-2026-10-06/r444-measure.txt create mode 100644 documentation/audits/ten-answers-2026-10-06/r99-ep0-listing.txt create mode 100644 documentation/runbooks/pbs-phantom-cleanup.md create mode 100644 documentation/runbooks/pbs-phantom-list.py diff --git a/documentation/audits/ten-answers-2026-10-06/r444-measure.txt b/documentation/audits/ten-answers-2026-10-06/r444-measure.txt new file mode 100644 index 00000000..a5a7802a --- /dev/null +++ b/documentation/audits/ten-answers-2026-10-06/r444-measure.txt @@ -0,0 +1,16 @@ +== R-444 trim measurement, demo-hp, 2026-10-06T09:13:53Z +-- before + data 53.87g 65.53 + vm-9201-disk-0 32.00g 11.45 + vm-9201-disk-1 70.00g 45.20 +/dev/mapper/pve-vm--9201--disk--0 32716560 1161396 29861060 4% / +/dev/mapper/pve-vm--9201--disk--1 72064432 15456272 52921772 23% /var/lib/felhom +-- pct fstrim 9201 (start 09:13:59) +/var/lib/lxc/9201/rootfs/: 30.1 GiB (32277680128 bytes) trimmed +/var/lib/lxc/9201/rootfs/var/lib/felhom: 53.9 GiB (57865633792 bytes) trimmed +rc=0 duration_s=24.367926462 +-- after + data 53.87g 33.40 + vm-9201-disk-0 32.00g 6.01 + vm-9201-disk-1 70.00g 22.96 +-- probe (UTC time, http code, seconds): 18 samples; non-200: 0; max latency: 09:14:01 200 1.115079192 diff --git a/documentation/audits/ten-answers-2026-10-06/r99-ep0-listing.txt b/documentation/audits/ten-answers-2026-10-06/r99-ep0-listing.txt new file mode 100644 index 00000000..05c4647b --- /dev/null +++ b/documentation/audits/ten-answers-2026-10-06/r99-ep0-listing.txt @@ -0,0 +1,16 @@ +== R-99 ep0 listing, read only, 2026-10-06 ~11:25 CEST: datastore felhom-offsite (/mnt/pbs-datastore), every snapshot directory +ns | group | snapshot | manifest (index.json.blob) | directory bytes +Tester-2 | ct/9201 | 2026-10-04T16:31:13Z | True | 35841 +demo-felhom | ct/9201 | 2026-09-22T04:12:20Z | True | 86404 +demo-felhom | ct/9201 | 2026-09-29T04:16:43Z | True | 101070 +demo-felhom | ct/9201 | 2026-10-06T04:21:09Z | True | 43485 +demo-hp | ct/9201 | 2026-09-24T20:06:25Z | True | 279204 +demo-hp | ct/9201 | 2026-10-01T20:15:29Z | True | 187064 +operator | host/dooplex-hub | 2026-10-05T14:23:25Z | True | 10371 +operator | host/dooplex-hub | 2026-10-06T00:32:21Z | True | 10372 +tester-1 | ct/9201 | 2026-10-04T19:57:18Z | True | 72686 +leftover candidates (no manifest, or the PBS-reported size < 1 MiB): 0 +== the backup server's own view (proxmox-backup-debug api get …/snapshots, every namespace): 9 snapshots, sizes +2096481498, 6956268441, 5840921439, 2749038679, 23236593219, 15223817295, 372921211, 369808250, 4913374016 bytes; every +verification state "ok". Smallest 369808250 B, 352x above the 1 MiB phantom line. Real snapshots per namespace: +Tester-2 1, demo-felhom 3, demo-hp 2, operator 2, tester-1 1. Nothing deleted (nothing to delete). diff --git a/documentation/backlog/CLOSED-ITEMS.md b/documentation/backlog/CLOSED-ITEMS.md index 2289a4fe..58363b96 100644 --- a/documentation/backlog/CLOSED-ITEMS.md +++ b/documentation/backlog/CLOSED-ITEMS.md @@ -26,6 +26,14 @@ --- +## 2026-10-06 (midday) — the ten answers + +The full text of every row below: `git show 8c65ff0c:documentation/backlog/OPEN-ITEMS.md`. + +| Row | What | Closed | Evidence | +|---|---|---|---| +| **R-618** | **[P1-HIGH] THREE apps are presented to the household as UNHEALTHY while they are working perfectly — and because the guarded Update waits on that same probe, a SUCCESSFUL update ends by STOPPING the working app and sending the household to a restore they do not need.** (P4) | CLOSED 2026-10-06 — RULED (operator, `09` §3 decision 141): app updates keep our own health check only | The three wrong probes and their gate shipped 2026-09-22 (the row). The open idea — let `verifying` accept Docker's own `healthy` — is REJECTED by the ruling: Docker's healthy does not end a wait early (R-635: a probe can be green on a broken app). Nothing to build. | + ## 2026-10-06 (morning) — R-889 delivered, the held catalog fixes proven and pushed The full text of every row below: `git show e6cfe6d6:documentation/backlog/OPEN-ITEMS.md`. diff --git a/documentation/backlog/OPEN-ITEMS.md b/documentation/backlog/OPEN-ITEMS.md index 4bb0db97..9fc5f07b 100644 --- a/documentation/backlog/OPEN-ITEMS.md +++ b/documentation/backlog/OPEN-ITEMS.md @@ -149,7 +149,6 @@ stopping line that lies. | **R-469** | App updates | P3 | **[P3-LOW] REMOVE THE ENGINE-MAJOR RULE when Slice 4 (R-448) ships — a tracked act, not a lapse.** Since 2026-09-13 `app-catalog-felhom.eu` `CLAUDE.md` rules that *until the Update button takes a verified backup as its precondition, no template may move a database-engine image across a major version* (four MariaDB, eleven PostgreSQL services), and `scripts/check-engine-major.py` (fourth row of `catalog_gates.py`, run by `.githooks/pre-push` with the push range) refuses one, naming the rule and this expiry. **Why the rule:** every `mariadb:` sidecar now carries `MARIADB_AUTO_UPGRADE=1` (R-459), so a MariaDB major move CONVERTS the customer's datadir on the next Update; PostgreSQL converts nothing and refuses to start (R-463). Either way a customer-data event with no backup in front of it. **Honest limit, not re-filed:** the gate needs a parent commit and CI fetches at `--depth 1` — the R-452 gap — so on a shallow clone the runner skips it out loud and only the hook bites. **When R-448 ships:** delete the CLAUDE.md rule, the gate's row and the gate, in one commit that cites this row; then close this. **2026-09-13 — UNBLOCKED, NOT LIFTED.** R-448 shipped in controller v0.237.0/v0.238.0 (slice 4): an update now refuses without a restorable, proven Tier-2 copy, backs up first when it is stale, takes a safety dump, and holds an app that does not come up — the precondition this rule was waiting for. **The rule stays in force until someone deliberately removes it**, which is a separate act (and is worth weighing against R-475: an app with no Tier-2 copy cannot be updated at all, so the guard does not yet cover every app a major engine move would touch). **HALF-LIFTED 2026-09-21, catalog `5ff36d098cbc`.** Slice 4 shipped 2026-09-13, so the rule's own expiry condition is met — **for MariaDB**: the four `mariadb:` sidecars have both halves they need, a verified backup in front of the Update (any tier since v0.239.0) and `MARIADB_AUTO_UPGRADE=1` whose conversion the harness WATCHED run on E3/E3b with the seeded data read back after. **PostgreSQL and MySQL stay refused** — postgres performs no `pg_upgrade` and REFUSES to start on an older major's datadir across eleven templates (R-463); a backup is a route BACK, not a conversion. The refusal text now cites R-463 instead of the shipped R-448. **R-450's second half is enforced in its place:** a MariaDB major must be the ONLY image move in its template in that commit (the bookstack `0b73e5e` shape — two migrations behind one edge). The gate now PRINTS what it allowed, by name — a lifted rule that goes quiet is a lifted rule nobody can audit. Two new decoy cases; two red-proofs, each seen to fail; 40 cases green. **What remains of this row:** the PostgreSQL half, which is R-463's to clear — see `09` §3b **Q5**. **-- NARROWED 2026-09-25 (evening), catalog `6a4a5f0`, `09` §3 decision 35:** the PostgreSQL half now passes ONE app at a time — only a template whose ladder entry for the step is proven on BOTH venues and carries `engine_conversion` (the box converts it, controller v0.273.0), as the only image move in its commit. Every other PostgreSQL app stays refused; the postgis family is judged now (it was not). CLAUDE.md rule text updated the same commit. Decoys + red-proof: `audits/night-2026-09-26/C/`. | **NARROWED** — **PARTLY CLOSED 2026-09-21 — MariaDB lifted; the PostgreSQL half stands until R-463** | — | — | CC | | **R-683** | App updates | P3 | **[P3-LOW] Watch: after a power cut during an update's health check, the hold named an HOUR-OLD second-drive copy, not the one the update's own backup should have just made.** 2026-09-24 chaos round 3 (nextcloud, `backup_max_age: 1m`): no `backing-up` phase was seen and the hold named Tier 2 at 13:04 for an update pressed at 14:04; the pre-cut controller log was lost with the container (the runner now saves it at arm time — R-320). Round 11, the same action without a power cut, named a fresh 14:34 copy and logged the Tier-2 copy. The sentence was TRUE (it named the copy it offered); the question is why the update did not back up first. Not reproduced; watch the next power-cut drill. `audits/night-2026-09-24/E/round-03*.json`, `E/round-11-controller-pre.log` | **OPEN — P3; owner: CC (watch)** | — | — | CC | | **R-785** | App updates | P3 | **[P3-LOW] SparkyFitness is pinned 11 releases and a major behind upstream (v0.17.3; upstream v1.7.3, v1.6.0 dated 2026-07-24).** READ 2026-10-01 (`audits/visitors-2026-10-01/C/bench/C1-previous-tag.txt`). **Needs:** an update walk 0.17 → 1.x through the ladder (bench + box), after R-784 is decided. | **OPEN — rank P3-LOW; owner: CC (after R-784)** | — | — | CC | -| **R-618** | App updates | P4 | **[P1-HIGH] THREE apps are presented to the household as UNHEALTHY while they are working perfectly — and because the guarded Update waits on that same probe, a SUCCESSFUL update ends by STOPPING the working app and sending the household to a restore they do not need.** **RANK RAISED FROM P2 TO P1 BY A LIVE MEASUREMENT taken the same night, and the escalation is the whole point:** tandoor's Update 2.6.13 → 2.6.15 was pressed at 21:16:47 and entered `verifying` at 21:17:46. At **21:18:28** the NEW version was `Up 25 seconds` and answering **HTTP 200** on `/accounts/login/` through the household's own front door — while the controller, probing port 8080 where nothing listens, could not see it. `verifying` therefore cannot pass, the full `update.health_timeout` is spent, `Manager.failAndHold` runs `compose down`, and the app is STOPPED. **Nothing is lost** — the data is in the volumes and the restore works — **but one wrong port number in a template converts every successful update of that app into an outage plus an unnecessary restore, for every household running it.** Evidence: `audits/update-night-2026-09-21/14-tandoor-serving-while-verifying.txt`. MEASURED 2026-09-21 on guest 9202 (controller v0.261.0, catalog `f5f6a152b513`). Two shapes, one class: **(a) `tandoor` — the WRONG PORT.** `.felhom.yml` probes `port: 8080`; the container listens on **80 and nothing else** (`ss -ltn` inside it), the compose's own traefik label routes to 80, its own docker healthcheck reads `healthy`, and `/accounts/login/` answers **200** through the household's real front door. `GET /api/stacks/tandoor` nevertheless reads `state: "unhealthy"`. **(b) `zipline` — the WRONG PATH.** `.felhom.yml` probes `/api/health`, which zipline 4.6.1 answers **404 `Route GET:/api/health not found`**; **the compose healthcheck in the very same file uses `/api/healthcheck` and is correct and green.** `/dashboard` answers 200. The controller reads `unhealthy`. **This is the MIRROR of R-613** — that is a probe that passes on a broken app (a false GREEN, which no alarm catches); this is a probe that fails on a working app (a false RED). **IT DOES NOT ALARM, AND THAT SETS THE RANK:** `08-alarm-ladder.md` §4 puts `unhealthy` deliberately in the NOT-down set, so no dead-app event and no customer mail follows — the damage is what the household READS, plus anything that gates on `state`. **IT ALREADY COST A MEASUREMENT TONIGHT:** this drill's harness waited for `state == "running"` and hung for its full budget on tandoor, an app that was up the whole time. An instrument waiting for a wrong answer looks exactly like a slow app. **THE GATE THIS WANTS IS CHEAP AND STATIC, AND THAT IS THE FINDING'S REAL VALUE.** Both halves of the answer live in the same template: compare the `.felhom.yml` probe's port and path against the compose's **own** `healthcheck: test:` URL. A sweep of all 53 templates on that rule was run tonight and returns **five** disagreements: `tandoor` (PORT — **CONFIRMED live**), `zipline` (PATH — **CONFIRMED live**), `wger` (PORT, probe 80 vs compose 8000 — **CONFIRMED live the same night**), `home-assistant` (PATH, `/api/` vs `/manifest.json` — **NOT MEASURED**), and `adventurelog` (a FALSE POSITIVE of the sweep's own regex — it reads `running` live). **So the rule finds both real defects, with two candidates and one false positive out of 53** — a good enough signal for a fast gate, provided it reports candidates rather than convictions and a person or a runtime check resolves them. The earlier, cruder rule (probe port vs the *traefik* port) is strictly worse: it clears zipline and convicts adventurelog. **AND A SECOND FIX SHAPE, ON THE CONTROLLER SIDE, WORTH CONSIDERING BESIDE THE CATALOG ONE:** in both confirmed cases the container's OWN docker healthcheck was **green** the whole time. A `verifying` phase that is about to stop a working app could ask that too — if the compose declares a healthcheck and docker reports `healthy`, the app is alive whatever our probe thinks. That does not excuse a wrong probe, but it turns this failure direction from an outage into a wrong label. It is a design question, not a defect, and is raised here rather than decided. **Needs:** fix tandoor's port (80), zipline's path (`/api/healthcheck`) and wger's port (8000) — **all three are now CONFIRMED live, none is a guess**; add the static gate with a decoy each way (R-421) — a template whose probe agrees must not read as a disagreement, and vice versa. **THE GATE'S RULE WAS THEN SHARPENED BY READING `healthprobe.go` RATHER THAN ASSUMING IT, and the sharpening REMOVED a false conviction.** `type: http` treats **any** response as healthy (`healthprobe.go:258-261`), and `type: api` with **no** `expect` block does the same (`:265-268`); only `type: api` WITH `expect.status` cares about the path or the code. So a PATH difference is a candidate only for the third shape, while a PORT difference is a candidate for all of them. Under that rule the 53-template sweep returns **four** candidates — `tandoor`, `zipline` and **`wger` (all three CONFIRMED live — probe `type: http, port: 80`; inside the container port 80 is `refused` and port 8000 `ANSWERED`; docker's own healthcheck green; front door 302; the box reads `unhealthy`)**, and `adventurelog` (a false positive: its compose lists two containers' ports and the probe targets the backend; measured `running`). **`home-assistant` is correctly CLEARED by the sharpened rule** — `type: api`, no `expect`, so its `/api/` answering 401 without a token is healthy, and its edge was PROVEN on the box tonight. The crude rule convicted it; the rule read from the code does not. **That is the gate to build: two of 53 convicted, one suspected, one false positive, and the false positive is resolvable by one live check.** Evidence: `audits/update-night-2026-09-21/10-probe-port-sweep.txt`, `12-probe-vs-compose-healthcheck.txt` and `13-probe-sweep-sharpened.txt`. **CLOSED 2026-09-22.** All three fixed in one commit (`app-catalog-felhom.eu@793c4fb`): tandoor `8080 -> 80`, wger `80 -> 8000`, zipline `/api/health -> /api/healthcheck`. No `image:` line moved, so no `catalog_since` moved. **RED-PROOFED LIVE ON 9202 THROUGH THE PRODUCT, BOTH DIRECTIONS.** Before the fix, at the LIVE pin, all three read **`Nem egészséges` / `Not healthy`** on their own app page while docker reported every container healthy and each front door served a real page through the household's own route — tandoor 200 `Login / Sign In`, zipline 200 `Zipline`, wger 200 `wger Workout Manager`. The fix was applied through the REAL sync (`POST /api/sync` answered *frissítve: tandoor, wger, zipline*) and all three read **`Fut` / `Running`** at the next poll, with no redeploy and no restart. **AND THE EDGE THAT FAILED WAS RE-WALKED AND PASSED:** tandoor `2.6.13 -> 2.6.15` via the drill catalog ended **`done` at +41.1 s** with the seed read back through tandoor's own front door and both containers running with zero restarts — where the identical edge on 2026-09-21 entered `verifying` at +58.4 s and ended `failed` at **+361.9 s** with the app stopped. **Same app, same versions, same button; the only change is one port number.** tandoor's verdict moved `failed -> proven` and it is now on the live catalog. **THE GATE SHIPPED WITH IT:** `scripts/check-probe-matches-compose.py`, a `--fast` row in `catalog_gates.py`, comparing the probe against the SAME service's own compose healthcheck. The rule was read out of `healthprobe.go` rather than guessed: a wrong PORT refuses for every check type; a wrong PATH refuses only for `type: api` WITH an `expect` block and WARNS otherwise, which is why `home-assistant` is warned about and not convicted. Four red-proofs and five decoys, plus five more with PyYAML shadowed out (R-630's sibling problem — see below). **Residual, filed separately:** R-630 (paperless-ngx's probe can never run at all) and R-631 (five templates the static rule cannot judge). | **NARROWED** (2026-10-03 triage: the row's verdict was finished, but it names open work no other row carries — the controller-side idea — let `verifying` accept docker's own `healthy` before it stops a working app — was raised and never decided) — **CLOSED 2026-09-22 — three probes fixed, red-proofed live both ways, gate shipped with decoys; tandoor re-walked `failed -> proven`** **2026-10-05 (burn-down night): NEEDS THE OPERATOR.** The three wrong probes and the gate shipped (2026-09-22); what is left is whether `verifying` may accept docker's own `healthy` — that widens the update guard (R-635: a probe can be green on a broken app). Next: the ruling. **RULED 2026-10-06 10:41 (`09` §3 decision 141): A — keep our own check only.** Being built. | — | — | CC + operator | | **R-734** | App updates | P4 | **[P3-LOW] The harness marks immich `files_may_change` because immich rewrites six 13-byte `.immich` folder markers at every start.** MEASURED 2026-09-30 on the bench (v3.2.2 → v3.2.4): the bind-tree hash of `appdata/immich` changed; the only changed files were `{encoded-video,library,backups,profile,thumbs,upload}/.immich`, rewritten at each start — no household file. The mark is honest by the harness's rule and the ladder writer copies it (never edited by hand), so immich's v3.2.4 night step needs a fresh WHOLE copy (decision 13); on a box without one the night leg skips it and a person presses. The 2026-09-23 immich entry did not carry it (`files_changed []`). **Needs:** a decision whether app-owned marker files are excluded from the file hash (a per-template ignore list, or a size/name rule), or the mark stays. **-- 2026-09-30 (evening):** the harness now NAMES the files behind the mark (`files_changed_detail`, catalog `5b1972b`); on immich's step `0b82…` re-proof it named exactly the six `.immich` markers again. calibre-web's step (v4.0.6 → v4.0.8) carries the mark too, and the named files are its LIBRARY DATABASE: `media/books/metadata.db`, `metadata.db-shm`, `metadata.db-wal` changed; the book file did not (bench names re-run, `A/calibre-names/`). That is household data (the template's backup class `mandatory` holds DB and books as one unit), so the mark is right there and the night leg takes that step only with a fresh whole copy. On 9202 the same step changed no file in that folder (per-file hashes before/after) — not explained. `audits/more-night-apps-2026-09-30/` | **READY — rank P3-LOW; owner: CC (harness); the rule change needs a word** **Re-ranked 2026-10-03: P3→P4: the effect is an update that waits for a press; no data risk.** **RULED 2026-10-06 10:41 (`09` §3 decision 145): A — a per-app list of marker files the update test ignores, each with a reason.** Being built. | — | — | CC + operator | ## Backup & restore — 34 rows (P2 8, P3 11, P4 15) @@ -176,7 +175,7 @@ stopping line that lies. | **R-645** | Backup & restore | P3 | **[P3-LOW] Lifting an update hold by the operator CLI lets the recovery unit be re-captured with the FAILED new definition within seconds — the copy the hold sentence names is overwritten.** MEASURED 2026-09-23 on 9202 during the undo bake-off: docmost was held at 07:59:16Z after a failed 0.95.0 → 0.96.0 update; `--clear-restore-hold docmost` + the controller restart it requires ran at ~07:59:23Z, and at **07:59:26Z** the controller logged *Recovery unit captured for docmost* — the unit's `compose/docker-compose.yml` now named `docmost/docmost:0.96.0`, the version that had just failed. The hold sentence had pointed the household at that unit („saját meghajtó, … 09:55"). The hold is what keeps the nightly legs off a held app (`isHeld`, v0.238.1); once it is lifted by hand, the checksum-gated refresh sees a changed definition and captures it. **Who it hits:** an operator who lifts a hold to inspect or repair, before restoring. With the undo (R-637) a failed update no longer holds unless the undo also fails, so the path is rarer — it does not go away. Candidate shapes, none chosen: the CLI refuses to lift an UPDATE hold (only a restore lifts it); or the lift also puts the pin back; or the capture skips an app whose pin is not what it is running. Evidence: `audits/undo-bakeoff-2026-09-23/docmost-40-undoF.txt` (the invalid run) and README §"Three things". **-- 2026-09-23 (controller v0.263.0):** the undo never reads the unit, so this no longer affects the automatic undo; it still affects an operator who lifts a hold by hand before restoring. | **OPEN — P3; owner: CC** **2026-10-05 (burn-down night): NEEDS THE OPERATOR.** Three shapes, none chosen (the CLI refuses to lift an UPDATE hold; the lift also restores the pin; the capture skips an app whose pin is not what it runs) — each changes the operator's tool or the nightly backup. Next: the pick. **RULED 2026-10-06 10:41 (`09` §3 decision 142): A — the night backup skips an app whose saved version is not the one it runs.** Being built. | — | — | CC | | **R-698** | Backup & restore | P3 | **[P3-LOW] A backup stores the image's NAME, not the image — a restore of a version its maker has deleted cannot start.** `RecoveryManifest.image_pins` ("image NOT stored — re-pulled on restore"); since controller v0.275.0 each data file also records its running `ref@digest`, and a restore brings the data back AT ITS OWN VERSION (`07` §6.6) — so a restore asks for exactly the old image. **Measured 2026-09-26** (`audits/version-travel-2026-09-26/A7/`, registry HEADs, no pulls): the catalog's 42 ladder `ref@digest` pairs all resolve (200); an invented digest answers 404 on Docker Hub and ghcr.io (negative control). Not measured: the digests recorded on boxes (older than any ladder entry), how often makers delete versions, the catalog's 66 digest-less compose lines. **Options (decide nothing yet):** (a) keep — a restore of a deleted version fails at the pull and the household uses the next copy or a newer version; (b) mirror every INSTALLED image into the DooPlex registry, restore falls back to it — storage + bandwidth on DooPlex, a new part on the recovery path; (c) mirror only ladder-named versions — bounded, misses pre-ladder boxes; (d) `docker save` into the unit — hundreds of MB per app per copy on every tier. **-- 2026-09-30 late (decision 53):** a box now keeps only an app's running and previous image; a restore to an older version re-pulls it — as every restore already did. The limit above is unchanged. | **OPEN — P3; owner: operator (a decision), CC measures** **Operator ruling 2026-10-05 18:23: kept OPEN as a known risk to a household's restore; owner the operator; not worked on in the burn-down.** | — | — | operator | | **R-91** | Backup & restore | P4 | Old 13 GB datastore copy at `/srv/pbs-felhom` on ep0's root disk **Checked from source 2026-10-05 (burn-down round 2):** Gate is long past (row waits on demo-felhom's first post-migration PBS backup, migration 2026-07-27). Last positive record of the copy: audits/CAMPAIGN-9-restore-proof-2026-07-28.md:759 'ep0 : /srv/pbs-felhom rollback copy intact (13G)'; CONTEXT.md:3666 still says it is 13 G of dead weight awaiting R-91. No later record of deletion found (grep srv/pbs-felhom across felhom.eu). Deleting is on ep0 (protected) and needs an operator word. | WATCHING | demo-felhom's first **post-migration** PBS backup | Delete once it lands; fix `CONTEXT.md:1018` same commit | CC | -| **R-99** | Backup & restore | P4 | Server-side prune **never removes** a phantom snapshot. Confirmed it does NOT count them toward `keep-last` (dry-run kept 2 real + the phantom) so there is **no retention/data-loss bug** — but one accumulates per aborted upload, forever | READY (S) **2026-10-05 (burn-down night): NEEDS THE OPERATOR.** Removing a phantom means deleting on a customer datastore (the row's own separate ruling), and the prune runs on ep0 (fenced). Next: the ruling. **RULED 2026-10-06 10:41 (`09` §3 decision 140): A — leftovers are deleted, by a runbook, when one is seen.** Being built. | — | Decide a cleanup path. Deletion on a **customer** datastore is a separate ruling — detection shipped, removal deliberately not automated | CC | +| **R-99** | Backup & restore | P4 | Server-side prune **never removes** a phantom snapshot. Confirmed it does NOT count them toward `keep-last` (dry-run kept 2 real + the phantom) so there is **no retention/data-loss bug** — but one accumulates per aborted upload, forever | READY (S) **2026-10-05 (burn-down night): NEEDS THE OPERATOR.** Removing a phantom means deleting on a customer datastore (the row's own separate ruling), and the prune runs on ep0 (fenced). Next: the ruling. **RULED 2026-10-06 10:41 (`09` §3 decision 140): A — leftovers are deleted, by a runbook, when one is seen.** Being built. **2026-10-06: runbook written** (`runbooks/pbs-phantom-cleanup.md` + `runbooks/pbs-phantom-list.py`). **Read-only listing of ep0 today: NO phantom** — 9 snapshots in 5 namespaces, all ≥ 369,808,250 B, all verification `ok`, every directory has its manifest; nothing deleted (`audits/ten-answers-2026-10-06/r99-ep0-listing.txt`). Left: the agent's WARN names the runbook (agent change, then release). | — | Decide a cleanup path. Deletion on a **customer** datastore is a separate ruling — detection shipped, removal deliberately not automated | CC | | **R-164** | Backup & restore | P4 | **C2's chain: the DB volume tar cannot be dropped until a SOUND dump predicate exists.** The unit carries both a volume tar and a SQL dump; the restore uses **both** — the dump is authoritative and replayed *after* the tar so it WINS (F17), with only the DB service up (R-47) — `internal/backup/restore_unit.go:262-266`. Dropping the DB container's tar would halve DB-app units **and** close the R-127(b) initdb-skip password trap (restored PGDATA ⇒ `POSTGRES_PASSWORD` ignored). | **BLOCKED** — on the predicate | a dump-validity predicate that is not `accounts has rows` | **The obvious gate is DEAD, measured:** `ValidateDump` warns when the `accounts` table is empty, and that warning was **correct** — the live DB genuinely had 0 accounts, and seeding one stopped the warning and put the row in the dump. But **a fresh appliance legitimately has zero accounts**, so promoting that predicate to a gate would **block every new customer's first backup**. Order: (1) a sound predicate — dump vs **live** per-table counts, not an absolute expectation; (2) warn→gate; (3) tar-drop. **Until (1), the tar is load-bearing** — not because dumps are bad, but because nothing can yet prove one is good. Pairs with **R-127** | CC | | **R-213** | Backup & restore | P4 | **Putting files back in place — the half the recovery screen deliberately does not do.** The screen (R-193, controller v0.200.0) unlocks the repository and LISTS what is in it; restoring is per-app and lives in the backups area, and the operator ruled the two separate on 2026-08-05: *a screen that unlocks and then offers to overwrite is two decisions wearing one button*. What is missing is the step after the listing — a customer who can now SEE their files still has to work out, per app, which restore to choose. **The operator named its requirement: a live-versus-backup comparison** — the customer must be able to see what would change before anything is overwritten | **OPEN — not started, deliberately** | the comparison design (nothing exists for it yet) | Design the live-vs-backup comparison, then the put-back flow on top of it. Do NOT fold it into the recovery screen | Operator + CC | | **R-246** | Backup & restore | P4 | **A leftover staleness flag has silently disabled the new recovery discriminator on `demo-hp` since 2026-08-04, and the flag is WRONG.** Found by a read-only spike, 2026-08-08. **Q1 — traced to an act, to the second:** at `2026-08-04 20:15:49` the hub emitted `offsite_reissued` and `escrow_stale` in the same second — an operator **Re-issue** pressed during the R-201 drill, three minutes after `escrow_blob_served` at 20:12:40/20:12:54. That was `offsite.ReissueCredentials`'s **precautionary** `MarkEscrowStale` call, which **hub v0.95.0 REMOVED the very next day** (R-196 / R-204 item 2) precisely because it marked healthy escrows stale. **Q2 — the flag is wrong, measured on both sides:** the hub's blob seals `restic_pw_sha256 = 8a9e33aa4da6769c…d080a`, and the key the box is actually using hashes to **the identical value**. The blob covers the key. **Q3 — nothing clears it by itself:** the ONLY writer of `stale_at = NULL` is `SaveHostEscrow`'s `ON CONFLICT` — i.e. a fresh escrow ceremony, **which is the one act that would supersede the good blob**. So the only exit from a false alarm is the destructive act the false alarm recommends. **Q5 — the blast radius, enumerated rather than assumed:** (1) `GetEscrowStatusForCustomer` withholds `restic_pw_sha256` from the ACK; (2) a **pending** box can never auto-confirm, so (3) **every off-site run is refused indefinitely** — neither bites `demo-hp`, which was already `escrowed` and is backing up healthily (12 snapshots, last success 2026-08-07T02:15:35Z); (4) the customer is told to create a new code; (5) **NEW — v0.206.0's shape (c) is inert**, because the box records an empty hub hash and falls back to (a)/(b), so the recovery screen would stay silent even if recovery were needed. **Q6 — a fresh box CANNOT reach this state:** `MarkEscrowStale` has **no production caller anywhere in the tree** (census: only its own definition, two comments and two test references). **The next walk cannot meet it.** **NOT CLEARED, deliberately** — Q2's answer says the fix is to clear this instance, but whether to also stop the column being settable at all is a separate ruling, and the spike was scoped read-only. **✅ THE FLAG IS CLEARED — operator-approved and applied 2026-08-08.** One row, identity-matched on `host_id` and guarded on `stale_at IS NOT NULL`; `changes()` returned **1**. **Verified end to end, not just in the database:** the hub now serves the hash again, the box recorded `hub_escrow_key_sha256 = 8a9e33aa4da6769c…d080a` at `11:10:19Z`, and that is **byte-identical to the key it is using** — so shape (c) compares, matches, and correctly stays silent. **The false stale warning is gone, proven with a positive control** rather than an absent line: **0** `escrow-confirm` lines since the restart while **5** scheduler lines in the same window prove the box was logging, and the recorded hash proves an ACK was processed. *(Method note: the hub pod is Alpine with no `sqlite3`; it was installed into the container's ephemeral writable layer — image and node untouched, gone on restart. SQLite's own file locking coordinated the write with the live hub; an earlier attempt failed cleanly on quoting and changed nothing, which is the fail-safe working.)* **STILL OPEN under this ID: the ruling on whether `stale_at` keeps a live setter at all.** It currently has NO production caller, so the column is write-only-by-accident — a field that changes behaviour, that nothing sets and nothing can see (R-248). Either give it an evidential setter or retire it; **do not leave it as a trap that only a database read can spring.** | **READY — flag cleared; the column ruling is still owed** — owner Viktor **Folded R-248 2026-10-03** (the same ruling: give `stale_at` a visible, evidence-bearing setter, or retire it). | — | — | operator | diff --git a/documentation/runbooks/pbs-phantom-cleanup.md b/documentation/runbooks/pbs-phantom-cleanup.md new file mode 100644 index 00000000..c856ee96 --- /dev/null +++ b/documentation/runbooks/pbs-phantom-cleanup.md @@ -0,0 +1,62 @@ +# Runbook — delete a phantom backup snapshot (an aborted upload's leftover) on the off-site backup server + +**Why it exists:** operator ruling 2026-10-06 (`architecture/09-update-architecture.md` §3 decision 140, R-99): *"Leftovers +should be deleted."* A PBS daemon killed in the middle of an upload leaves a snapshot that holds nothing (measured 1 B, +2026-07-28, F-CRIT-2). The box already refuses to count it as a backup (felhom-agent `internal/backup/runner.go`, +`archivePlausiblyComplete`, the 1 MiB floor) and logs it once at WARN. Server-side prune never removes it (dry-run +2026-07-28: keep-last 2 kept two real snapshots PLUS the phantom). So one accumulates per aborted upload, until a person +deletes it — by this runbook, when one is seen. **ep0 is Tier 2, protected** (`runbooks/target-selection.md`): this runbook +is the ONLY deletion the ruling allows there, and only of a snapshot this runbook proves to be a phantom. + +## When you see one + +The box's agent log says `… size N B is below the 1048576 B plausibility floor — an aborted/incomplete archive, not a +successful backup` (once per volid). Or a session reads one in step 1. + +## 1. List — read only + +From DooPlex (`ssh root@167.233.158.164`; never via `felhom-pve → 10.77.0.1`): + +```bash +# the server's own view, every namespace (size = what the manifest references; verification state) +for ns in $(ls /mnt/pbs-datastore/ns); do + proxmox-backup-debug api get /admin/datastore/felhom-offsite/snapshots --ns "$ns" --output-format json-pretty +done +# the directories themselves (a snapshot with no manifest may not be listed by the API at all) +python3 - < documentation/runbooks/pbs-phantom-list.py # run on ep0: ssh root@… python3 - < that file +``` + +**A snapshot is a PHANTOM only when BOTH hold:** +1. the server reports `size` below **1 MiB** (1,048,576 B) — or the directory has **no `index.json.blob`** (no manifest); +2. it is **not** `protected`, and it is **not the newest** snapshot of its group while an upload for that group may still be + running (check `proxmox-backup-manager task list --all --limit 20` — no running `backup` task for that group). + +The smallest real backup ever measured is ~584 MiB; real ones on 2026-10-06 were ≥ 352 MiB. Anything between 1 MiB and the +smallest real size is **NOT** a phantom by this runbook — leave it and ask the operator. + +**Write down, per namespace, the count of real snapshots** (everything that is not a phantom). This count is the control. + +## 2. Delete — one snapshot at a time, the command shown first + +```bash +# print it, read it, then run it +echo proxmox-backup-debug api delete /admin/datastore/felhom-offsite/snapshots \ + --ns --backup-type --backup-id --backup-time +proxmox-backup-debug api delete /admin/datastore/felhom-offsite/snapshots \ + --ns --backup-type --backup-id --backup-time +``` + +Never `rm -rf` a snapshot directory, never `prune`, never touch a datastore, a namespace or a group. One `delete` per +proven phantom. + +## 3. Prove nothing real moved + +Re-run step 1. **The count of real snapshots in every namespace must be exactly what it was before.** The deleted +phantom is gone; nothing else changed. If a count moved, stop and tell the operator at once. Save both listings (before, +after) under `documentation/audits//`. + +## The record of each use + +| Date | Namespace / group / time | Size | Why a phantom | Real counts before = after | +|---|---|---|---|---| +| 2026-10-06 | — | — | **No phantom existed** (9 snapshots, all ≥ 369,808,250 B, all verification `ok`; no directory without a manifest) — nothing deleted | Tester-2 1, demo-felhom 3, demo-hp 2, operator 2, tester-1 1 (`audits/ten-answers-2026-10-06/r99-ep0-listing.txt`) | diff --git a/documentation/runbooks/pbs-phantom-list.py b/documentation/runbooks/pbs-phantom-list.py new file mode 100644 index 00000000..9d87e3b6 --- /dev/null +++ b/documentation/runbooks/pbs-phantom-list.py @@ -0,0 +1,23 @@ +import os, re, json +root = "/mnt/pbs-datastore" +rows = [] +def walk_ns(base, ns): + for typ in ("ct", "vm", "host"): + tdir = os.path.join(base, typ) + if not os.path.isdir(tdir): continue + for gid in sorted(os.listdir(tdir)): + gdir = os.path.join(tdir, gid) + for snap in sorted(os.listdir(gdir)): + sdir = os.path.join(gdir, snap) + if not os.path.isdir(sdir) or not re.match(r"\d{4}-\d{2}-\d{2}T", snap): continue + files = os.listdir(sdir) + size = sum(os.path.getsize(os.path.join(sdir, f)) for f in files) + man = "index.json.blob" in files + manifest_files = [] + rows.append({"ns": ns or "(root)", "group": f"{typ}/{gid}", "snap": snap, "files": sorted(files), "dir_bytes": size, "manifest": man}) + nsdir = os.path.join(base, "ns") + if os.path.isdir(nsdir): + for n in sorted(os.listdir(nsdir)): + walk_ns(os.path.join(nsdir, n), (ns + "/" if ns else "") + n) +walk_ns(root, "") +print(json.dumps(rows))