Compare commits
322 Commits
d9da5a0678
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 2fa1efc5e5 | |||
| 330e4a051e | |||
| 90f2545679 | |||
| 8144a70a72 | |||
| 34d83f5a02 | |||
| c24f1920d9 | |||
| bb50e1293c | |||
| 3e3ee94b7b | |||
| ae10f64806 | |||
| 3ed5e3e770 | |||
| 3168a78935 | |||
| 89712563a0 | |||
| 1b66010298 | |||
| 68f3e12398 | |||
| f87be3575f | |||
| 38f4535bfa | |||
| 397d62136f | |||
| 86a78c6767 | |||
| b762a37097 | |||
| c732fe1283 | |||
| fcffaf573a | |||
| 37b5ba08a7 | |||
| 27d1165962 | |||
| 62998aab4f | |||
| 8dbbc98ff2 | |||
| 3d3b4496f3 | |||
| 0a9158d53e | |||
| 72368654e4 | |||
| de39e47f53 | |||
| a5d90ff801 | |||
| a491abef6c | |||
| 763de3a025 | |||
| c6b69d888e | |||
| 53e9bf0224 | |||
| 4d349d1106 | |||
| 9dc26459ea | |||
| 66d80efb9f | |||
| 7db42c5fec | |||
| a62bb3874b | |||
| 7534ea203d | |||
| c7446f2d6a | |||
| 1e759a16ec | |||
| 05cf352a2f | |||
| a3499d1807 | |||
| a315d623b8 | |||
| 62b85ecf13 | |||
| 636c51e542 | |||
| be3c5fa7f6 | |||
| 992803c10b | |||
| a91f055960 | |||
| 1214bae0a2 | |||
| 68f195676b | |||
| 33fcc502e4 | |||
| 2e936f43bf | |||
| 73b6dbc27d | |||
| f4796e0d00 | |||
| 58c703bd44 | |||
| a96c3d9473 | |||
| 73efb091d9 | |||
| 532f5712a8 | |||
| 1b1366bb6e | |||
| bdab80c933 | |||
| 9640e51321 | |||
| 0887fd676d | |||
| 88897a224e | |||
| db0d4b129d | |||
| 6c43bf6156 | |||
| fef07c3923 | |||
| 4be6467b50 | |||
| d5be67b913 | |||
| 9a3c4855d7 | |||
| 5adae4dad9 | |||
| cf48214f6c | |||
| 95eb5c2c1a | |||
| e6311f9fbc | |||
| 4bad6e06c9 | |||
| dcc3363d2f | |||
| 582135f861 | |||
| 3446609420 | |||
| a8f7c61d41 | |||
| dbcb306fcf | |||
| e7c44c0e0f | |||
| dcc400e175 | |||
| eaded79b18 | |||
| 7c32c74140 | |||
| 8cb3d7af91 | |||
| c432f701dd | |||
| 4115e88f68 | |||
| 4ed938cce4 | |||
| 2f27a363d5 | |||
| b331f18424 | |||
| cdaeb36972 | |||
| 3f7cf2a965 | |||
| 4d6c8a6056 | |||
| c1a63de1c7 | |||
| ff058a4f10 | |||
| fd50a73e65 | |||
| d8b3279731 | |||
| 3f048e042b | |||
| 3db8bfb953 | |||
| e000e201af | |||
| 4056feccee | |||
| fb91c8d766 | |||
| a63409c843 | |||
| 079265ad8e | |||
| 8f46495426 | |||
| ca013c8d27 | |||
| 86ea482fc1 | |||
| ba8bf9cd75 | |||
| e9c99566b0 | |||
| ccefff4f39 | |||
| b8598361b8 | |||
| 32200c7b5f | |||
| 3f0420ff9c | |||
| f5e106440d | |||
| 9e5ea56853 | |||
| de96efc0c5 | |||
| 47fda06ba1 | |||
| 9056f01fae | |||
| c7a3a90782 | |||
| 8fadbd9891 | |||
| 4773809334 | |||
| 2958946517 | |||
| 3b672ba74c | |||
| 7d5b0163ef | |||
| f6a8249593 | |||
| de14eedb9f | |||
| 9cc8424954 | |||
| 2487681396 | |||
| 0a582ea07b | |||
| dbf631312e | |||
| c97975c1df | |||
| e164fef70c | |||
| 82c67e32e1 | |||
| e33c1aeabc | |||
| d37bb1eb6a | |||
| 59d182a6b9 | |||
| e9a365e59c | |||
| dbd1ff0c17 | |||
| bf44216e79 | |||
| 1fd070615d | |||
| a04afc367b | |||
| 570fb30147 | |||
| 15206314ab | |||
| 8e5edb2865 | |||
| c23a0f6d2d | |||
| 77956d8df2 | |||
| 2c80868c63 | |||
| 7a53cb43ef | |||
| ea432ca74a | |||
| 4aa7d41cac | |||
| 987e915bf2 | |||
| a71cc58327 | |||
| cb8bf14599 | |||
| ce8531426c | |||
| 0eba37d5cd | |||
| 59cd260e57 | |||
| 9610906916 | |||
| 7013a5fd2e | |||
| 4aa2ce4b61 | |||
| 0fbd272ad2 | |||
| 5fdd2039fd | |||
| ea0d3f1764 | |||
| a96226a8ae | |||
| 4ab793ef31 | |||
| 94fa4d6fb9 | |||
| ac3790a11b | |||
| 83f20c8293 | |||
| 1dad3c97fd | |||
| 984ea8c8bd | |||
| 285dd1032f | |||
| 0f9b29a19a | |||
| f134eab609 | |||
| 9d1b4983f5 | |||
| 70cb21b058 | |||
| 3a9d744360 | |||
| b30e2e5a28 | |||
| 8d33a90888 | |||
| 332b2b84bc | |||
| 4978d355b8 | |||
| 78ff991f1c | |||
| fd40b29119 | |||
| 37e12c82a7 | |||
| 5c105fb49b | |||
| badf17bebd | |||
| 8db9232dea | |||
| 9f436c8a3b | |||
| 4646be1515 | |||
| c059fe4c28 | |||
| 9d001771ea | |||
| 29eda5d86e | |||
| 062357f778 | |||
| 2fcae041ae | |||
| ac7323dc9a | |||
| da56c3e994 | |||
| 86de16f6dd | |||
| 10ca8ac884 | |||
| 63a22e5911 | |||
| 111369dd10 | |||
| 77e8d5590b | |||
| b5d78d1e0f | |||
| 7f0b41c3e7 | |||
| fd93020f39 | |||
| 24d23b80bd | |||
| 57be785bad | |||
| cc50918244 | |||
| a5d870ea0e | |||
| d0c1d77491 | |||
| 3b70a9e9ab | |||
| 900c870212 | |||
| 85b76e0fc3 | |||
| c81df55dcb | |||
| 3dfc49e578 | |||
| a4c82a2651 | |||
| b409f5eee2 | |||
| 2eef9b2e4a | |||
| 4f08e5e7c3 | |||
| 1d26a69dd4 | |||
| 0dcbea90b2 | |||
| b0c5ef4823 | |||
| f42f3e0e08 | |||
| ac13966f30 | |||
| a4a7de3d8f | |||
| 3286c7faaf | |||
| f900c83eed | |||
| 596505ed64 | |||
| 1452dd2b17 | |||
| 0515d153db | |||
| 0025a3b090 | |||
| 5be2449267 | |||
| 2dd05670ae | |||
| f665bbed45 | |||
| fe9266f53f | |||
| 8f3564c137 | |||
| 120332103a | |||
| e99c675fe4 | |||
| c124d0adf2 | |||
| 0ef6648e12 | |||
| cf9ce01917 | |||
| 5f216138e5 | |||
| 3603d1fc7f | |||
| 1245a6c46e | |||
| 483b2186cb | |||
| 0cfcc42464 | |||
| 482d0d98e3 | |||
| 2d20859858 | |||
| 0c6e151c1c | |||
| 2668ac4da3 | |||
| 95f3180ab4 | |||
| 0649f9a3e6 | |||
| af98c53c82 | |||
| 68f0e0cf5c | |||
| b42904bbab | |||
| 3679a759ba | |||
| b49076db4b | |||
| a829cdc91f | |||
| c6d8bc82a2 | |||
| 8967ba7cb4 | |||
| eb3bf4a391 | |||
| 7465713a2f | |||
| 3c9de42c20 | |||
| d752f159b9 | |||
| 61c1391ff6 | |||
| db16371f47 | |||
| 59e4ca70de | |||
| 8eed73b38a | |||
| 900adf733c | |||
| 788f8e65fa | |||
| 6f72e5e0c2 | |||
| 156e7d5565 | |||
| a5041dac80 | |||
| 08a966b92f | |||
| 51c871ad9b | |||
| c739003379 | |||
| 6136461de6 | |||
| f8df1e9f16 | |||
| 93c95f0fb5 | |||
| 21d0f66c15 | |||
| 41c8fc23cf | |||
| 912e6a62b6 | |||
| 02c84e1ce2 | |||
| 458cf4a4c2 | |||
| 67494f9890 | |||
| 4f06149abc | |||
| 3eacb6c326 | |||
| 7b3e7f2e71 | |||
| e0618af438 | |||
| 466f42708e | |||
| 811a0ef7a7 | |||
| 753cd83456 | |||
| aa967fbf69 | |||
| 0ece2ba85c | |||
| df7ad378f4 | |||
| b482860c03 | |||
| a00afcc79d | |||
| 8987ce0f67 | |||
| 40b53047f1 | |||
| b4e0a197f9 | |||
| d4641c221d | |||
| 3cf49c7fd5 | |||
| dec6fef20d | |||
| 9d05fa5c35 | |||
| 92d670e8a6 | |||
| 40498254c6 | |||
| e030b8d9a9 | |||
| 6ab8f943af | |||
| d8f6069b46 | |||
| a6da64da15 | |||
| 5d91fc8cce | |||
| 0f311adaa1 | |||
| e3903be0f1 | |||
| 68b3a3932e | |||
| b389e73fff | |||
| 4a9c54a105 | |||
| c0f3e12483 | |||
| 6e9dd1bfa1 | |||
| 458d7e1ebc | |||
| 26a43708b7 | |||
| 647116480a | |||
| b6842a4bc3 | |||
| ac6adaaa44 | |||
| 53ab971fe4 |
@@ -0,0 +1,43 @@
|
||||
---
|
||||
paths: ["controller/internal/agentapi/**"]
|
||||
---
|
||||
|
||||
# Coupling to the host agent — felhom-controller
|
||||
|
||||
`internal/agentapi` is **the disk seam**: the pinned-TLS client to the host agent's per-guest local
|
||||
API. The controller holds no Proxmox credentials; everything disk/host/Proxmox goes through here.
|
||||
|
||||
## Declaring a coupled feature
|
||||
|
||||
Controller behaviour that depends on a specific agent version needs **all three**, or it ships broken
|
||||
on an older box:
|
||||
|
||||
1. a `featureProbes` table row in `internal/agentapi/features.go`
|
||||
2. a `Supports` gate call **at the feature's entry point** — not somewhere on the path to it
|
||||
3. `MinAgent: X.Y.Z` in the CHANGELOG entry header
|
||||
|
||||
Rules: `felhom.eu/documentation/runbooks/publish-train-rules.md`.
|
||||
|
||||
## Never push a controller past the agent it depends on
|
||||
|
||||
The R-216 guard compared the box's agent against the **golden's** MinAgent while serving a **floor**
|
||||
that could point elsewhere. Raise a floor above the vouched golden — which the day-0 runbook
|
||||
recommends and a per-customer override makes trivial — and the guard checks a version it is not
|
||||
serving. A box then landed on a controller needing a newer agent, and its customer was told a correct
|
||||
recovery code was wrong.
|
||||
|
||||
**A floor above the vouched golden is HELD, with its own reason** (hub v0.97.0).
|
||||
|
||||
## Distinguish "could not reach" from "wrong answer"
|
||||
|
||||
A failed bundle FETCH must not be reported to a customer as a bad recovery code. Classify by **value**
|
||||
(`ErrBundleFetch` → HTTP 502), never by error string — a string is not something a caller can branch
|
||||
on. Unknown class → neutral message, never the typing message.
|
||||
|
||||
<!--
|
||||
R-224, measured live 2026-08-05 (CAMPAIGN-11 F3/F4) with a correct current code: 0.0556 s with the
|
||||
hub firewalled off and 0.0299 s with the agent stopped, against ~1.0 s for a genuine unseal — the
|
||||
machine accused the customer of something it had not attempted. A green test named this exact
|
||||
consequence since v0.125.0 and did not prevent it, because it asserted this package's error STRING
|
||||
one layer below where the merge happened. Fixed agent v0.126.0 + controller v0.202.0.
|
||||
-->
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
paths: ["controller/internal/backup/**", "controller/internal/appbackup/**", "controller/internal/recovery/**", "controller/internal/appexport/**", "controller/internal/quiesce/**"]
|
||||
---
|
||||
|
||||
# Backup, recovery units and export — felhom-controller
|
||||
|
||||
## Assert the consequence across the whole run, not the mechanism inside one function
|
||||
|
||||
The R-181 recovery-unit refusal claimed *"the previous unit is untouched and NOTHING was deleted"*.
|
||||
*Nothing deleted* held; **untouched was measured false** — the floor was checked ONLY in
|
||||
`captureAllRecoveryUnits`, while the two dump legs wrote the bulk into the same tree first and
|
||||
unguarded, so a 182,272 B tar became 2,147,666,432 B under a manifest that had not moved. A full
|
||||
green suite plus three of its own red-proofs missed it, because every one asserted the mechanism
|
||||
inside `captureAllRecoveryUnits`.
|
||||
|
||||
**The test that catches this class: fingerprint the tree before and after the whole backup run, and
|
||||
compare.** Full doctrine and the other eight instances: the `felhom-testing` skill.
|
||||
|
||||
## Presence is not success
|
||||
|
||||
A timestamp recording an **attempt** must never be read as evidence of a **result**. Where a status
|
||||
field travels alongside a timestamp, the verdict consults both — or the timestamp records only
|
||||
successes. Ask of any timestamp: *what exactly must have happened for this to be set?* If the answer
|
||||
is "we tried", it cannot answer "did it work".
|
||||
|
||||
**Corollary:** when a verdict changes which field it counts from, the alarm text has to change with
|
||||
it. `last run 8h ago` while alarming on a six-day-old success turns a true alarm into one the
|
||||
operator dismisses.
|
||||
|
||||
<!--
|
||||
Two instances. F-CRIT-2: a phantom snapshot's ctime set tier freshness — an aborted 1-byte upload
|
||||
made the tier look backed up. R-100: LastRun is written on failure, so a nightly-failing offsite
|
||||
tier kept the staleness clock fresh forever.
|
||||
-->
|
||||
|
||||
## Storage keys and paths
|
||||
|
||||
- Never guess a persisted key — it is `offbox`, not `offbox_target` (R-7b).
|
||||
- `.fab` export/import uses strict segment validation; bundles from controller ≤0.124.0 are hollow.
|
||||
- Recovery-unit restore and tier-2 copies share `appbackup`'s path primitives — change them there,
|
||||
once, not per caller.
|
||||
@@ -0,0 +1,54 @@
|
||||
---
|
||||
paths: ["controller/**/*.go", "controller/**/*.html", "controller/**/*.css", "controller/scripts/**"]
|
||||
---
|
||||
|
||||
# Gates and logging — felhom-controller
|
||||
|
||||
## The ONE entry point
|
||||
|
||||
**Run `python3 controller/scripts/controller_gates.py` (from `controller/`) after ANY change in this
|
||||
repo.** It runs all seven local gates — `template_id_gate`, `emoji_gate`, `native_confirm_gate`,
|
||||
`offbox_rename_gate`, `app_row_dedup_gate`, `mojibake_gate`, `docker_run_volume_path_gate` — plus
|
||||
`reuse_refs_check` and `instructions_gate` on the repo root, streaming each gate's own output and
|
||||
exiting non-zero if any fails.
|
||||
|
||||
- `--fast` selects the gates that touch no network and no container runtime; today that is all of them.
|
||||
- **A missing gate script is a FAILURE, never a skip.**
|
||||
- **The shared `reuse_refs_check.py` and `instructions_gate.py` live in `felhom.eu/scripts/` and are
|
||||
never copied here** — a copy would recreate the drift they detect; an absent sibling clone FAILS.
|
||||
- **The pre-push hook** (`.githooks/pre-push`) runs it with `--fast` and refuses a failing push. It is
|
||||
per-clone — switch it on once with `git config core.hooksPath .githooks`, and a manual run WARNS
|
||||
when this clone is unarmed. `git push --no-verify` bypasses it deliberately; **say so in the session
|
||||
report when you use it** — CI re-runs the same entry point on every push and **emails the operator
|
||||
on failure**, so a bypass is noticed even though it is not blocked (R-168, CLOSED 2026-08-02).
|
||||
|
||||
<!--
|
||||
WHY A RUNNER AND NOT SEVEN INVOCATIONS (2026-08-02, R-29) — rationale, not a directive.
|
||||
A census of all thirteen gates across the four repos found that every check a CLAUDE.md named was
|
||||
passing, and two of the four nobody is told to run were failing. This repo's CLAUDE.md used to name
|
||||
two of the seven; the other five were reachable only through a line in REUSE.md, and
|
||||
docker_run_volume_path_gate.py was RED. The single-entry-point shape is the only one that
|
||||
demonstrably gets run. app-catalog-felhom.eu/scripts/catalog_gates.py is the canonical version of
|
||||
the runner (R-161); repo_gates.py copies it. site_gates.py is a *gate*, not a runner — do not model
|
||||
new work on it.
|
||||
-->
|
||||
|
||||
## Logging
|
||||
|
||||
New leveled lines use `internal/logx` — DEBUG always reaches the debug ring; stdout respects
|
||||
`logging.level`. English, keys-never-values, durations on outcomes. Full rules:
|
||||
`felhom.eu/documentation/runbooks/logging-conventions.md`.
|
||||
|
||||
## Health checks issue no block I/O
|
||||
|
||||
A probe that touches a wedged device enters uninterruptible sleep, survives `SIGKILL`, and cannot be
|
||||
recovered until the device returns or the host reboots — so `systemctl restart` hangs too. A timeout
|
||||
protects the caller's control flow and nothing else: the blocked thread remains. Liveness is decided
|
||||
from `/proc` and kernel state, never by reading or writing the filesystem.
|
||||
|
||||
<!--
|
||||
Measured, R-117 spike §6.3 (felhom.eu/documentation/audits/SPIKE-r117-bind-liveness-2026-07-30.md):
|
||||
a probe stayed in D state 3m50s after kill -9; a buffered write with no fsync blocked too (O_CREAT
|
||||
needs journal access); and statfs/getdents returned HEALTHY on a namespace that EIOs every byte —
|
||||
fast, and wrong.
|
||||
-->
|
||||
@@ -0,0 +1,27 @@
|
||||
---
|
||||
paths: ["controller/internal/web/templates/**", "controller/internal/web/**/*.go", "**/*.css", "**/*.html"]
|
||||
---
|
||||
|
||||
# UI and Hungarian copy — felhom-controller
|
||||
|
||||
- **All UI text is Hungarian**, Budapest timezone.
|
||||
- Design tokens, badge/colour rules, the 2px/no-shadow/no-emoji/BOM hard rules and the mechanical
|
||||
gate to run after each surface: **use the `felhom-ui-design` skill.**
|
||||
- Template methods need **value receivers** — pointer receivers compile, pass `go vet`, pass the
|
||||
suite, and then 500 at render time.
|
||||
|
||||
## Grep fetched pages with ASCII-only substrings
|
||||
|
||||
Accented Hungarian patterns get mangled through the `ssh → pct exec → bash -c` chain and return a
|
||||
false `0` — which reads exactly like the banner or string being gone. Use `kezel`, `Utols`,
|
||||
`Biztons`. **Never let an accented pattern gate a conclusion.**
|
||||
|
||||
<!--
|
||||
From the 2026-07-20 remediation: an accented grep nearly produced a wrong "banner cleared" claim.
|
||||
This is the "an absent line is not evidence" rule aimed at a UTF-8 transport, not at a log.
|
||||
-->
|
||||
|
||||
## Credentials containing `!` or `'` break in heredoc-built helper scripts
|
||||
|
||||
History expansion eats `!!`. Use the proven inline `-d "password=$PW"` form for authed curl, and
|
||||
delete any credential-bearing helper from `/tmp` (host AND guest) when done.
|
||||
@@ -0,0 +1,104 @@
|
||||
# gates — re-run this repo's gate entry point on every push, on a machine that does not care who
|
||||
# pushed or what they typed.
|
||||
#
|
||||
# *** THIS REPORTS. IT CANNOT REFUSE. ***
|
||||
#
|
||||
# felhom repos push straight to `main` with no pull request, so there is no merge for a status
|
||||
# check to stand at. The refusing half is `.githooks/pre-push`, which is local to a clone and which
|
||||
# `git push --no-verify` skips; this half is what notices when that happened. Neither half is the
|
||||
# whole thing, and both are named in felhom.eu documentation/backlog/OPEN-ITEMS.md R-168.
|
||||
#
|
||||
# NO `uses:` STEP ANYWHERE, deliberately: JavaScript actions need a node runtime in the runner, and
|
||||
# the runner is a host-mode container with python3 and git and nothing else (see
|
||||
# homelab-manifests/gitea-system/act-runner.yaml for why it is not privileged). Probe P3 measured
|
||||
# that a plain `git fetch` of the pushed SHA from the in-cluster Gitea service is enough.
|
||||
#
|
||||
# A failing run must reach a person — a detector nobody hears is the defect R-29 filed, rebuilt one
|
||||
# layer up. That is the last step, and it runs ONLY on failure.
|
||||
name: gates
|
||||
on: [push]
|
||||
|
||||
jobs:
|
||||
gates:
|
||||
runs-on: felhom-gates
|
||||
steps:
|
||||
- name: Fetch the pushed commit and the sibling clone it needs
|
||||
# This repo's entry point invokes a SHARED checker that lives in the felhom.eu clone next
|
||||
# door and is deliberately never copied here — so CI has to reproduce the workspace's
|
||||
# sibling layout or the gate fails closed with "gate is MISSING". The sibling is also
|
||||
# needed for CONTENT: this repo's REUSE.md cites a path that lives in the hub.
|
||||
run: |
|
||||
# Shallow, and pinned to the exact SHA that was pushed — not to the branch tip,
|
||||
# which can move under us if two pushes race.
|
||||
mkdir -p ws/felhom-controller
|
||||
cd ws/felhom-controller
|
||||
git init -q .
|
||||
git remote add origin http://gitea.gitea-system.svc.cluster.local:3000/admin/felhom-controller.git
|
||||
git fetch -q --depth 1 origin "$GITHUB_SHA"
|
||||
git checkout -q FETCH_HEAD
|
||||
echo "checked out $(git rev-parse HEAD)"
|
||||
cd .. && git clone -q --depth 1 http://gitea.gitea-system.svc.cluster.local:3000/admin/felhom.eu.git felhom.eu
|
||||
echo "sibling felhom.eu present at $(cd felhom.eu && git rev-parse --short HEAD)"
|
||||
|
||||
- name: Run the gate entry point
|
||||
# The ONLY thing CI runs. No go build, no go test, no linting, no deploy. The
|
||||
# exit code IS the result: no `|| true`, no pipe that could swallow it.
|
||||
run: cd ws/felhom-controller/controller && python3 scripts/controller_gates.py --fast
|
||||
|
||||
- name: Alarm on failure
|
||||
# THE POINT OF THE WHOLE THING. Probe P5 measured that a failed run produces NO mail, NO
|
||||
# notification row and NO log line from Gitea itself — a red tick in a web UI nobody watches
|
||||
# is exactly the shape R-29 filed against. So the run sends its own alarm, on the project's
|
||||
# existing transactional path (Resend, the same one the hub uses), and prints the provider's
|
||||
# accepted id so "a message left the machine" is an observable, not an assumption.
|
||||
#
|
||||
# Pure python3 and urllib, NOT curl: the runner image carries python3 and git and nothing
|
||||
# else on purpose, and the first version of this step died on `curl: command not found`.
|
||||
# Reaching for a bigger image to send one HTTP request would have been the wrong trade.
|
||||
if: failure()
|
||||
env:
|
||||
RESEND_API_KEY: ${{ secrets.RESEND_API_KEY }}
|
||||
run: |
|
||||
python3 - <<'PY'
|
||||
import json, os, sys, urllib.request, urllib.error
|
||||
|
||||
key = os.environ.get("RESEND_API_KEY", "")
|
||||
if not key:
|
||||
sys.exit("ALARM FAILED: RESEND_API_KEY is empty — the alarm cannot be sent, and a "
|
||||
"silent alarm is worse than none. Set the user-level Actions secret.")
|
||||
|
||||
repo = os.environ.get("GITHUB_REPOSITORY", "?")
|
||||
sha = os.environ.get("GITHUB_SHA", "?")
|
||||
run = os.environ.get("GITHUB_RUN_NUMBER", "?")
|
||||
srv = os.environ.get("GITHUB_SERVER_URL", "https://gitea.dooplex.hu")
|
||||
|
||||
body = json.dumps({
|
||||
"from": "Felhom CI <monitoring@felhom.eu>",
|
||||
"to": ["admin@felhom.eu"],
|
||||
"subject": "[felhom CI] gates FAILED in %s" % repo,
|
||||
"text": (
|
||||
"The gate entry point exited non-zero.\n\n"
|
||||
"Repository : %s\n"
|
||||
"Commit : %s\n"
|
||||
"Run : %s/%s/actions/runs/%s\n\n"
|
||||
"The failing gate names itself in the run log.\n\n"
|
||||
"If the local pre-push hook was GREEN for this commit, then CI and the hook\n"
|
||||
"disagree - that is a finding about the gates themselves, not about CI, and it\n"
|
||||
"outranks whatever the push was for.\n"
|
||||
) % (repo, sha, srv, repo, run),
|
||||
}).encode()
|
||||
|
||||
req = urllib.request.Request(
|
||||
"https://api.resend.com/emails", data=body, method="POST",
|
||||
headers={"Authorization": "Bearer %s" % key,
|
||||
"Content-Type": "application/json",
|
||||
# Cloudflare fronts api.resend.com and BLOCKS the default
|
||||
# "Python-urllib/3.x" agent with its own 403 (error 1010) — which looks
|
||||
# exactly like an auth failure and is not one. Measured 2026-08-02.
|
||||
"User-Agent": "felhom-ci/1.0"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
print("RESEND-ACCEPTED id=%s" % json.load(r)["id"])
|
||||
except urllib.error.HTTPError as e:
|
||||
sys.exit("ALARM FAILED: Resend returned HTTP %s: %s" % (e.code, e.read().decode()[:300]))
|
||||
PY
|
||||
Executable
+82
@@ -0,0 +1,82 @@
|
||||
#!/bin/sh
|
||||
# pre-push — refuse a push that carries a broken gate. (2026-08-02, R-29 leg (b) first half.)
|
||||
#
|
||||
# Runs this repo's ONE gate entry point in --fast mode: only checks that touch no network and no
|
||||
# container runtime, so a push stays a push and never pulls images or starts containers. The slow
|
||||
# gates stay deliberate periodic runs; a hook that takes minutes gets bypassed within a week and
|
||||
# the bypass becomes the habit.
|
||||
#
|
||||
# BOTH LINES BELOW ARE DELIBERATE. An absent log line is not evidence a hook ran — a silent pass is
|
||||
# equally consistent with "gates green" and "hook never fired", so a passing push says so out loud.
|
||||
#
|
||||
# HONEST LIMITS, stated so this is not mistaken for enforcement it cannot provide:
|
||||
# * per-clone — core.hooksPath is local config and a clone does not carry it. Arm a clone once:
|
||||
# git config core.hooksPath .githooks
|
||||
# Any manual entry-point run WARNS when the clone is unarmed.
|
||||
# * skippable — `git push --no-verify` bypasses this entirely. That is on purpose: an escape
|
||||
# hatch that cannot be reached is one that gets removed the first time it is
|
||||
# inconvenient. USING IT MUST BE STATED IN THE SESSION REPORT.
|
||||
# The half that is neither per-clone nor skippable is CI — felhom.eu OPEN-ITEMS.md R-168.
|
||||
#
|
||||
# Measured 2026-08-02 (git 2.47.3): a relative core.hooksPath resolves correctly and the hook's cwd
|
||||
# is the repo root whether `git push` is issued from the root or from any subdirectory. The
|
||||
# explicit rev-parse below does not depend on that.
|
||||
set -u
|
||||
|
||||
root=$(git rev-parse --show-toplevel 2>/dev/null) || {
|
||||
echo "pre-push: FAIL - cannot resolve the repo root (git rev-parse --show-toplevel)." >&2
|
||||
exit 1
|
||||
}
|
||||
cd "$root" || exit 1
|
||||
|
||||
# ── WORKSPACE-ROOT ASSERTION (2026-08-05, R-204 rider) ───────────────────────────────────────────
|
||||
# Refuse a push from a clone outside the felhom workspace.
|
||||
#
|
||||
# WHY THIS IS A HOOK AND NOT A LINE IN A DOCUMENT: the workspace root is ALREADY written down, in
|
||||
# documentation/runbooks/workspace-CLAUDE.md and in the workspace-root CLAUDE.md ("stay inside it"),
|
||||
# and work drifted into a home directory anyway. A rule that has failed once as a reminder is not
|
||||
# fixed by writing it down again — it has to be asserted where it can bite.
|
||||
#
|
||||
# A PUSH IS THE RIGHT TRIGGER, deliberately: throwaway clones under /tmp for probes and red-proofs
|
||||
# never push, so nothing legitimate breaks. Reads and builds elsewhere stay unaffected.
|
||||
#
|
||||
# Symlinks are resolved on BOTH sides before comparison, so a symlinked path neither falsely passes
|
||||
# nor falsely fails. If the workspace root does not exist on this machine the check is SKIPPED, not
|
||||
# failed — this hook must not brick a legitimate clone on a different host.
|
||||
#
|
||||
# The only bypass is the documented `git push --no-verify`, whose use is already reportable.
|
||||
FELHOM_WORKSPACE_ROOT=/mnt/5_hdd/felhom.eu
|
||||
if [ -d "$FELHOM_WORKSPACE_ROOT" ]; then
|
||||
ws_real=$(cd "$FELHOM_WORKSPACE_ROOT" 2>/dev/null && pwd -P) || ws_real=""
|
||||
root_real=$(pwd -P) || root_real=""
|
||||
if [ -n "$ws_real" ] && [ -n "$root_real" ]; then
|
||||
case "$root_real/" in
|
||||
"$ws_real"/*) : ;; # inside the workspace — proceed
|
||||
*)
|
||||
echo "pre-push: PUSH REFUSED - this clone is OUTSIDE the felhom workspace." >&2
|
||||
echo " clone: $root_real" >&2
|
||||
echo " expected: under $ws_real (repos live in $ws_real/git/<repo>)" >&2
|
||||
echo " Work in the workspace clone, or bypass with 'git push --no-verify'" >&2
|
||||
echo " and state that you did in the session report." >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
fi
|
||||
|
||||
if ! command -v python3 >/dev/null 2>&1; then
|
||||
echo "pre-push: FAIL - python3 not found, so the gates CANNOT run. This is a failure, never a" >&2
|
||||
echo " pass by default. Install python3, or push with --no-verify and say so." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "pre-push [felhom-controller]: running controller/scripts/controller_gates.py --fast ..."
|
||||
python3 "controller/scripts/controller_gates.py" --fast
|
||||
rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo "pre-push [felhom-controller]: PUSH REFUSED - gates exited $rc. Fix the finding above, or bypass with" >&2
|
||||
echo " 'git push --no-verify' and state that you did in the session report." >&2
|
||||
else
|
||||
echo "pre-push [felhom-controller]: gates OK - push proceeding."
|
||||
fi
|
||||
exit $rc
|
||||
+5322
File diff suppressed because it is too large
Load Diff
@@ -1,168 +1,110 @@
|
||||
# CLAUDE.md — Project Instructions for Claude Code (`felhom-controller`)
|
||||
# CLAUDE.md — `felhom-controller`
|
||||
|
||||
> Read automatically at session start. Stable orientation only — **current state lives in
|
||||
> `CONTEXT.md` and the top of `CHANGELOG.md`**, never here. Cross-repo orientation: workspace-root
|
||||
> `e:\git\CLAUDE.md`.
|
||||
> Stable orientation only — **current state lives in `CONTEXT.md` and the top of `CHANGELOG.md`**,
|
||||
> never here. Cross-repo conventions (clean-tree gate, secrets, trunk-based, artifact taxonomy):
|
||||
> workspace-root `/mnt/5_hdd/felhom.eu/git/CLAUDE.md`. Path-scoped detail: `.claude/rules/`.
|
||||
|
||||
!!! IMPORTANT !!!
|
||||
- Always update CHANGELOG.md whenever you modified the code, and pushed to git!!
|
||||
- IF controller feature changed (new/modify/remove) always update the relevant part of controller/README.md with the architectural change!!
|
||||
## What this repo is
|
||||
|
||||
## Project overview
|
||||
|
||||
Felhom is a managed home-server business for Hungarian customers. This repo contains the
|
||||
**felhom-controller** — the Go application that manages Docker Compose stacks inside each customer
|
||||
LXC guest via a Hungarian-language web dashboard.
|
||||
|
||||
Read in this order:
|
||||
- **`REUSE.md`** — before writing new code (canonical helpers, patterns, traps, seams).
|
||||
- `CONTEXT.md` — current project state, decisions, roadmap (update after each session).
|
||||
- `controller/README.md` — full feature/architecture reference (update when features change).
|
||||
- `TASK.md` — the current task to implement (if it exists).
|
||||
|
||||
## System context — the three-component model
|
||||
|
||||
The project runs **on Proxmox**, with a locked three-component model:
|
||||
- **Hub** (`felhom.eu/hub/`) — operator backend on k3s.
|
||||
- **Host agent** (`felhom-agent/`) — one per Proxmox host; operator-tier; owns ALL Proxmox interaction.
|
||||
- **In-guest controller** (THIS repo) — one per customer LXC; **Docker-only; holds NO Proxmox
|
||||
credentials**. De-privileged: disk/host/Proxmox concerns are delegated to the host agent via the
|
||||
pinned local-API client (`internal/agentapi`); the controller keeps the app domain — stack/deploy
|
||||
management, the Hungarian web UI, app-data backup, metrics/telemetry, integrations, git-sync,
|
||||
notifications. Whole-guest backup (PBS vzdump) is the agent's.
|
||||
|
||||
> **Authoritative maps:** `felhom.eu/documentation/architecture/01/02/03-*.md` (topology/trust,
|
||||
> controller module map, host agent) + the code-verified feature docs in
|
||||
> `felhom.eu/documentation/controller/`. Match the current code, not summaries, if they drift.
|
||||
The **in-guest controller** — one per customer LXC, Docker-only, **holds NO Proxmox credentials**. It
|
||||
owns the app domain: stack/deploy management, the Hungarian web UI, app-data backup, metrics,
|
||||
integrations, git-sync, notifications. Disk/host/Proxmox concerns are delegated to the host agent via
|
||||
`internal/agentapi`. Whole-guest backup (PBS vzdump) is the agent's, not ours.
|
||||
|
||||
**Don't confuse the two ex-"controllers":** `felhom-agent` (host, operator-tier, was
|
||||
`proxmox-controller`) vs this `felhom-controller` (in-guest, was `deploy-felhom-compose`).
|
||||
`proxmox-controller`) vs this repo (in-guest, was `deploy-felhom-compose`).
|
||||
|
||||
## Layout (verified against the tree)
|
||||
## Doing X → read Y
|
||||
|
||||
```
|
||||
controller/cmd/controller/ entry point + startup wiring (scheduler block, init-only setters)
|
||||
controller/internal/
|
||||
agentapi/ pinned-TLS client to the host agent's per-guest local API (THE disk seam)
|
||||
api/ REST /api/* router (writeJSON envelope, limitBody, config writes)
|
||||
appbackup/ felhom-data paths/namespaces, DB dumps, userdata skeleton (shared primitives)
|
||||
appexport/ .fab export/import bundles (password crypto, strict segment validation)
|
||||
assets/ app logo/screenshot sync from the hub
|
||||
backup/ app-data backup manager, recovery units, tier-2 copies, offbox restic
|
||||
bootstrap/ bootstrap.json ingest → controller.yaml (Day-0 + refresh)
|
||||
channelhealth/ agent-channel health checker (debounce + born-down alerting)
|
||||
cloudflare/ geo-enforcement remnant (agent-delegated)
|
||||
config/ controller.yaml load/validate (LoadPermissive = setup-mode only)
|
||||
crypto/ AES-256-GCM app.yaml secret encryption (ENC: prefix)
|
||||
infra/ traefik/cloudflared/filebrowser base-stack templates
|
||||
integrations/ app-to-app integrations (e.g. OnlyOffice)
|
||||
mailrelay/ app-email SMTP shim → hub relay
|
||||
metrics/ telemetry collection
|
||||
monitor/ health checks, protected containers
|
||||
notify/ hub event push (typed Notify* wrappers)
|
||||
quiesce/ quiesce loop for whole-guest backup (marker + recover)
|
||||
recovery/ recovery-unit restore
|
||||
report/ hub report builder/pusher + pull-based config refresh
|
||||
scheduler/ background jobs (Every/Daily, Budapest DST-safe)
|
||||
selftest/ startup self-checks
|
||||
selfupdate/ controller image self-update via the agent swap
|
||||
settings/ settings.json persistence (registry, flags, corruption recovery)
|
||||
setup/ first-boot setup wizard (own CSRF)
|
||||
stacks/ compose ops: deploy/delete/migrate/state (THE app domain core)
|
||||
sync/ git-sync of the app catalog
|
||||
system/ mounts/probes (linux + permissive _other stubs)
|
||||
util/ small shared helpers
|
||||
web/ dashboard UI: server, auth/CSRF, handlers, funcmap, templates (Hungarian)
|
||||
```
|
||||
| Doing | Read |
|
||||
|---|---|
|
||||
| writing any new code | `REUSE.md` — canonical helpers, patterns, traps, seams |
|
||||
| needing current state / roadmap | `CONTEXT.md` |
|
||||
| needing a feature or architecture reference | `controller/README.md` |
|
||||
| build, deploy, publish, verify a version | the **`felhom-build-deploy`** skill |
|
||||
| writing or reviewing a test, fixing a bug | the **`felhom-testing`** skill |
|
||||
| UI, tokens, badges, Hungarian copy | the **`felhom-ui-design`** skill |
|
||||
| which box may I break | `felhom.eu/documentation/runbooks/target-selection.md` |
|
||||
| host addresses, break-glass, node facts | `felhom.eu/documentation/operations/nodes.md` |
|
||||
| what version is live anywhere | ask the hub (`/hosts`, `/configs`) or the box — **never a doc** |
|
||||
| the authoritative design | `felhom.eu/documentation/architecture/01/02/03-*.md` |
|
||||
|
||||
Per-package helpers/seams/traps: **`REUSE.md`** (maintained same-commit as helper changes).
|
||||
## Session-critical invariants
|
||||
|
||||
## Conventions & cardinal rules
|
||||
|
||||
- **Trunk-based — no branches.** All shippable work commits directly to `main`; `main` equals what is
|
||||
deployed. Report-only artifacts → `felhom.eu/documentation/` (`audits/`, `backlog/`). Risky fixes
|
||||
are implemented during the supervised session itself, on `main`; if a fix can't be verified/shipped,
|
||||
revert + report — never park on a branch.
|
||||
- Code quality: double-check for bugs/edge cases; add debug logging; **ask rather than guess**.
|
||||
- All UI text is Hungarian (Budapest timezone). Design tokens/gates: use the `felhom-ui-design`
|
||||
skill; templates must pass `controller/scripts/template_id_gate.py` + `emoji_gate.py`.
|
||||
- Testing doctrine (non-hollow tests, red-proofs, seams): use the `felhom-testing` skill.
|
||||
- Update `REUSE.md` if you added/changed/deprecated a shared helper or pattern (same commit).
|
||||
- **Coupled features** (controller behavior that depends on a specific agent version): add a
|
||||
`featureProbes` table row in `internal/agentapi/features.go` + a `Supports` gate call at the
|
||||
feature's entry point; declare `MinAgent: X.Y.Z` in the CHANGELOG entry header. Rules:
|
||||
`felhom.eu/documentation/runbooks/publish-train-rules.md`.
|
||||
|
||||
> **In every repository where you make a change, update both files in that repo:**
|
||||
> - **`CHANGELOG.md`** — cumulative log, newest on top.
|
||||
> - **`REPORT.md`** — **overwrite** with the most recent implementation/validation summary only.
|
||||
>
|
||||
> **Never write secrets** into any committed file — reference them as "stored out-of-band".
|
||||
|
||||
## Live validation
|
||||
|
||||
Exercise the SERVER-SIDE PIPELINE a real user triggers, end-to-end (connect → enroll → deploy). The
|
||||
forbidden shortcut is BYPASSING that pipeline (the F9 episode: raw agent guest-attach + hand-set
|
||||
state). Invoking the exact endpoint the UI invokes is an acceptable proxy when a browser tool isn't
|
||||
available — no server logic is skipped, only rendering; say which method was used. For strict
|
||||
end-to-end UI coverage use claude-in-chrome (attaches only to sessions started AFTER the bridge
|
||||
connected) or a manual click-through.
|
||||
|
||||
## Environment & access
|
||||
|
||||
Claude Code runs on Windows 11; repos in `E:\git\` (`/e/git/` in Git Bash). All repos hosted at
|
||||
`gitea.dooplex.hu/admin/`. **SSH binary MUST be** `SSH=/c/Windows/System32/OpenSSH/ssh.exe`
|
||||
(Git Bash's ssh lacks the Windows agent — fails silently).
|
||||
|
||||
| Host | Access | Role |
|
||||
|------|--------|------|
|
||||
| Build server (k3s) | `$SSH kisfenyo@192.168.0.180` | build + push images (`~/build/felhom-controller`) |
|
||||
| Demo Proxmox host `demo-felhom` | `$SSH felhom-pve` (root@192.168.0.162) | `pct` into guests; live validation |
|
||||
| Demo guest 9201 | `pct exec 9201 -- ...` on felhom-pve | the live demo controller (golden/bootstrap-managed) |
|
||||
| felhotest (legacy) | `$SSH -p 33022 kisfenyo@router.abonet.hu` | OLD /opt/docker compose mechanism |
|
||||
|
||||
External access via Cloudflare Tunnel → Traefik; Pi-hole forwards `*.demo-felhom.eu` → .162 locally.
|
||||
|
||||
## Build & deploy — MANDATORY after code changes
|
||||
|
||||
**Full runbook: use the `felhom-build-deploy` skill.** Summary (guest 9201 is bootstrap-managed —
|
||||
**no compose file**; `felhom-controller-bootstrap.service` runs the tag in `/etc/felhom-controller-image`):
|
||||
|
||||
| Step | Command |
|
||||
|------|---------|
|
||||
| 1. Commit + push | `git add -A && git commit -m "..." && git push` |
|
||||
| 2. Build + push image | `$SSH kisfenyo@192.168.0.180 "cd ~/build/felhom-controller && git -C ~/git/felhom-controller pull && ./build.sh <VER> --push"` (build.sh does NOT pull — the explicit pull is load-bearing) |
|
||||
| 3. Deploy (9201) | `$SSH felhom-pve "pct exec 9201 -- bash -c 'docker pull gitea.dooplex.hu/admin/felhom-controller:<VER> && echo gitea.dooplex.hu/admin/felhom-controller:<VER> > /etc/felhom-controller-image && systemctl restart felhom-controller-bootstrap.service'"` |
|
||||
| 4. Verify | `$SSH felhom-pve "pct exec 9201 -- docker ps --filter name=felhom-controller --format '{{.Image}} {{.Status}}'"` + container logs |
|
||||
|
||||
Hub build/deploy lives in `felhom.eu` (GitOps) — see that repo's CLAUDE.md / the skill. Catalog
|
||||
changes (`app-catalog-felhom.eu`): commit+push; controller sync picks them up ≤15 min or via the
|
||||
"Sablonok frissítése" button.
|
||||
|
||||
## Session-critical invariants (the rest live in REUSE.md)
|
||||
The rest live in `REUSE.md`. These cost incidents to learn:
|
||||
|
||||
- `docker compose restart` does NOT pick up new images/env — always `up -d` (`RedeployFromEnv`).
|
||||
- Docker's `.State` says "running" even for unhealthy containers — `.Status` parse is the truth.
|
||||
- In-memory `Deployed` flag is set BEFORE `compose up -d` (slow-pull race); reverted on failure.
|
||||
- `compose up -d` exits 0 on crash-loops — post-start status check is the detection.
|
||||
- Docker's `.State` says "running" even for unhealthy containers — the `.Status` parse is the truth.
|
||||
- In-memory `Deployed` is set BEFORE `compose up -d` (slow-pull race); reverted on failure.
|
||||
- `compose up -d` exits 0 on crash-loops — the post-start status check is the detection.
|
||||
- Env var KEYS are logged, never values. Protected stacks (traefik, cloudflared, felhom-controller)
|
||||
can't be stopped from the UI.
|
||||
cannot be stopped from the UI.
|
||||
- Verify a container image HAS the healthcheck tool before using it (BusyBox wget / python3 / curl —
|
||||
catalog REUSE.md maps the families).
|
||||
the catalog `REUSE.md` maps the families).
|
||||
- `IsRunning()` is CONCURRENCY, false during a verification restore — display MUST use
|
||||
`RestoreStatus()`.
|
||||
|
||||
## Live validation — the fence
|
||||
|
||||
Exercise the SERVER-SIDE PIPELINE a real user triggers, end-to-end (connect → enroll → deploy). **The
|
||||
forbidden shortcut is BYPASSING that pipeline** — the F9 episode was a raw agent guest-attach with
|
||||
hand-set state, and it proved nothing.
|
||||
|
||||
`claude-in-chrome` is NOT available on DooPlex. The standard method is endpoint-level: invoke the
|
||||
exact endpoint the UI invokes (no server logic is skipped, only rendering) and **say which method was
|
||||
used**. Strict end-to-end UI coverage is a manual click-through by the operator.
|
||||
|
||||
Two traps in that method live in `.claude/rules/ui-hungarian.md` (ASCII-only greps; `!` in
|
||||
credentials) — they load when you touch a template or stylesheet.
|
||||
|
||||
## Commands — one per surface
|
||||
|
||||
| Surface | Command |
|
||||
|---|---|
|
||||
| Gates (after ANY change) | `python3 controller/scripts/controller_gates.py` — from `controller/` |
|
||||
| Green gate | `go build ./... && go vet ./... && go test ./...` |
|
||||
| Build + deploy | the **`felhom-build-deploy`** skill — do not hand-roll it |
|
||||
|
||||
Guest 9201 is **bootstrap-managed — there is no compose file**;
|
||||
`felhom-controller-bootstrap.service` runs the tag written in `/etc/felhom-controller-image`. Catalog
|
||||
changes (`app-catalog-felhom.eu`) are picked up by controller sync ≤15 min, or via the "Sablonok
|
||||
frissítése" button.
|
||||
|
||||
## Working with CHANGELOG.md
|
||||
|
||||
**DO NOT read the full file** — it is large and will waste context.
|
||||
- Session start: use `CONTEXT.md` + `controller/README.md` for current state.
|
||||
|
||||
- Session start: `CONTEXT.md` + `controller/README.md` for current state.
|
||||
- Adding an entry: Read only the top ~30 lines for format, then Edit-insert after line 1.
|
||||
- History: Grep for topics instead of reading.
|
||||
|
||||
## End-of-session checklist
|
||||
|
||||
1. **Commit and push** all code changes
|
||||
2. **Build, push, and deploy** the new controller image (if controller code changed)
|
||||
3. **Update CHANGELOG.md** with what was done
|
||||
4. **Update CONTEXT.md** with decisions made, state and what's next
|
||||
5. **Update controller/README.md** if architecture or features changed
|
||||
6. **Verify** the deployment is working (check `docker ps` and logs)
|
||||
7. **Update REUSE.md** if you added/changed/deprecated a shared helper or pattern (same commit)
|
||||
1. **Commit and push** all code changes (explicit paths; no `git add -A`).
|
||||
2. **Build, push, and deploy** the new controller image, if controller code changed.
|
||||
3. **`CHANGELOG.md`** — always, whenever code changed and was pushed.
|
||||
4. **`CONTEXT.md`** — decisions made, state, what is next.
|
||||
5. **`controller/README.md`** — whenever a feature was added, modified or removed.
|
||||
6. **`REPORT.md`** — overwrite with this run's summary only.
|
||||
7. **`REUSE.md`** — if a shared helper or pattern was added/changed/deprecated (same commit).
|
||||
8. **Verify** the deployment (`docker ps` + logs).
|
||||
|
||||
<!--
|
||||
WHY THIS FILE IS SHORT (2026-08-06, instruction-trim task).
|
||||
Removed from here and rehomed, not lost:
|
||||
- the `## Layout (verified against the tree)` block -> derivable by `ls internal/`; REUSE.md
|
||||
carries the per-package seams and traps that the annotations were really for.
|
||||
- the `!!! IMPORTANT !!!` header -> its two requirements are checklist items 3 and 5. One voice,
|
||||
one place; a rule stated twice in one file is a rule that gets edited in one of them.
|
||||
- the host/access table -> documentation/operations/nodes.md is the single home. The copy here
|
||||
had drifted: it gave demo-felhom as plain root@192.168.0.162 (the LAN fallback, not the route),
|
||||
pinned "agent 0.93.0" against the project's own no-versions-in-docs rule, and claimed no drill
|
||||
VM was provisioned on demo-hp. Measured 2026-08-06: `qm list` on demo-hp shows VM 300
|
||||
`drill-r50` present. felhom-agent/CLAUDE.md was right; this file was wrong.
|
||||
- the "felhom-pve is back on the home LAN" block -> it was bookkeeping about a retired block; the
|
||||
record is in documentation/audits/AUDIT-vacation-remote-ops-2026-07-20.md.
|
||||
- the "Legacy: Windows workstation" block -> the workspace-root CLAUDE.md carries the full version.
|
||||
- the gates/logging/coupling/UI paragraphs -> .claude/rules/*.md, which load when a matching file
|
||||
is read instead of in every session.
|
||||
Full per-block accounting: felhom.eu/documentation/audits/LEDGER-instruction-trim-2026-08-06.md
|
||||
-->
|
||||
|
||||
+1594
-1
File diff suppressed because it is too large
Load Diff
@@ -1,123 +1,441 @@
|
||||
# REPORT — v0.113.0: NAS verify-before-commit + page redesign + protocol-honest guidance
|
||||
# REPORT — controller v0.215.0 → v0.216.0: disk-health severity ladder, escalation, and the alert that never sent
|
||||
|
||||
**Date:** 2026-07-11 · **Version:** controller v0.113.0 (from v0.112.0) · **Pairs with:** agent
|
||||
v0.81.0 + felhom.eu host-install v1.13.0
|
||||
**Spec:** TASK — NAS verify-before-commit + page redesign + protocol-honest guidance
|
||||
**Evidence base:** `felhom.eu/documentation/audits/SPIKE-nas-verify-2026-07-11.md` (b57f6c1)
|
||||
**Date:** 2026-08-14 · **Task class:** Implementation · **Repos touched:** `felhom-controller` (code),
|
||||
`felhom.eu` (documentation only — no hub code, no manifest bump, no ArgoCD sync)
|
||||
|
||||
## Confirmed baselines → shipped
|
||||
---
|
||||
|
||||
| Repo | Baseline (`main`) | Shipped commits | Version |
|
||||
|---|---|---|---|
|
||||
| felhom-agent | `300f06722b` (v0.80.0) | `added9d` | **v0.81.0** — deployed on felhom-pve |
|
||||
| felhom-controller | `3db9126121` (v0.112.0) | `bb8737a` (orchestration) + `a65dcff` (UI) + this docs commit | **v0.113.0** — live on 9201 |
|
||||
| felhom.eu | `e80e14d6` | `27e2fb0` | host-install **v1.13.0** + feature doc |
|
||||
## 1. Confirmed baselines used (as read at the start of the run)
|
||||
|
||||
## What shipped (files)
|
||||
| Repo | `main` @ start | Version | → Shipped |
|
||||
|------|----------------|---------|-----------|
|
||||
| felhom-controller | `3e3ee94b7bbe6b66663c468e22aa86616365a45a` | v0.214.0 | **v0.215.0**, then **v0.216.0** (a defect found live in v0.215.0 — §14) |
|
||||
| felhom.eu | `e0b56c976f8e4a7352309d78754fec448dd55f99` | n/a (docs only) | n/a |
|
||||
|
||||
**Agent (`added9d`)** — `internal/storage/netmount.go` (NFS `retry=0` + exported
|
||||
`NetworkMountedAt`/`NetworkEndpointReachable`), NEW `internal/storage/netverify.go`
|
||||
(`ClassifyNetVerifyFailure`, Q4-verbatim table), NEW `internal/localapi/netverifyjob.go` (in-memory
|
||||
single-slot detached verify + auto-rollback + `GET /netstorage/verify-status`),
|
||||
`internal/localapi/netstorage.go` (sync fast-fail: full validation + 2 s TCP pre-probe before ANY
|
||||
install; verify-job start), `server.go` (seams + route). **No new sudoers grants** — journal read is
|
||||
unprivileged via the `systemd-journal` group.
|
||||
Both trees verified clean (`git status --porcelain` empty, `HEAD == origin/main`) before any build.
|
||||
The controller hash matched the spec's stated baseline exactly. `MinAgent` stays **0.129.0** — no
|
||||
agent change; every field read here has been on the wire since agent v0.94.0/v0.95.0.
|
||||
|
||||
**Controller (`bb8737a` + `a65dcff`)** — `internal/agentapi/client.go` (verify fields, typed
|
||||
`NetAddRefusedError`, `NetVerifyStatus`), NEW `internal/web/netprobe.go` + `netprobe_linux.go` +
|
||||
`netprobe_other.go` (uid-1000 re-exec probe; `--netprobe` hidden mode in `cmd/controller/main.go`),
|
||||
NEW `internal/web/netstorage_job.go` (detached single-flight orchestration + §3.2 Hungarian map),
|
||||
`netstorage_handlers.go` (job-start add + status endpoint + orphan rows + `netListFn` seam),
|
||||
`storage_handlers.go` (route), `templates/storage_network.html` (full redesign).
|
||||
---
|
||||
|
||||
**felhom.eu (`27e2fb0`)** — `scripts/felhom-host-install.sh` v1.13.0 (`usermod -aG systemd-journal
|
||||
felhom-agent`, idempotent; header/const drift v1.11.0-vs-1.12.0 fixed), NEW
|
||||
`documentation/controller/network-storage-nas.md` (authoritative feature doc).
|
||||
## 2. Files created / modified
|
||||
|
||||
## Tests + companion red-proofs (all: mutation run → FAIL observed → reverted → suite green)
|
||||
**felhom-controller**
|
||||
- `controller/internal/agentapi/diskverdict.go` — modified (14-row ladder, `DiskPrior`, `UncorrectableSectors`, `TemperatureFailC`)
|
||||
- `controller/internal/agentapi/diskverdict_test.go` — modified
|
||||
- `controller/internal/agentapi/diskverdict_ladder_test.go` — **created**
|
||||
- `controller/internal/notify/notifier.go` — modified (severity, `DiskAlert`, `DiskAlertKind`, `Severity()`, 5 message shapes)
|
||||
- `controller/internal/notify/disk_health_test.go` — rewritten
|
||||
- `controller/internal/web/disk_health_state.go` — **created** (persistence + decision)
|
||||
- `controller/internal/web/disk_health.go` — modified
|
||||
- `controller/internal/web/disk_health_test.go` — rewritten
|
||||
- `controller/internal/web/server.go` — modified (seam signature)
|
||||
- `controller/cmd/controller/main.go` — modified (6h → 1h)
|
||||
- *(v0.216.0)* `controller/internal/web/disk_health.go` + `disk_health_test.go` — the R-335 dedup fix and its test
|
||||
- `CHANGELOG.md`, `CONTEXT.md`, `REUSE.md`, `controller/README.md`, `REPORT.md`
|
||||
|
||||
Test funcs added: agent storage +3 (netverify_test.go incl. the pure `networkMountedIn` table),
|
||||
agent localapi +5 (netverifyjob_test.go), controller web +9 (netstorage_job_test.go +
|
||||
storage_network_template_test.go).
|
||||
**felhom.eu** (documentation only)
|
||||
- `documentation/audits/DIAG-smart-passed-trap-2026-08-14.md` — **created**
|
||||
- `documentation/audits/fixtures/smart-ST3000VX010-failing-2026-08-14.json` — **created** (raw `smartctl -a -j`, verbatim)
|
||||
- `documentation/audits/fixtures/smartd-history-sdg-2026-08-14.txt` — **created** (406 `smartd` journal lines)
|
||||
- `documentation/architecture/00-capability-map.md`, `documentation/backlog/ROADMAP.md`, `documentation/backlog/OPEN-ITEMS.md` — modified
|
||||
|
||||
| # | Test | Red-proof mutant | Observed failure (verbatim core) |
|
||||
|---|---|---|---|
|
||||
| A1 | `TestMountOptions_NFSRetry0_SMBWithout` (+ NFS rendering test) | reverted `retry=0` | `NFS options missing retry=0 (Q4-vi): "vers=4.1,...,_netdev"` |
|
||||
| A2 | `TestClassifyNetVerifyFailure` (Q4-verbatim table incl. merged `nfs_export` + empty-journal degradation) | exit-code classifier (`return mount_failed` — everything is rc=32) | every non-generic row: `= "mount_failed", want "unreachable"/...` — an exit-code mutant cannot split smb_auth/smb_share |
|
||||
| A3 | `TestNetVerify_MountFailed_RollsBackAndClassifies` (+ journal-unavailable sibling) | rollback call dropped | `RemoveNetworkMount not called for the failed install: removed=[]` + `creds file must be removed` |
|
||||
| A4 | `TestNetVerify_TruthTable` (§8: ReadDir-ok+unmounted=FAIL; EACCES+mounted=OK) | readability-based verdict (`terr == nil`) | BOTH rows failed: `a readable-but-unmounted path must FAIL verify, got done` / `unreadable-but-mounted must PASS, got failed` |
|
||||
| A5 | `TestNetVerify_UnreachablePreProbe_InstallsNothing` | pre-probe skipped | `unreachable add: got 200 want 502` (the install would have run) |
|
||||
| A6 | `TestNetVerify_SingleFlight_AndNoJobShape` | running-check dropped | `second add while verifying: got 200 want 409` |
|
||||
| C1 | `TestNetAdd_HappyPath_RegisterOnlyAfterProbe` | register moved BEFORE the probe | `phase = failed (category=register_failed detail=... already registered)` + C2's `must NOT be registered (got 1 paths)` |
|
||||
| C2 | `TestNetAdd_ProbeFail_RollsBackNotRegistered` | rollback dropped | `probe-fail must roll the agent install back: removes=[]` |
|
||||
| C3 | `TestNetAdd_AgentVerifyFailed_MappedMessage` | (kill-surface shared with C4's mutant) | asserts the EXACT §3.2 merged nfs_export string + probe-not-run + no controller double-remove |
|
||||
| C4 | `TestNetAdd_VerifyLost_RollsBack` | `none` treated as success | `probe must not run when the verify was lost` |
|
||||
| C5 | `TestNetProbeChild` (ok / unwritable→2→not_writable / nonce-tamper→3→probe_io / cleanup-fail→4=OK+warn) | — (pure child body + verdict table; Credential wiring live-validated in Scenario C) | — |
|
||||
| C6 | `TestNetAdd_SingleFlight` | acquire check dropped | `second add: got 200 want 409` |
|
||||
| C7 | `TestNetStorage_OrphanRow` | orphan sweep disabled (registry-only filter) | `items = 1, want 2 (registered + orphan)` |
|
||||
| C8 | `TestStorageNetworkTemplate_CanonicalClasses` (renders through the PRODUCTION template tree) | `form-input` reintroduced | `rendered page still contains the banned pattern "form-input"` |
|
||||
---
|
||||
|
||||
Green gates: agent `go build && go vet && go test ./...` PASS (one hit of the KNOWN
|
||||
`TestGenerateRecoveryCode` wordlist flake — clean on re-run, pre-documented); controller full suite
|
||||
PASS; `template_id_gate.py` + `emoji_gate.py` green.
|
||||
## 3. Commits pushed to `main`
|
||||
|
||||
## Deployed + verified
|
||||
| Repo | Hash | What |
|
||||
|------|------|------|
|
||||
| felhom.eu | `848de81` | Part 0 — fixtures + findings doc |
|
||||
| felhom-controller | `bb50e12` | Parts 1–3 — ladder, severity, persisted state + tests |
|
||||
| felhom-controller | `c24f192` | Group L strengthened to two post-restart checks |
|
||||
| felhom-controller | `34d83f5` | Part 4 — cadence 6h → 1h (measured) |
|
||||
| felhom-controller | `8144a70` | Part 5 — CHANGELOG / CONTEXT / README / REUSE |
|
||||
| felhom.eu | `767960b` | Part 5 — capability map, ROADMAP, register rows R-328…R-334 |
|
||||
| felhom-controller | `90f2545` | **v0.216.0** — R-335, one physical disk evaluated once per run |
|
||||
| felhom.eu | `fa4748d` | R-335 register row |
|
||||
|
||||
- **Agent v0.81.0** on felhom-pve: `.bak-0.80.0` kept; `usermod -aG systemd-journal felhom-agent`
|
||||
applied BEFORE restart; `felhom-agent --version` → 0.81.0; `id felhom-agent` →
|
||||
`groups=990(felhom-agent),999(systemd-journal)`; journal shows a clean start (enrolled drive
|
||||
bound under shared parent, local-api listening, hub desired-state gen 10, no capability
|
||||
degradation).
|
||||
- **Controller v0.113.0** on 9201 (golden/bootstrap mechanism): `docker ps` →
|
||||
`gitea.dooplex.hu/admin/felhom-controller:0.113.0 Up (healthy)`.
|
||||
---
|
||||
|
||||
## Live validation (anti-F9: the exact endpoint the UI invokes)
|
||||
## 4. Per-test results — all twelve groups
|
||||
|
||||
Method: `curl` from inside guest 9201 to the controller container
|
||||
(`http://172.17.0.2:8080` + `Host: felhom.demo-felhom.eu` — the dashboard is host-routed; the demo
|
||||
has no password so auth/CSRF do not apply) — POST `/api/storage/netstorage/add` + status-poll, the
|
||||
byte-identical server pipeline behind the UI; the residual is client-side rendering only.
|
||||
Sim NAS = isolated `/srv/nas-spike3/*` on DooPlex (`.bak-nasspike3` backups; pre-counts exports=2,
|
||||
smb_sections=6; NO iptables — the unreachable case used the ping+neigh-verified-unused
|
||||
192.168.0.199). Throwaway users `spike3b` (uid 1060) + `spike3smb`; passwords never committed.
|
||||
| Group | Scenario | Test | Result |
|
||||
|-------|----------|------|--------|
|
||||
| A | real drive, 2nd observation | `TestDiskCheck_RealDrive_HibaAndOneCriticalEvent` + `TestLadder_RealDrive_ReachesHiba` | **PASS** |
|
||||
| B | the transient that cleared | `TestDiskCheck_FirstSightingIsWarnOnly` | **PASS** |
|
||||
| C | sustained → Hiba | `TestDiskLadder_SustainDrivesTheEscalation` + `TestLadder_SustainIsWhatFires` | **PASS** |
|
||||
| D | recovered, silent | `TestDiskCheck_RecoveryIsSilentAndClearsState` | **PASS** |
|
||||
| E | flap damping | `TestDiskCheck_FlapDamping` | **PASS** |
|
||||
| F | escalation beats damping | `TestDiskCheck_EscalationBeatsDamping` | **PASS** |
|
||||
| G | still getting worse | `TestDiskCheck_RealertWhenStillWorsening` | **PASS** |
|
||||
| H | below both bars | `TestDiskCheck_NoRealertBelowBothBars` | **PASS** |
|
||||
| I | heat | `TestLadder_Temperature` + `TestDiskCheck_TemperatureShape` | **PASS** |
|
||||
| J | no data never alarms | `TestDiskCheck_UnknownNeverAlarmsNorErasesPrior` + `TestLadder_UnknownNeverAlarms` | **PASS** |
|
||||
| K | severity routes | `TestNotifyDiskHealthDegraded_SeverityRoutes` | **PASS** |
|
||||
| L | state survives restart (seam) | `TestDiskCheck_StateSurvivesRestart_ProductionPath` | **PASS** |
|
||||
|
||||
| Scenario | Result | Key evidence |
|
||||
|---|---|---|
|
||||
| **A** bogus export on a reachable server (THE bug) | **PASS** — failed in **4 s**, category `nfs_export`, the exact merged Hungarian message, the REAL journal in detail (`reason given by server: No such file or directory` — the unprivileged journal read works), list EMPTY, **zero spike3 units left on felhom-pve** — the "Készenlét forever" behavior is dead |
|
||||
| **B** SMB wrong password / wrong share | **PASS** — `smb_auth` („Hibás SMB felhasználónév vagy jelszó.") vs `smb_share` (distinct message), both 4 s, **creds file gone after each**, no units |
|
||||
| **C** squash trap (`anonuid=1000`, NO all_squash — mounts fine) | **PASS** — agent verify passed → **uid-1000 probe refused** → `not_writable` with the Route-A message incl. the computed **101000**; full rollback (no units/mounts/registration); **no probe file** left on the export |
|
||||
| **D** happy paths | **PASS** — d1 (NFS anonuid=101000), d2 (NFS **Route A**, alien uid 1060), d3 (SMB plain user, no force user): all `done` in **4 s**, registered Schedulable, health `ok`. d2 production proof: the installed unit carries `retry=0`; a uid-1000 write shows guest-view `65534:65534` and lands **server-side `1060:1060`** |
|
||||
| **E** unreachable IP | **PASS** — failed in **2.0 s** (the agent's sync TCP pre-probe), category `unreachable`, message names the IP, **nothing installed** |
|
||||
| **G** single-flight | unit-proven (A6 + C6); not re-proven live — the 4 s happy windows make a live race impractical |
|
||||
Supporting: `TestLadder_CountBackstopBoundary`, `TestLadder_ZeroPriorIsFailSafe`,
|
||||
`TestUncorrectableSectors`, `TestDegradedAttributes_NamesFailCounters`,
|
||||
`TestNotifyDiskHealthDegraded_{WarnShape,FailShapes,CopyDiscipline}`,
|
||||
`TestDiskAlertDecision_Table`, `TestDiskState_CorruptFileFallsBackToNoPrior`,
|
||||
`TestDiskCheck_DisappearedDiskIsForgotten`, `TestDiskCheck_UnreachableAgentIsInert` — all PASS.
|
||||
|
||||
**Teardown verified:** d1/d2/d3 removed via the real remove endpoint (list empty); 180 restored from
|
||||
`.bak-nasspike3` (post-counts exports=2, smb_sections=6, 0 spike3 exports, both users deleted,
|
||||
scratch gone); felhom-pve: 0 spike3 mounts/units/creds, mountpoint dirs removed,
|
||||
`systemctl reset-failed` cleared the residual failed-unit listings; guest 9201: 0 spike3 mounts,
|
||||
controller healthy.
|
||||
---
|
||||
|
||||
## NOT yet live-validated
|
||||
## 5. Red-proof outcomes — all twelve, individually
|
||||
|
||||
- Guest-restart trigger survival (Q1c follow-up — needs a restart window; no guest restart allowed).
|
||||
- Synology/QNAP appliance pass (virtual-dsm) — the sim-vs-real caveat stands.
|
||||
- Peti rollout: floor bump + his `usermod -aG systemd-journal felhom-agent` step — when the NAS
|
||||
feature reaches him. STOP honored: no publish train (agent 0.81.0 NOT published to Gitea, Day-0
|
||||
manifest untouched), Peti's box untouched, guest 9201 never restarted.
|
||||
- Scenario F live (agent restart mid-verify) — covered by unit tests A6/C4 + the orphan surface
|
||||
(C7); a live kill inside a 4 s verify window is impractical.
|
||||
Each mutation was applied by script, **asserted present in the source before the run** (the harness
|
||||
aborts with `MUTATION-NOT-APPLIED` if the target text is absent), the named test run, and the file
|
||||
reverted with `git checkout --`. The tree was confirmed clean after the sweep.
|
||||
|
||||
## Observations (documented, NOT acted on)
|
||||
| # | Mutation applied | Target test | Outcome |
|
||||
|---|------------------|-------------|---------|
|
||||
| A | remove truth-table row 6 (the sustain rule) | `TestDiskCheck_RealDrive_HibaAndOneCriticalEvent` | **RED-PROOF PASSED — FINDING, see below** |
|
||||
| B | make row 9 return `Fail` | `TestDiskCheck_FirstSightingIsWarnOnly` | failed as required |
|
||||
| C | pass a zero `DiskPrior` in `RunDiskHealthCheck` | `TestDiskLadder_SustainDrivesTheEscalation` | failed as required |
|
||||
| D | let Rendben fall through the silence guard | `TestDiskCheck_RecoveryIsSilentAndClearsState` | failed as required |
|
||||
| E | compare against last **observed** verdict, not last **alerted** | `TestDiskCheck_FlapDamping` | failed as required |
|
||||
| F | let damping cover escalations | `TestDiskCheck_EscalationBeatsDamping` | failed as required |
|
||||
| G | remove the re-alert branch | `TestDiskCheck_RealertWhenStillWorsening` | failed as required |
|
||||
| H | make cooldown/doubling an **OR** instead of an AND | `TestDiskCheck_NoRealertBelowBothBars` | failed as required |
|
||||
| I | remove truth-table rows 3 **and** 13 | `TestLadder_Temperature`, `TestDiskCheck_TemperatureShape` | failed as required |
|
||||
| J | let UNKNOWN delete the prior record | `TestDiskCheck_UnknownNeverAlarmsNorErasesPrior` | failed as required |
|
||||
| K | restore `severity := "warn"` | `TestNotifyDiskHealthDegraded_SeverityRoutes` | failed as required |
|
||||
| L | skip loading the persisted state | `TestDiskCheck_StateSurvivesRestart_ProductionPath` | failed as required |
|
||||
|
||||
- The demo host had NO pre-existing `nas-media` unit (the spike-era shares were fully torn down) —
|
||||
nothing predating `retry=0` exists; the "don't rewrite installed units" rule was moot in practice.
|
||||
- `RemoveNetworkMount` leaves the EMPTY mountpoint dir under `/mnt/felhom-drives/` and does not
|
||||
`reset-failed` a failed mount unit's residual state (the unit FILES are removed; a `not-found
|
||||
failed` listing lingers until reset-failed/reboot). Cosmetic; cleaned by hand this run.
|
||||
- Pre-existing (since 2026-07-08, unrelated to this task): `lanresolver: cannot list provisioned
|
||||
guests: permission denied` — `/var/lib/felhom-agent/guests` is root-0700 while the agent runs
|
||||
non-root.
|
||||
- host-install header/SCRIPT_VERSION drift (v1.11.0 vs 1.12.0): the v1.12.0 bump had shipped with
|
||||
no changelog entry and no header sync; both fixed at v1.13.0.
|
||||
- The controller container publishes NO ports (bridge-only, traefik-fronted): in-guest API testing
|
||||
needs the container IP + `Host:` header — the "POST to in-guest 127.0.0.1:8080" note in older
|
||||
session memory is stale.
|
||||
### A thirteenth red-proof, added after the deploy (R-335)
|
||||
|
||||
| # | Mutation applied | Target test | Outcome |
|
||||
|---|------------------|-------------|---------|
|
||||
| M | delete the `if seen[key] { continue }` dedup guard | `TestDiskCheck_SameDiskTwiceIsEvaluatedOnce` | failed as required |
|
||||
|
||||
Observed failure: `first sighting of an aliased disk must be silent, got 1: [{Label:felhom-backup … Kind:2 Sectors:8}]`
|
||||
— i.e. `Kind:2` is `DiskAlertFailSectors`, a **Hiba on a first sighting of 8 sectors**. Reverted.
|
||||
|
||||
### FINDING — red-proof A passed, and it is the spec's mutation that is at fault, not the code
|
||||
|
||||
The task specified group A's red-proof as *"remove truth-table row 6 → verdict is Warn"*. **That
|
||||
mutation cannot fail a test built on the real drive's values**: the real drive carries **352**
|
||||
unreadable sectors, so with row 6 deleted it still reaches Hiba via **row 8** (count ≥ 64). The test
|
||||
correctly stayed green, so the mutation proves nothing about row 6.
|
||||
|
||||
This was anticipated while writing the tests and is documented in the test's own comment rather than
|
||||
discovered afterwards. **Row 6 is genuinely pinned**, by two tests that hold the counters at **8**
|
||||
(far below the 64 backstop) and vary *only* the prior:
|
||||
|
||||
- `TestLadder_SustainIsWhatFires` (agentapi) — same `SmartSummary`, `DiskPrior{}` → Warn,
|
||||
`DiskPrior{SawUncorrectable:true}` → Fail.
|
||||
- `TestDiskLadder_SustainDrivesTheEscalation` (web) — the event-level twin.
|
||||
|
||||
Both were run under the row-6-deleted mutation and **both failed**, as recorded:
|
||||
`SAME 8 sectors, now sustained = 2 (Figyelmeztetés), want Fail/Hiba` and
|
||||
`severity = "warning", want critical` / `chip = "Figyelmeztetés", want Hiba`. So the invariant is
|
||||
covered; only the spec's chosen mutation was invalid.
|
||||
|
||||
### A second finding, from building red-proof L
|
||||
|
||||
The first version of the Group L seam test ran **one** check after the restart and **passed under the
|
||||
mutation** — because a controller that has forgotten its state is also silent on its first check. The
|
||||
test was strengthened to run **two** checks (commit `c24f192`), after which the mutation fails. This
|
||||
is the exact shape §10 warns about, caught by running the red-proof rather than assuming it.
|
||||
|
||||
---
|
||||
|
||||
## 6. Test count and suite state
|
||||
|
||||
- **Before:** 1391 test functions (at `3e3ee94`) · **After:** 1414 (+23)
|
||||
- `go build ./... && go vet ./... && go test ./...` — **all green**, no failures, no skips introduced.
|
||||
- `python3 controller/scripts/controller_gates.py` — **all 11 gates OK.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Cadence measurement (Part 4)
|
||||
|
||||
Measured on **demo-hp** (Tier 0, disposable), through `fetchDisks`' real path — the agent local API
|
||||
`GET /disks`, not the 60 s card cache. Ten consecutive calls, all **HTTP 200**:
|
||||
|
||||
```
|
||||
0.840558 0.817000 0.815783 0.831992 0.809066
|
||||
0.832899 0.824788 0.804886 0.812539 0.833547 (seconds)
|
||||
```
|
||||
|
||||
| min | median | max | disk count |
|
||||
|-----|--------|-----|------------|
|
||||
| **0.804886 s** | **0.820894 s** | **0.840558 s** | **3 physical rows** across 2 devices (SanDisk X600 M.2 SATA SSD; Toshiba KXG50PNV1T02 NVMe, counted twice as `c11-scratch` + `felhom-backup`) |
|
||||
|
||||
**Branch taken: median < 5 s → `6*time.Hour` → `1*time.Hour`.** The median is ~6× under the bar. The
|
||||
detection argument is the real one: the observed benign excursion lasted about **one hour**, so a
|
||||
6-hourly sampler can land either side of it and then catch the terminal run half a day late.
|
||||
|
||||
No spin-up signature appeared in the timings (uniform ~0.82 s; demo-hp is all-flash), so the
|
||||
measurement did not suggest the spun-down-drive concern. That question is recorded as an Observation
|
||||
below and deliberately **not acted on**.
|
||||
|
||||
---
|
||||
|
||||
## 8. Live validation
|
||||
|
||||
Deployed to **demo-hp guest 9201** via the bootstrap path (`docker pull` →
|
||||
`/etc/felhom-controller-image` → `systemctl restart felhom-controller-bootstrap.service`).
|
||||
|
||||
```
|
||||
gitea.dooplex.hu/admin/felhom-controller:0.215.0 Up 19 seconds (healthy) # 06:23Z
|
||||
gitea.dooplex.hu/admin/felhom-controller:0.216.0 Up 6 seconds (healthy) # after the R-335 fix
|
||||
```
|
||||
|
||||
### Leg 1 — no over-correction (the load-bearing check)
|
||||
|
||||
Method: **endpoint-level** — authenticated `GET /dashboard` on the real controller
|
||||
(`https://felhom.enkisfelhom.hu/dashboard`, HTTP 200, 44 036 bytes), i.e. the exact endpoint the UI
|
||||
invokes; only rendering is skipped. No browser is available on DooPlex.
|
||||
|
||||
Card contents, parsed from the response body:
|
||||
|
||||
| Disk | Chip | Class | Temp |
|
||||
|------|------|-------|------|
|
||||
| KXG50PNV1T02 NVMe TOSHIBA 1024GB | **Rendben** | `state-text-run` | 53 °C |
|
||||
| KXG50PNV1T02 NVMe TOSHIBA 1024GB | **Rendben** | `state-text-run` | 53 °C |
|
||||
| SanDisk X600 M.2 2280 SATA 128GB | **Rendben** | `state-text-run` | 44 °C |
|
||||
|
||||
`Figyelmeztetés` = 0, `Hiba` = 0, `Nincs adat` = 0, `state-text-warn` = 0, `state-text-crit` = 0.
|
||||
**No healthy disk was over-corrected.**
|
||||
|
||||
**Positive observable, at deploy:**
|
||||
`[INFO] [scheduler] Registered periodic job: disk-health-check (every 1h0m0s)` — the new cadence is
|
||||
in force, not merely compiled.
|
||||
|
||||
**Positive observable, per cycle** — two full hourly cycles observed after the deploy, from the
|
||||
container log:
|
||||
|
||||
```
|
||||
2026/08/14 06:23:13 [INFO] [scheduler] Registered periodic job: disk-health-check (every 1h0m0s)
|
||||
2026/08/14 07:23:13 [INFO] [scheduler] Running job: disk-health-check
|
||||
2026/08/14 07:23:14 [INFO] [web] disk-health check complete: 3 disk(s) evaluated, 0 alert(s)
|
||||
2026/08/14 07:23:14 [INFO] [scheduler] Job disk-health-check completed (took 849ms)
|
||||
2026/08/14 08:23:13 [INFO] [scheduler] Running job: disk-health-check
|
||||
2026/08/14 08:23:14 [INFO] [web] disk-health check complete: 3 disk(s) evaluated, 0 alert(s)
|
||||
2026/08/14 08:23:14 [INFO] [scheduler] Job disk-health-check completed (took 843ms)
|
||||
```
|
||||
|
||||
`grep -c disk_health_degraded` over the whole container log: **0**. Both cycles ran (849 ms / 843 ms,
|
||||
matching the §7 measurement), evaluated every disk, and emitted nothing. **Zero alerts from a check
|
||||
that demonstrably ran** — not silence.
|
||||
|
||||
Persisted state written by the first cycle (`/opt/docker/felhom-controller/data/disk-health-state.json`,
|
||||
428 bytes, on the `felhom-controller-data` docker volume, so it survives container recreation):
|
||||
|
||||
```json
|
||||
{"version": 1, "disks": {
|
||||
"path:/var/lib/vz": {"verdict": 1, "saw_uncorrectable": false, ...},
|
||||
"uuid:91d2dc2d-2d28-4929-9bdd-3e11fa2f41ae": {"verdict": 1, "saw_uncorrectable": false, ...}}}
|
||||
```
|
||||
|
||||
`verdict: 1` is `DiskVerdictOK` for both, `saw_uncorrectable: false`, never alerted.
|
||||
|
||||
**Reading those two artefacts against each other is what exposed R-335** — see §14.
|
||||
|
||||
### Leg 2 — the severity fix arrives (the point of the task)
|
||||
|
||||
Two synthetic `disk_health_degraded` events pushed for customer `demo-hp` **through the real hub
|
||||
event endpoint** (`POST https://hub.felhom.eu/api/v1/event`), from the guest's own controller using
|
||||
its own hub credentials — the genuine controller→hub path, not a hand-crafted operator call. Both
|
||||
returned `HTTP 200 {"ok":true}`. The hub DB was read with its `-wal` and `-shm` copied alongside
|
||||
`hub.db` (a `hub.db`-only read is stale).
|
||||
|
||||
**As STORED by the hub (`events`):**
|
||||
|
||||
| id | severity pushed | severity STORED |
|
||||
|----|-----------------|-----------------|
|
||||
| 2964 | `warning` | **`warning`** |
|
||||
| 2965 | `warn` | **`info`** ← coerced |
|
||||
|
||||
**`notification_log` rows for those two events:**
|
||||
|
||||
| id | event_type | severity | channel | status | error |
|
||||
|----|-----------|----------|---------|--------|-------|
|
||||
| 689 | `disk_health_degraded` | `warning` | `operator` | **`sent`** | *(none)* |
|
||||
| — | *(the `"warn"` push)* | — | — | **NO ROW EXISTS** | — |
|
||||
|
||||
**That pair is the proof.** The identical event, differing only in one word of the severity string,
|
||||
is the difference between *delivered to the operator* and *stored as an informational notice and
|
||||
delivered to nobody*. This is the first time this leg has been observed end to end.
|
||||
|
||||
Only the operator leg fired because **demo-hp has no `customer_notifications` row at all** (no
|
||||
customer email, no `enabled_events`), so no customer row was possible for either push — verified
|
||||
directly, not assumed. **One real email was sent to the operator**, as the task anticipated.
|
||||
|
||||
---
|
||||
|
||||
## 9. NOT yet live-validated — stated explicitly
|
||||
|
||||
**The Fail-from-counters path has never fired on real hardware.** Everything in §4/§5 exercises it
|
||||
against the committed fixture's values in unit tests only. The live legs above prove the *negative*
|
||||
(no false alert on three healthy disks) and the *severity wire* (end to end, through the hub) — they
|
||||
do **not** prove a live disk reaching Hiba. The fixture tests must not be read as a live proof.
|
||||
|
||||
Tracked as **R-332 (WATCHING)**. Closing condition: a live disk reaching Hiba from counters, or a
|
||||
deliberate injection through the real pipeline (agent `/disks` → controller check → hub event) — not
|
||||
a hand-set verdict.
|
||||
|
||||
**One item originally listed here has since been proven live** and is no longer part of this gap: the
|
||||
**persisted state surviving a controller restart**. The v0.215.0 → v0.216.0 redeploy destroyed and
|
||||
rebuilt the container, and the new one read back a `changed_at` written by the previous version rather
|
||||
than re-baselining — see §14. What remains unproven is the stronger half: an already-**alerted** disk
|
||||
not re-alerting after a restart, which needs a disk that has actually alerted. The drive that produced the fixture lives in DooPlex, which is Tier 2 and never a
|
||||
drill target; the demo boxes are all-flash and healthy.
|
||||
|
||||
---
|
||||
|
||||
## 10. Teardown
|
||||
|
||||
**This run provisioned nothing** — no VM, no guest, no hub customer record, no storage. Nothing was
|
||||
formatted, mounted, unmounted, repaired or written on any monitored disk; the only write is the
|
||||
controller's own `disk-health-state.json` inside its data volume.
|
||||
|
||||
Disposition of what the run did create:
|
||||
|
||||
- **Two synthetic hub events (`events` id 2964, 2965) and one `notification_log` row (id 689)** on the
|
||||
live hub. **Left in place deliberately.** Both messages are self-labelling
|
||||
(`"R-328 severity probe (…) - synthetic, no real disk fault"`), and deleting rows from the
|
||||
production hub DB is a riskier act than leaving two clearly-marked probe rows. Named here so they
|
||||
are not mistaken later for a real disk fault on demo-hp.
|
||||
- **One real operator email** resulting from row 689.
|
||||
- A local copy of `hub.db`/`-wal`/`-shm` in the session scratchpad only (not committed, not exported).
|
||||
|
||||
---
|
||||
|
||||
## 11. Register rows
|
||||
|
||||
| Row | State | Owner |
|
||||
|-----|-------|-------|
|
||||
| **R-328** — the severity drop: `"warn"` coerced to `info`, emailed to nobody | **CLOSED** (controller v0.215.0), proven live side by side | CC |
|
||||
| **R-329** — `app_start_failed` carries the identical defect | READY — **not fixed here**; needs a decision on whether it should notify at all | Viktor |
|
||||
| **R-330** — Phase 2: collect SMART attrs 187/199/188 + persist samples | READY — a declared wire change, hub models it in the same session under G-1 | CC |
|
||||
| **R-331** — Phase 3: growth-rate detection; revisit the static 64 | READY, blocked on R-330 | CC |
|
||||
| **R-332** — the Fail path has never fired on real hardware | **WATCHING** | CC |
|
||||
| **R-333** — NVMe temperature bands; agent `smartctl` has no `-n standby` | READY (S each) | Viktor decides (a); CC does (b) |
|
||||
| **R-334** — released with no golden carrying it (gate waiver) | READY — now applies to **v0.216.0** | CC bakes; **Viktor vouches** |
|
||||
| **R-335** — one physical disk walked twice per run, sustaining against itself | **CLOSED** (controller v0.216.0) | CC |
|
||||
|
||||
`smartd`-on-DooPlex-alerts-nobody is recorded in `DIAG-smart-passed-trap-2026-08-14.md` §8 as the
|
||||
same shape one layer out.
|
||||
|
||||
---
|
||||
|
||||
## 12. Observations — noticed, NOT acted on
|
||||
|
||||
1. **`app_start_failed` has the identical severity defect** (`notifier.go` ~L546, `"warn"`). Left
|
||||
untouched per scope. It needs a prior decision — should a stopped app email the customer at all? —
|
||||
because flipping the string alone converts a silent event into a mail flood on a crash-looping box.
|
||||
**R-329.**
|
||||
|
||||
2. **The 55/60 °C bands are spinning-disk bands being applied to NVMe, and this is close to biting.**
|
||||
Adopted unchanged from the operator's Prometheus config by explicit decision — but demo-hp's
|
||||
**healthy** Toshiba NVMe idles at **53 °C**, i.e. **2 °C below Figyelmeztetés and 7 °C below Hiba**,
|
||||
and NVMe routinely passes 60 °C under sustained write with no fault. As shipped, a healthy customer
|
||||
NVMe under load can be reported as **Hiba** — the single worst outcome this feature can produce, and
|
||||
the one leg 1 exists to guard. Not changed here because the threshold is a stated, settled operator
|
||||
decision; flagged rather than overridden. **R-333(a) — recommend splitting the bands by device
|
||||
class, or dropping them for NVMe and relying on `critical_warning`.**
|
||||
|
||||
3. **The agent runs bare `smartctl -a -j` with no `-n standby`**
|
||||
(`felhom-agent/internal/storage/hostops.go:368`), so every poll wakes a spun-down drive, and 6h → 1h
|
||||
multiplies that by six. Recorded, not acted on, per the task's instruction. demo-hp is all-flash so
|
||||
the measurement could not reveal it. Mitigating datum from the fixture: the failing drive logged
|
||||
only **3375 load cycles in 60505 power-on hours** (~one per 18 h), so this duty cycle barely spins
|
||||
down at all. **R-333(b).**
|
||||
|
||||
4. **`source ~/.config/credentials` prints two recovery codes to the terminal.** The file contains
|
||||
hyphenated keys (`R_DEMO-FELHOM`, `R_DEMO-HP`) that bash cannot assign, so sourcing it emits
|
||||
`command not found` errors **containing the secret values**. Anything that sources that file leaks
|
||||
them into logs, scrollback and transcripts. Not a code defect and out of scope; worth quoting
|
||||
values from it by other means, or renaming the keys.
|
||||
|
||||
5. **`golden_currency_gate.py` has no waiver parser.** Its own failure text says *"record a waiver in
|
||||
`OPEN-ITEMS.md` — never a bypass"*, but nothing reads such a waiver, so the only way past it is the
|
||||
bypass it warns against. See §13.
|
||||
|
||||
---
|
||||
|
||||
## 13. Deviations, stated plainly
|
||||
|
||||
- **`git push --no-verify` was used once**, on the `felhom.eu` docs push (`767960b`), and only there.
|
||||
Cause: `golden_currency_gate.py` correctly convicts the fact that controller **v0.215.0 is released
|
||||
and no golden carries it** (newest bake 0.214.0), so a *newly installed* machine would receive
|
||||
0.214.0 — without the severity fix. A golden bake was out of the task's scope, and its second half
|
||||
(vouching in the hub's day-0 artifact manifest) is operator-password-gated, so CC cannot complete it;
|
||||
a baked-but-unvouched golden is worse than none. Recorded as **R-334** with the bake+vouch owners
|
||||
named. CI re-runs the same entry point and will mail the operator. The running fleet is unaffected.
|
||||
- **One pre-existing test changed meaning by design:** `TestDiskVerdictFor`'s
|
||||
`critical_warning>0 → warn` case is now `→ fail` (truth-table row 4 — NVMe's own critical flag is a
|
||||
device declaration, not a drifting counter). `TestDiskHealthCheck_DegradationOnce` and its siblings
|
||||
were rewritten into the scenario groups because they encoded the pre-v0.215.0 single-alert behaviour
|
||||
the task deliberately replaces (Scenario C).
|
||||
|
||||
---
|
||||
|
||||
## 14. R-335 — a defect in v0.215.0, found live, fixed as v0.216.0
|
||||
|
||||
**How it was found.** Not by a test and not by review: by reading the release's own **positive
|
||||
observable** against the release's own **persisted artefact**. The hourly check logged *"3 disk(s)
|
||||
evaluated"*; `disk-health-state.json` held **two** records. Two artefacts that should have agreed did
|
||||
not.
|
||||
|
||||
**Cause.** demo-hp's `c11-scratch` and `felhom-backup` are the same physical NVMe (`/dev/nvme0n1`) and
|
||||
resolve to the same `diskKey`, so one disk was walked twice in a single run.
|
||||
|
||||
**Why it mattered.** `RunDiskHealthCheck` writes a disk's new record before the next entry reads it, so
|
||||
the **second** copy of an aliased disk consumed the **first** copy's write as its prior. The disk
|
||||
therefore **sustained against itself and reached Hiba on a first sighting** — defeating truth-table
|
||||
row 6, the single rule separating a one-hour benign excursion from a false critical alert — and would
|
||||
have emitted **two identical events** for one drive.
|
||||
|
||||
**Severity in practice: latent, not active.** Nothing fired on demo-hp because all three entries are
|
||||
healthy with zero counters. But any aliased disk developing one pending sector would have gone
|
||||
straight to Hiba, which is precisely the outcome §8 leg 1 exists to prevent. Aliasing is not exotic —
|
||||
it is the *normal* shape whenever a box has two PVE storage entries on one physical device.
|
||||
|
||||
**Fix (v0.216.0, `90f2545`).** Each `diskKey` is evaluated once per run. Both entries stay marked
|
||||
`seen`, so neither is mistaken for a disappeared disk, and the card still renders **both** storage
|
||||
rows — the dedup is about state and alerts, not display. Pinned by
|
||||
`TestDiskCheck_SameDiskTwiceIsEvaluatedOnce`, red-proof run and reverted (§5).
|
||||
|
||||
**Deployed:** `gitea.dooplex.hu/admin/felhom-controller:0.216.0 Up 6 seconds (healthy)`.
|
||||
|
||||
**Confirming cycle on v0.216.0 — CONFIRMED LIVE, 09:31:35Z:**
|
||||
|
||||
```
|
||||
live image: gitea.dooplex.hu/admin/felhom-controller:0.216.0 Up About an hour (healthy)
|
||||
2026/08/14 09:31:35 [INFO] [web] disk-health check complete: 2 disk(s) evaluated, 0 alert(s)
|
||||
grep -c disk_health_degraded: 0
|
||||
```
|
||||
|
||||
**`2 disk(s) evaluated` now matches the 2 persisted records.** The count and the artefact agree, which
|
||||
is the disagreement that exposed R-335 in the first place. Still zero alerts, still both card rows.
|
||||
|
||||
### The redeploy also proved persistence live — a gap §9 had listed as unproven
|
||||
|
||||
The 0.215.0 → 0.216.0 redeploy **replaced the container**, and the state file came back intact:
|
||||
|
||||
```json
|
||||
"path:/var/lib/vz": {"verdict": 1, "changed_at": "2026-08-14T07:23:14.640216851Z", ...}
|
||||
"uuid:91d2dc2d-…": {"verdict": 1, "changed_at": "2026-08-14T07:23:14.640216851Z", ...}
|
||||
```
|
||||
|
||||
That `changed_at` was written by **v0.215.0's first cycle at 07:23Z**, before the container was
|
||||
destroyed and rebuilt. The v0.216.0 container read it back and preserved it rather than stamping a
|
||||
fresh time — so the new container **loaded the pre-restart record instead of silently re-baselining**.
|
||||
That is Scenario L observed on real hardware, not just through the production-path unit test, and it
|
||||
is exactly the behaviour that was impossible before v0.215.0 (the baseline was in-memory).
|
||||
|
||||
It also incidentally confirms the unchanged-verdict path: `changed_at` is preserved across four checks
|
||||
and two controller versions because the verdict never changed, rather than being churned every cycle.
|
||||
|
||||
**What this still does NOT prove:** these disks are healthy and were never alerted, so the stronger
|
||||
half — *an already-ALERTED disk not re-alerting after a restart* — remains unit-tested only. R-332
|
||||
stands.
|
||||
|
||||
**Process note, recorded because it nearly cost the fix.** The red-proof harness reverts with
|
||||
`git checkout --`, which restores to `HEAD`. Running a red-proof against an **uncommitted** fix
|
||||
therefore *deletes the fix* along with the mutation — which happened here and was caught only by
|
||||
re-grepping the source afterwards. Commit the fix before red-proofing it, or snapshot outside git.
|
||||
|
||||
@@ -12,10 +12,16 @@
|
||||
|---|---|---|---|---|
|
||||
| `NamespaceRoot` | controller/internal/appbackup/paths.go | `(drivePath string, inGuestDrive bool) string` | Resolve felhom-data root for a drive | `inGuestDrive=true` returns path AS-IS (Model A: guest mount IS the ns root); false appends `felhom-data`. Never double-nest |
|
||||
| `PrimaryBackupPath` / `RecoveryUnitPath` / `RecoveryUnitComposePath` / `RecoveryUnitManifestPath` | controller/internal/appbackup/paths.go | `(nsRoot[, stackName]) string` | All backup dir layout | Take the NAMESPACE ROOT, not a bare drive path |
|
||||
| `AppDBDumpPath` / `AppVolumeDumpPath` / `AppDataDir` | controller/internal/appbackup/paths.go | `(nsRoot, stackName) string` | Per-app dump/data dirs | Same nsRoot contract |
|
||||
| `UserdataDir` / `EnsureUserdataSkeleton` / `EnsureDirOwned` | controller/internal/appbackup/userdata.go | `(nsRoot)` / `(path, gid int)` | userdata/ tree w/ 2775 setgid gid-1000 convention | Linux-only chown via build-tag twin userdata_linux.go |
|
||||
| `AppDBDumpPath` / `AppVolumeDumpPath` / `AppDataDir` | controller/internal/appbackup/paths.go | `(nsRoot, stackName) string` | Per-app dump/data dirs | Same nsRoot contract. `AppDataDir`'s final segment is the app's real appdata dir NAME — NOT always the stack name (paperless-ngx → `paperless`); resolve via `AppDataDirNames` first (F-S2/F-S3) |
|
||||
| `AppDataDirNames` / `AppDataBindsPresent` | controller/internal/appbackup/paths.go | `(hddPath, stackName string, hddMounts []string) []string` / `(hddPath, hddMounts) bool` | Resolve the real `appdata/<name>` dir(s) from compose `${HDD_PATH}` binds (F-S2/F-S3) | `hddMounts` = ParseComposeHDDMounts shape. Deduped+sorted; falls back to `[stackName]` when no appdata bind. Tier-2 (`backup.Manager.tier2AppDataName`) refuses N>1; migrate (`stacks.Manager.ResolveAppDataDirNames`) loops N. `BindsPresent` drives the WARN-on-missing-declared-dir |
|
||||
| `UserdataDir` / `ImportDir` / `EnsureUserdataSkeleton` / `EnsureDirOwned` | controller/internal/appbackup/userdata.go | `(nsRoot)` / `(nsRoot)` / `(nsRoot, dirs []string)` / `(path, gid int)` | userdata/ tree w/ 2775 setgid gid-1000 convention. **R-75:** `ImportDir` is the CANONICAL drop-zone (`<nsRoot>/userdata/import`) and callers MUST resolve it against the SYSTEM namespace, never an app's HDD_PATH — use `stacks.Manager.GetImportRoot()`. `EnsureUserdataSkeleton` now takes the dir set: build it with `BuildUserdataSkeleton(DeriveUserdataDirs(stacksDir))`, or via `Manager.EnsureUserdataSkeleton` / `web.Server.ensureUserdataSkeleton`. | Linux-only chown via build-tag twin userdata_linux.go. **The set MUST stay sorted** — `fbNeedsRecreate` force-recreates FileBrowser on any byte diff and the naive map-order derivation measured 20/20 distinct (SPIKE P6). `UserdataSkeletonCarry()` is the old hardcoded list, retained forever so derivation can only ADD (zero removals). |
|
||||
| `BuildUserdataSkeleton` / `UserdataSkeletonCarry` / `DeriveUserdataDirs` | appbackup/userdata.go, stacks/skeleton_derive.go | `([]string)` / `()` / `(stacksDir)` | catalog-derived userdata skeleton (R-75) | Derives `${USERDATA_PATH}` binds only — `${IMPORT_PATH}` is NOT part of a drive skeleton (one root, system drive, `Manager.EnsureImportRoot`). Do NOT wire the catalog sync to `SyncFileBrowserMounts`. |
|
||||
| `appbackup.ValidateRelPath` / `ValidRoot` | controller/internal/appbackup/classify.go | `(root, path)` / `(root)` | THE single path-safety refusal set for every `${VAR}`-relative catalog path | Shared by `backup:` and `data_paths:`. **Do not write a second path validator.** |
|
||||
| `stacks.ValidateDataPaths` | controller/internal/stacks/datapaths.go | `(entries, binds, appName, logger)` | `data_paths:` annotation validation | ASYMMETRIC on purpose (Fork-3): malformed PATH ⇒ whole-block reject (data handling, `backup:` precedent); unknown ROLE ⇒ fails OPEN, one WARN (presentation, `Lifecycle` precedent). |
|
||||
| `web.fileBrowserLink` / `importFolderLink` | controller/internal/web/filebrowser_link.go | `(domain, sourceName, relPath)` | FileBrowser Quantum deep link | Template read out of the shipped router (SPIKE P2). **`url.PathEscape` per segment — NEVER `QueryEscape`** (space→`+` is a literal plus in a path). Let `html/template` do the attribute escaping; do not pre-escape. |
|
||||
| `HumanizeBytes` | controller/internal/appbackup/appdata.go | `(b int64) string` | Human byte sizes | Exported canonical; private clones exist (§6) |
|
||||
| `stablePathForName` / `agentWhere` | controller/internal/web/intermediary.go | `(name/registeredPath) string` | Map registry stable path `/mnt/felhom-drives/<n>` ↔ raw agent mount | Registry stores STABLE path; agent ops take the RAW mount — always convert |
|
||||
| `offsiteRestoreRootFor` | controller/internal/backup/offbox_verify_copies.go | `(drivePath string) string` | THE only place `backups/offsite-restore` is spelled | `offboxRestoreScratchDir` builds on it — the listing/delete surface MUST resolve byte-identical paths to what the restore wrote. Do not re-hardcode the segments (they were open-coded in 3 places before v0.147.0) |
|
||||
| `ProtectedHDDPaths` | controller/internal/stacks/delete.go | `(hddPath string) map[string]bool` | Never-delete set (root, appdata, backups, media, legacy felhom-data) | Consult before ANY recursive delete under a drive |
|
||||
|
||||
### Subprocess + timeout + exit-code discipline
|
||||
@@ -39,6 +45,9 @@
|
||||
| `jsonResponse` / `jsonError` | controller/internal/web/handler_export.go | `(w, v)` / `(w, msg, code)` | Export/import API | Third envelope shape — keep within export surface |
|
||||
| `limitBody` | controller/internal/api/router.go | `(w, req)` | Bound request bodies (1MB) | Apply before decode on any new POST |
|
||||
| `offboxRedirect` | controller/internal/web/offbox_handlers.go | `(w, r, msg string, isErr bool)` | Flash-message redirects | Flash = `?flash=` / `?flash_error=` query params, read by page handlers |
|
||||
| `offboxRedirectTo` | controller/internal/web/offbox_handlers.go | `(w, r, page, msg string, isErr bool)` | Same, to an EXPLICIT page | **TRAP (fixed v0.154.0): the separator is chosen, not `"?"`.** Targets may already carry a query — the R-48 wizard is `/backups/restore/app?name=<app>` — and a hardcoded `"?"` buries the flash inside the previous parameter's value |
|
||||
| `restoreOpInFlight` + `hasRecentRestoreResult` | controller/internal/web/restore_wizard.go | `(backup.RestoreOpStatus) bool` / `(st, app, now) bool` | THE "is a restore running / did one just finish" display reads | **TRAP (v0.154.0 shipped this bug): `Manager` has TWO running flags.** `IsRunning()` reads the CONCURRENCY flag, acquired inside the goroutine — and `RestoreOffboxScratch` never acquires it, so it is false for the whole verification restore. Display must read `RestoreStatus().Running` (set synchronously by `BeginRestoreOp`). Read the status ONCE per render or the strip and the suppression can disagree. `hasRecentRestoreResult` is app-bound and window-bounded — a process-wide result must not light another app's „Eredmény" |
|
||||
| `restoreWizardPath` / `deriveWizardStep` / `resolveWizardApp` | controller/internal/web/restore_wizard.go | `(app) string` / `(restoreWizardInput) restoreWizardView` / `([]OffboxAppRow, name) *OffboxAppRow` | R-48 offsite restore wizard: URL builder + the PURE step/unlock derivation + the app-resolution refusals | The step is **never** taken from the request. Precedence is load-bearing: op-running outranks a stale `?full_prep=`, else a commit button reappears mid-restore. Truth table + red-proof: `restore_wizard_test.go`. Adding a form here that posts anywhere new breaks `TestRestoreWizard_NoNewMutationEndpoints` **by design** — R-48 adds no mutation surface |
|
||||
| `redirectTier2` | controller/internal/web/tier2_config_handler.go | `(w, r, name, flash, flashErr)` | Tier2 page flash redirects | Same convention |
|
||||
| `validStackName` | controller/internal/web/validate.go | `(name string) bool` | Any stack name from a request | Single-segment, no `/ \ ..` — blocks path traversal into stacks/userdata |
|
||||
| `ValidateSegment` | controller/internal/appexport/validate.go | `(kind, s string) error` | Any attacker-controlled path segment (.fab manifest fields) | CTRL-001 guard; deliberately NOT for dotfile ConfigFiles |
|
||||
@@ -48,6 +57,10 @@
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
|---|---|---|---|---|
|
||||
| `backup.ErrOffboxSealedPackageHeld` + `IsOffboxSealedPackageHeld` + `sealedPackageHeld` + `OffboxAwaitingRecoveryKey` (R-241, v0.206.0) | controller/internal/backup/offbox.go | sentinel; `(error) bool`; `() bool`; `() bool` | **THE MINT GUARD** — a box never creates a repository key while the hub holds a sealed package for it | **The guard is a CONJUNCTION** (package held AND no key present). Widening it to "never mint" leaves a first-time box unable to start, waiting for a package that will never exist — pinned by `TestR241_ScenarioB_FirstTimeBoxStillMints`. **The refusal is a HOLDING state, not a failure:** `ApplyOffsiteTarget` catches the sentinel and still writes the transport, so `/recovery`'s synchronous tier-up (R-219) can bring the tier up the instant the key arrives; returning the error instead leaves `needsOffsiteCredential` true and the hub re-staging a consumed credential for ever. `OffboxAwaitingRecoveryKey` is **DERIVED, never stored** — and **`t.Enabled` is load-bearing in it**: a customer who switched off-site OFF is not awaiting anything (the Scenario-E carve-out `needsOffsiteCredential` makes two functions above; the first draft omitted it and an existing test caught it). A nil settings store reads as "no package held" — a transient read failure must never become a permanently-held tier |
|
||||
| `settings.HubEscrowKeySHA256` + `SetHubEscrowKeySHA256` / `GetHubEscrowKeySHA256`, and `OffsiteRecoveryOffer` **shape (c)** (R-241, v0.206.0) | controller/internal/settings/settings.go, controller/internal/backup/offbox.go | `(sha, checkedAt string) error` / `() (string, string)` | **THE DISCRIMINATOR the recovery screen asks** — does the hub hold a package for a key other than the one we use? | **The comparison was ALREADY computed on every ACK since SLICE 3 and persisted nowhere** — that is R-241's second half. Wire the recorder in `main.go`'s `EscrowAutoConfirmer` literal or shape (c) reads an empty hash for ever and the fix ships INERT (pinned by `TestMainWiresRecordEscrowKeyHash`). **§7.2 staleness, decided:** a KNOWN DIFFERENCE offers **however old the reading** — age is deliberately NOT gated on, because gating makes a box offline from the hub silently stop offering; an **ABSENT hash falls back to (a)/(b)** and does NOT offer, because `""` is the hub positively saying its package seals no key (legacy hash-less escrow), not an unknown. `CheckedAt` is for diagnosis, never a gate |
|
||||
| `backup.AbandonStatus` / `AbandonSweep` / `CancelAbandon` / `ClearAbandonPurgeIfConfirmed` / `ExtendAbandon` / `StopAbandon` + `AbandonGraceDays` (R-241, v0.206.0) | controller/internal/backup/offbox_abandon.go | see file | **The 14-day abandonment countdown** — the ONLY thing in the product that deletes a customer's off-site history | **BOTH HALVES OR NEITHER.** The set-aside store and the sealed package that protects it are two halves of one thing; removing only one leaves a package that opens nothing, or ciphertext nobody can decrypt. Not atomic across two machines, so it is a **two-phase commit**: delete the store, set `AbandonPurgeRequested`, and keep declaring it until the hub's ACK stops reporting a superseded package — the confirmation rides the SAME ACK as the request. **The countdown starts in `ResetOrphanedRepo`, NOT in the shared `resetOrphanedRepo`** — the helper is also the UNCLAIMED auto-reset, where nobody decided anything. **The recovery offer stays reachable for the whole grace** (a grace in which recovery is impossible is decorative). **Drive it with `SetOffboxClock`, never a shortened live timer** (§7.4). A transport failure leaves the countdown DUE so tomorrow retries; the operator levers REFUSE rather than no-op when nothing is running or the store is already gone |
|
||||
| `settings.SyncRecoveryOfferEpoch` / `PostponeRecoveryNoticeForEpoch` / `OptOutRecoveryRemindersForEpoch` + `web.recoveryBannerCookie` (R-241, v0.206.0) | controller/internal/settings/settings.go, controller/internal/web/recovery_handlers.go | `(offered bool, now) (RecoveryOfferView, error)` | **The offer EPOCH** — "once per entry into the offered state", not once ever | **Sync the epoch FIRST and UNCONDITIONALLY in `recoveryInterrupts`.** The first draft returned early when the offer was false, so the FALLING edge was never recorded, `RecoveryOfferActive` stayed true through a settled period, and the next entry counted as a continuation — **the exact defect the epoch exists to fix, reintroduced inside the fix**. Dismissals are recorded against the epoch they were made in, so a fresh entry resets them **by arithmetic**, with nothing to clear. **Three levers, three scopes, and NONE removes the entry point on `/backups/remote`:** the banner cookie is a browser SESSION cookie (no MaxAge — cleared on login) and persists nothing; the reminder opt-out is durable but silences the BANNER ONLY; "most nem" suppresses the full page only |
|
||||
| `atomicWrite` | controller/internal/backup/recovery_unit.go | `(path, data, perm) error` | Atomic file writes (backup pkg) | tmp+rename; no dir creation, no fallback |
|
||||
| `writeFileAtomic` | controller/internal/bootstrap/bootstrap.go | `(path, b) error` | controller.yaml writes from bootstrap | Always 0600 (holds local-api token + hub key) |
|
||||
| `writeConfig0600` | controller/internal/api/router.go | `(path, body) error` | config writes via API | ALWAYS chmods 0600 even pre-existing (F8); direct-write fallback on bind-mount EBUSY (non-atomic!) |
|
||||
@@ -55,7 +68,19 @@
|
||||
| `Settings.save` (unexported) | controller/internal/settings/settings.go | via mutator methods only | ALL settings.json persistence | tmp+rename, then `.bak` last-known-good AFTER rename succeeds. Never write settings.json by hand |
|
||||
| `settings.Load` | controller/internal/settings/settings.go | `(path, logger) (*Settings, error)` | Startup load | Corruption recovery: `.bak` restore → else preserve `.corrupt-<ts>` + safe defaults; never crash-loops |
|
||||
| `Manager.writeJournal` / `loadJournal` | controller/internal/stacks/migrate.go | `(j *MigrationJob)` | Migration crash journal | Enables `RecoverMigration` at startup |
|
||||
| `backup.SharesPseudoStack` / `DisplayStackName` | controller/internal/backup/shares_payload.go | `"_shares"` / `(key) string` | THE reserved key for the shares source (restic tag, `backups/secondary/_shares`, CrossDriveBackup record) + its display mapping | NEVER let the raw key reach a Hungarian surface — map at the notification/prose boundary ONLY; the persisted `EnlargedBlocked` set and the templates index by the RAW key |
|
||||
| `Manager.buildSharesPayload` / `classifiedShares` | controller/internal/backup/shares_payload.go | `() (dir, passdbOK, error)` / `() []classifiedShare` | the definitions+credential payload and the availability-filtered share set both tiers read | payload is SECRET-BEARING (0600 passdb.tar) — never log its bytes/name at INFO. `classifiedShares` is the single place a dead mount is dropped, so both jobs agree |
|
||||
| `Manager.selectTier2TargetFrom` | controller/internal/backup/tier2.go | `(stack, sourceDrive, fullSize, stateOnlySize) (*Tier2Target, error)` | tier-2 target choice with the source drive supplied EXPLICITLY | the seam the shares job reuses — NEVER fork the headroom math; `selectTier2Target` is now a thin wrapper over it |
|
||||
| `Manager.tier2ReconcileRoots` | controller/internal/backup/tier2.go | `(destBase, roots, legRels)` | staleness pruning with explicit dest roots | pure extraction from `tier2Reconcile` (which now calls it with `hdd`/`userdata`); reuse it rather than writing a second pruner |
|
||||
| `Manager.liveShareRootOK` / `scratchJoin` | controller/internal/backup/shares_restore.go | `(dst) bool` / `(scratch, abs) string` | THE place guard for shares restore + scratch path reconstruction | a snapshot is UNTRUSTED layout input: require a STRICT descendant of a live registered root, refuse `..` and the drive root itself. `scratchJoin` strips the volume name — plain `filepath.Join` splices a drive letter mid-path |
|
||||
| `infra.SambaContainerName` / `SambaPassdbVolume` / `SambaPassdbMount` | controller/internal/infra/samba.go | consts | single source of truth for the samba container identity | the compose renderer interpolates them; stacks/backup/monitor read them. The CONTAINER name (`felhom-samba`) is NOT the stack name (`samba`) — `EffectiveProtected` needs the container one |
|
||||
| `sambaWriteAtomic` | controller/internal/stacks/samba.go | `(path, data, mode) error` | samba smb.conf/compose writes | tmp+**fsync**+rename (the only one of these that fsyncs). Fourth atomic-write helper in the tree — see §6 |
|
||||
| `Loop.writeMarker` / `Recover` | controller/internal/quiesce/quiesce.go | `(m Marker)` / `()` | Quiesce crash-safety | Marker written BEFORE stopping stacks; Recover restarts stranded stacks at boot |
|
||||
| `quiesce.TieredBackend` + `Loop.resolveDueTiers` / `quiesceAndPollTiers` | controller/internal/quiesce/tiers.go, quiesce.go | `Tiers/DueFor/StartBackupFor/BackupStatusFor`; `resolveDueTiers(ctx) ([]dueTier,bool,error)` | THE R-82 multi-tier backup schedule — several whole-guest tiers (local daily + PBS weekly) reconciled into ONE quiesce window | **Both tiers due ⇒ ONE stop/start pair**, never two (two = two app outages for one night). Tiers run SEQUENTIALLY (vzdump holds a guest lock) and the app stays down until the LAST tier snapshots — resuming earlier loses app-consistency on the DR tier. Order is fast-first (agent advertises primary first) or downtime blows up. `ErrTiersUnsupported` (route 404) ⇒ pre-R-82 agent ⇒ degrade to the untargeted path and **STILL BACK UP** — never read it as "nothing due". |
|
||||
| `quiesce.failureBreaker` + `Loop.dropBackedOffTiers` / `noteTierFailure` / `noteTierSuccess` | controller/internal/quiesce/breaker.go, quiesce.go | `blocked/recordFailure/recordSuccess(target, now)`; `backoffFor(n) time.Duration` | **R-88** — a tier whose backups keep failing stops re-quiescing. Backoff 15m→30m→1h→2h→4h (cap), reset on success | **It gates the QUIESCE, not the backup** — the harm was never the failing backup, it was the app outage taken to attempt it, so backed-off tiers are dropped from the due set BEFORE any stack is stopped. **Per TARGET** — a broken offsite tier must never suppress a healthy local one (`TestBreaker_OneFailingTierDoesNotSuppressAHealthyOne`). **Never permanent** — the cap bounds the retry INTERVAL, it never stops retrying; a latched breaker is a silent backup outage, worse than the loop it replaces. **`TriggerNow` is never gated** (it already bypasses due-ness and the window gate), though a manual run still RECORDS its outcome. **`stillRunning` is NOT a failure** — a first full offsite snapshot legitimately runs for hours. State is **in-memory on purpose**: a restart forgets the backoff and re-attempts, which is the cheap direction to fail. Log the deferral ONCE when armed, never per tick. |
|
||||
| `quiesce.TierNotifier` + `Loop.SetTierNotifier` / `noteTierFailure` / `noteTierSuccess` | controller/internal/quiesce/breaker.go, quiesce.go | `BackupFailed(tier,msg,err)` / `BackupRecovered(tier,msg)`; `SetTierNotifier(n)` INIT-ONLY | **R-97a** — the whole-guest backup tier reports its outcome to the hub | A **seam, not an import** — quiesce keeps no dependency on `internal/notify` (same reason `windowStartFn` is injected). Wired by a setter because main.go builds the notifier AFTER the loop; `nil` = unprovisioned guest, not an error. **Edge-triggered:** failure fires only when the breaker ARMS (`n == 1`), never per retry — the cadence is 15m/30m/1h/2h/4h and an event per attempt is an inbox nobody reads. Recovery rides `recordSuccess`'s existing bool. **Event types are OPERATOR-ONLY** (`whole_guest_backup_failed`/`_recovered`, hub >= v0.78.0) — NOT `backup_failed`, which has a customerMessages entry AND sits in live `enabled_events`, so it would email the CUSTOMER about a backup they cannot act on. `WholeGuestBackupDetails.Tier` is load-bearing: the hub keys its per-tier cooldown on it. |
|
||||
| `quiesce.Loop.SuppressedStacks` + `markQuiesced` / `markUnquiesced` | controller/internal/quiesce/suppress.go | `() map[string]bool` (nil-safe on a nil *Loop) | **R-97b** — an app THIS controller stopped for a backup is not a fault | Consumed at the SINGLE derivation point `classifyRunStates` (which computes both the banner dead-list and the notifier Down-set — keep it one place). **Cycle-keyed, not state-based:** v0.164.0's `!= StateStopped` filter cannot see an app caught MID-RESTART (`starting`/`unhealthy`), which is how BookStack alarmed on 2026-07-27. The window (`quiesceAlarmGrace` = 180 s, derived from the deploy flow's 120 s health timeout and Mealie's 60 s start_period) **EXPIRES** — permanent suppression turns a loud false alarm into a silent real one. Open-ended while the cycle runs (a first offsite snapshot legitimately takes hours). |
|
||||
| `agentapi.BackupTiers` / `BackupDueFor` / `StartBackupFor` / `BackupStatusFor` | controller/internal/agentapi/backup_tiers.go | `(ctx[, target]) (…, error)` | The per-tier agent surface (agent >= v0.97.0) | `targetQuery("")` returns an EMPTY suffix so an untargeted call hits the pre-R-82 route byte-for-byte. `BackupTiers` maps a 404 to `ErrTiersUnsupported` — the documented ROUTE-PROBE capability signal, NOT a `featureProbes` row (the loop needs the tier LIST, not a yes/no). |
|
||||
|
||||
### Compose ops / stack lifecycle
|
||||
|
||||
@@ -63,14 +88,28 @@
|
||||
|---|---|---|---|---|
|
||||
| `Manager.DeployStack` | controller/internal/stacks/deploy.go | `(req DeployRequest) (string, error)` | Full deploy flow | Sets in-memory `Deployed` BEFORE compose up (slow-pull race), reverts on failure |
|
||||
| `Manager.RedeployFromEnv` | controller/internal/stacks/deploy.go | `(name, env map[string]string) error` | Re-up with changed env (migration flip, config edits) | `compose up -d`, never `restart` (restart won't pick up images/env) |
|
||||
| `Manager.StartStack/StopStack/RestartStack/UpdateStack` | controller/internal/stacks/manager.go | `(name string) error` | Lifecycle | Protected stacks refuse stop; all funnel through composeExec |
|
||||
| `Manager.PersistUnitRedeployConfig` (R-47, v0.153.0) | controller/internal/stacks/deploy.go | `(name, env map[string]string) error` | the PERSIST half of `RedeployFromEnv` — app.yaml + locked fields + in-memory flags, **starts nothing** | **TRAP: the restore paths must use THIS, never `RedeployFromEnv`.** RedeployFromEnv ends in a full `up -d`, which before the replay IS the H4 race. RedeployFromEnv is now literally this + the unchanged up-and-report tail |
|
||||
| `Manager.StartStackServices` (R-47, v0.153.0) | controller/internal/stacks/manager.go | `(name string, services []string) error` | scoped `compose up -d <svc>...` — the DB-only window a dump is replayed in | **REFUSES an empty list** (argument-less `up -d` is a FULL start — the one silent fall-through that would reintroduce the race). No `logPostStartStatus`: the app containers are absent on purpose. Never `RestartStack` here — it is a full up in disguise |
|
||||
| `appbackup.DBServiceNames` / `dbTypeForImage` (R-47, v0.153.0) | controller/internal/appbackup/dbservices.go | `(composePath string) ([]string, error)` | naming the compose SERVICE(s) holding a database, sorted | yaml.v3 `services:` MAP parse — **never a line scan** (immich's top-level `immich_ml_cache:` / `immich_postgres_data:` volume keys look exactly like services). `dbTypeForImage` is shared with `DiscoverDatabases`, which is what makes "a dump exists ⇒ a service can be named" hold. An error means CANNOT-TELL, never "no database" — callers refuse when a dump exists |
|
||||
| `Manager.StartStack/StopStack/RestartStack/UpdateStack` | controller/internal/stacks/manager.go | `(name string) error` | Lifecycle | Protected stacks refuse stop; all funnel through composeExec. **NOT writers of desired state (R-166)** — 14 call sites, only 2 are the customer; recording intent here would make a nightly backup indistinguishable from the customer pressing Stop. Use `SetDesiredState` at the intent point instead |
|
||||
| `Manager.SetDesiredState` / `DesiredStateOf` / `BackfillDesiredState` (R-166, v0.189.0) | controller/internal/stacks/desiredstate.go | `(name, desired string) error` / `(Stack) string` / `() int` | THE customer-intent record — `app.yaml` `desired_state`, tri-state `""`/`running`/`stopped` | **ONE OWNER: the customer's action.** Writers are the API action switch, `DeployStack`, `UpdateOptionalConfig`'s redeploy branch, and the `.fab` restore adapter — nothing else, ever. **`""` (absent) means UNKNOWN, never "running"**: every pre-v0.189.0 app.yaml reads absent, so treating it as running would start every deliberately-stopped app on upgrade. Write intent BEFORE the act and REFUSE the act if it fails (§8.2). Backfill is **running-only** — never infer `stopped` from zero containers, that inference IS the defect |
|
||||
| `Manager.DriveLive` (R-171, v0.190.0) | controller/internal/stacks/deploy.go | `(hddPath string) bool` | is an app's data drive a live mountpoint RIGHT NOW | Wraps the **same** `isMountPoint` seam the userdata belt uses (`manager.go`) — never write a second liveness check, the two would drift invisibly. The system/local path is legitimately not a mountpoint and returns true |
|
||||
| `bootrecon.StartGate` (R-171, v0.190.0) | controller/internal/bootrecon/bootrecon.go | `MayStart(stack) (bool, reason)` | THE one question the boot sweep asks before starting anything | **Fail-safe: cannot determine ⇒ return FALSE.** One seam for all three holders (absent drive · quiesce · an in-flight app-data operation) because they differ only in the reason string. Implemented in `main.go` (`bootDriveGate`) reusing `quiesce.SuppressedStacks()`, `AppStopGuard.HeldStacks()` and `Manager.DriveLive` — never re-derive any of them. Held apps go to `Result.HeldByDrive`, **never** `StillDown` (that is the dead-app alarm's bucket) |
|
||||
| the boot settle window (R-157 A, v0.190.0) | controller/cmd/controller/main.go | `bootReconcileSample` / `StableFor` / `Budget` | sample the fleet until it stops changing, then sweep ONCE | **settle + budget + one `DefaultRetryDelay` must stay under `deadAppBootGrace`** — pinned by `TestBootWindow_CommonCaseFitsInsideTheDeadAppGrace`, which is why the budget is 50 s and not 60 s. Sampling is READ-ONLY; sweeping per sample would never see a settled fleet (the sweep's own StartStack changes it). A late recovery is REPORTED (`recordLateRecovery`), never hidden by widening the grace |
|
||||
| `backup.AppStopGuard` (`Begin`/`End`/`Recover`) (R-166, v0.189.0) | controller/internal/backup/appstop_marker.go | `(opID, reason, stacks) error` / `()` / `() *AppStopRecovery` | THE crash marker for stop→work→start windows (volume dump, offbox reconstitute, `.fab` export) | Its **own** file (`appstop-state.json`), never quiesce's — one file, one writer. **A `defer` is NOT the mechanism** (Campaign 8 fault 10: SIGKILL runs no defer); the marker is. Written BEFORE the stop, cleared ONLY after a restart that succeeded; a FAILED restart deliberately KEEPS it. `Recover` RETURNS its outcome rather than notifying, because it must complete before the boot reconciler while the notifier does not exist yet |
|
||||
| `backup.ErrStartRefused` + `AppStopRecovery.Refused`/`Alarming()` (R-174, v0.191.0) | controller/internal/backup/appstop_marker.go | `errors.Is(err, ErrStartRefused)` / `() bool` | THE refusal-vs-failure split in the app-stop crash recovery | **A gated starter's refusal is NOT a restart failure.** `Recover`'s starter MUST be the gated `gatedAppStopStarter` (cmd/controller/main.go), never the raw `stacks.Manager` — that was the v0.189.0 defect, which started apps onto ABSENT drives at boot (R-171 one path over). A refusal goes to `Refused` (marker KEPT, silent), a real error to `Failed` (marker kept, ALARMS). Collapsing them routes a deliberate hold into `NotifyBackupFailed`, a customer-enabled type — the R-171 false alarm again. `main.go` must guard the notify with `Alarming()`, not `!= nil` |
|
||||
| `Manager.DeleteStack` / `RemoveStack` | controller/internal/stacks/delete.go | `(name, removeHDDData[, backupPaths])` | THE guarded removal paths | Orphan/protected/deploying/running checks + ProtectedHDDPaths filter before any RemoveAll |
|
||||
| `resolveContainerState` / `aggregateState` | controller/internal/stacks/manager.go | `(dockerState, dockerStatus)` / `([]ContainerInfo)` | State classification | `.State` says "running" even when unhealthy — `.Status` parse is the fix |
|
||||
| `Manager.logPostStartStatus` | controller/internal/stacks/manager.go | `(name, stackDir, env)` | Async post-start verification | compose up exits 0 on crash-loops; this is the detection. Goroutine + 3s, never blocks |
|
||||
| `Manager.EnsureBaseStack` | controller/internal/stacks/infra.go | `() error` | Traefik/cloudflared/FileBrowser infra convergence | Renders from `internal/infra` templates |
|
||||
| `appbackup.ClassifyBinds` / `ValidateBackupSpec` | controller/internal/appbackup/classify.go | `(spec, binds) ([]ClassifiedBind, bool)` / `(spec, binds) error` | Backup-classification (Task 2, referential coupling) — pure | Two-level default: explicit wins over `:ro`; unlisted writable→mandatory, unlisted `:ro`→excluded; nil spec→legacy/false. Validate REJECTS the WHOLE block on any defect (whole-block semantics). INERT — no tier consumes it yet |
|
||||
| `ParseComposeClassifiableBinds` | controller/internal/stacks/classify_binds.go | `(composePath) []appbackup.ComposeBind` | `${VAR}`-relative binds + `:ro` for classification | Do NOT use `ParseComposeHDDMounts`/`ExportDataMounts` as classifier input (§traps) — they resolve absolutes, drop `:ro`, or union the userdata ROOT. Short-syntax only |
|
||||
| `Metadata.EffectiveLifecycle` / `CanInstall` / `IsAbandoned` + `web.lifecycleBadge` / `web.visibleCatalogStacks` | controller/internal/stacks/metadata.go, controller/internal/web/metabadge.go, controller/internal/web/handlers.go | `meta.CanInstall() bool` | app lifecycle: `available` / `hidden` / `abandoned` (v0.158.0) | THE single interpretation of `.felhom.yml` `lifecycle:` — every surface must go through these, never compare the raw string. Listing drops `!Deployed && !Protected && !CanInstall()`; `api.deployStack` refuses server-side BEFORE any mutation (hiding a button is not a gate), `stacks.DeployStack` repeats it for non-API callers. **Unknown value fails OPEN** (→ available + one WARN) — opposite to the gate on purpose: a typo must never pull a working app out of every catalog. **NEVER let lifecycle reach orphan detection** (`getCatalogTemplateSlugs`) — a withdrawn template stays in the tree, or every deployed instance reads as `Elavult` and gets a Törlés button. Badges: `MetaBadge` + `meta_badge` partial, built generic for R-56 difficulty labels |
|
||||
| `Manager.ClassifiedBinds` + `StackDataProvider.GetStackClassifiedBinds` | controller/internal/stacks/metadata.go, appbackup/appdata.go | `(name) ([]appbackup.ClassifiedBind, bool)` | Per-stack classification through the REAL LoadMetadata validate path | The wired seam Task 3 consumes; LoadMetadata is the SINGLE validation choke point (bad block → nil + one ERROR → legacy) |
|
||||
| `backup.Manager.DumpAppVolumesSafe` | controller/internal/backup/backup.go | `(stackName) error` | Volume tar of a live app | Stops → dumps → restarts; surfaces BOTH errors (app may be left stopped). Check `GetDockerVolumes()!=0` + `IsProtectedStack` BEFORE calling — it stops the stack before its own volume check (see `runVolumeDumps`) |
|
||||
| `backup.Manager.ListRestorePoints` | controller/internal/backup/restore_points.go | `(stackName) ([]RestorePoint, bool)` | Restorable keep-side backups (the /api/backup/snapshots payload) | ONE point per app (the current unit); tier always 1 — never list Tier-2 (not restorable via /backup/restore) |
|
||||
| `backup.Manager.RestoreTier2Files` | controller/internal/backup/tier2_restore.go | `(stackName) (filesRestored int, err error)` | In-place ADDITIVE-ONLY class-C file restore from the recorded Tier-2 copy (`POST /backup/tier2/restore`) | Never overwrites/deletes live files; refusals (Hungarian) before any stop; source = recorded `DestinationPath`, never re-selected |
|
||||
| `backup.Manager.RestoreTier2Files` | controller/internal/backup/tier2_restore.go | `(stackName) (filesRestored int, err error)` | In-place ADDITIVE-ONLY class-C file restore from the recorded Tier-2 copy (`POST /backup/tier2/restore`) | Never overwrites/deletes live files; refusals (Hungarian) before any stop; source = recorded `DestinationPath`, never re-selected. **C9-F1 (v0.183.0): reads `hdd/` + `userdata/` ONLY — never `recovery-unit/`.** For 43 of 53 catalog apps that is a guaranteed no-op, so it now refuses with `ErrTier2NoRestorableData` BEFORE stopping the app. Ask `Tier2RestoreCoverage` first |
|
||||
| `backup.Manager.Tier2RestoreCoverage` | controller/internal/backup/tier2_restore.go | `(stackName) (Tier2Coverage{Legs, HasUnit}, error)` | Answers what a Tier-2 restore CAN and CANNOT return for an app, from the RECORDED copy on disk | **C9-F1.** `Legs` = subtrees the restore reads; `HasUnit` = the copy also holds DB dumps + volume tarballs it will NEVER read. Use it to refuse up front and to decide whether the success message must disclose uncovered data. Judged from the copy, not the catalog, so a retemplated app is judged by what it actually has |
|
||||
| `Manager.acquireRunning`/`releaseRunning`, `acquireMigrating` | controller/internal/backup/backup.go, controller/internal/stacks/migrate.go | `() error` | Single-flight for long ops | Copy this mutex-flag pattern for any new long-running manager op |
|
||||
|
||||
### Secrets hygiene
|
||||
@@ -82,10 +121,13 @@
|
||||
| `SaveAppConfig` / `LoadAppConfigDecrypted` | controller/internal/stacks/deploy.go | `(stackDir, cfg, encKey, sensitiveVars)` | app.yaml persistence | Encrypts only `SensitiveEnvVars(meta)`; never write app.yaml directly |
|
||||
| `generateValue` / `randomAlphanumeric` | controller/internal/stacks/deploy.go | `(spec "password:N\|hex:N\|base64key:N\|static:v")` | Auto-generated secrets | crypto/rand-backed; reuse the spec grammar |
|
||||
| `Manager.GenerateSecretForField` | controller/internal/stacks/deploy.go | `(stackName, envVar) (string, bool)` | Replacement value for a RESETTABLE secret from its catalog `generate` spec (O4 restore path via `backup.SetSecretGenerator`) | REFUSES `data_key` fields, spec-less and non-secret fields; never log the value |
|
||||
| `reconcileRestoreSecrets` | controller/internal/backup/restore_unit.go | `(nonSecretEnv, recoveredSecrets, secretNames, dataKeyNames)` | Recovery-unit restore env merge | Units are secret-FREE by design; secrets come from live app.yaml |
|
||||
| `reconcileRestoreSecrets` | controller/internal/backup/restore_unit.go | `(nonSecretEnv, unitSecrets, guestSecrets, secretNames, dataKeyNames)` | Recovery-unit restore env merge | **Precedence: UNIT WINS over guest** (the unit's secrets match the data being restored; the guest's are merely newest). Pure — new sources arrive as ARGUMENTS. Fail-closed data-key gate lives here |
|
||||
| `stacks.PortableSecretEnvVars` | controller/internal/stacks/deploy.go | `(meta) []string` | **THE D5 secret boundary**: which secrets may travel on a customer drive | `type: secret` travels, `type: password` NEVER, minus the `nonPortableSecrets` code register. Withholding the password class is what licenses plaintext — do not relax one without the other |
|
||||
| `buildUnitAppYaml` / `readUnitEnv` | controller/internal/backup/{recovery_unit,restore_unit}.go | `(info) []byte` / `(path, portableNames)` | The ONE place the unit's app.yaml is written / split back | Split is driven by the MANIFEST's portable names, never guessed from key names; write 0600; empty `portableNames` = schema-1 unit ⇒ everything is plain config |
|
||||
| `EncryptFile` / `DecryptFile` / `IsEncryptedFAB` | controller/internal/appexport/crypto.go | password-based file crypto | .fab export bundles | scrypt-derived AES+HMAC keys |
|
||||
| `maskRepoURL` | controller/internal/sync/sync.go | `(url) string` | Logging git URLs | Strips embedded credentials |
|
||||
| `metrics.RedactLine` | controller/internal/metrics/redact.go | `(s string) string` | ANY log line shipped off-box (issue context, log tails) | Masks password/passwd/secret/token/api-key/authorization/bearer values + 64-hex; apply BEFORE the line leaves the box — controller-side redaction is authoritative |
|
||||
| `settingsRetrievalPasswordRevealHandler` | controller/internal/web/handlers.go | `POST /settings/retrieval-password/reveal` | **THE PATTERN for showing a secret in the UI** — an XHR that returns only the value | **Never template a secret into a page and hide it with CSS.** `display:none` / `hidden` / `type="password"` stop a browser DRAWING the value; the plaintext is still in the response body, so a `curl` of the page returns it, and it reaches caches, history and any screen-share of the source. R-249 shipped exactly that for two months and was found by it landing in a transcript. The page carries a **boolean** (`HasRetrievalPassword`); the value comes from a POST (CSRF-covered, uncacheable) and the reveal is **logged as an act**. `escrow_handlers.go` states the same rule for R. **Test on the RESPONSE BODY** — a test asserting what the customer *sees* cannot see this class at all. **Both R-254 sites are now FIXED the same way** — `POST /apps/<slug>/initial-credentials/reveal` (re-reads the container, never a cached copy) and `POST /stacks/<name>/auto-field/reveal` (authorised on the field being a `type: secret` auto-field of that stack). **Per-secret, never one generic reveal-any-named-secret endpoint.** The PRE-DEPLOY hidden input is deliberate and untouched — a form must carry what it submits (README §318). Enforced by `scripts/secret_in_markup_gate.py`, whose measured blind spot (a secret under a neutral page-data key) is in its docstring; runtime body-assertion covers 4 of 27 pages — R-255. |
|
||||
|
||||
### Storage registry + mount detection
|
||||
|
||||
@@ -97,9 +139,14 @@
|
||||
| `system.IsMountPoint` / `IsWritable` / `PathsOverlap` | controller/internal/system/mounts_linux.go | `(path) bool` | Mount checks | `_other.go` stubs return permissive values — Linux behavior is the real one |
|
||||
| `system.CheckBackupDestination` | controller/internal/system/mounts_linux.go | `(path) DestinationHealth` | Tier2/offbox target vetting | Detects same-physical-device (`SamePhysicalDevice`) |
|
||||
| `system.ProbeStoragePath` | controller/internal/system/mounts_linux.go | `(path) ProbeResult` | Disconnect detection | — |
|
||||
| `appexport.DiskFree` | controller/internal/appexport/estimate.go | `(path) int64` | Free bytes for space gates (df-based, 0 on any error) | Exported v0.128.0 for the browser-upload gate; test seam = `web.uploadDiskFree` package var |
|
||||
| `stacks.ExportDataMounts` | controller/internal/stacks/delete.go | `(composePath, hddPath) []string` | THE .fab-export mount discovery (v0.130.0 C6B-F1) | Unions `${HDD_PATH}` binds + the `${USERDATA_PATH}` ROOT (single `userdata` entry — basename must round-trip the import's `<HDD_PATH>/<subdir>` mapping; NEVER return per-bind userdata subpaths). Containment-deduped. Backup-side `stackAdapter` deliberately does NOT use it |
|
||||
| `Server.deployedAppsOnPath` | controller/internal/web/netstorage_handlers.go | `(base) []string` | Deployed stacks whose HDD_PATH is base or a subpath | The C6B-F2 share-removal guard; nil-safe on stackMgr |
|
||||
| `planDriveGates` / `Server.ReconcileDriveGates` | controller/internal/web/intermediary.go | pure plan + executor | Drive appear/disappear reactions | `planDriveGates` is PURE (unit-testable); loop at `driveGateLoop` |
|
||||
| `Server.runStorageInit` / `runStorageAttach` | controller/internal/web/storage_handlers.go | wizard pipelines | New-drive enroll / re-attach | Format goes through the agent's two-step confirm (below) |
|
||||
|
||||
| `Server.sharingResolvePath` / `sharingResolveStorageRoot` | controller/internal/web/sharing_handlers.go | `(raw) (string, error)` | THE guard for every customer-supplied SMB share path | resolvePath validates a share TARGET (refuses the drive root); resolveStorageRoot validates the new-folder PARENT (accepts exactly a registered live root). Refusals are UNIFORM (no filesystem oracle). Never add a second deny-list — `stacks.SharingDeniedRoots` derives from `ProtectedHDDPaths` |
|
||||
|
||||
### Agent local-API client (cross-repo edge)
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
@@ -112,32 +159,49 @@
|
||||
| `Client.AddNetStorage/ListNetStorage/RemoveNetStorage` | controller/internal/agentapi/client.go | NAS mounts (A1) | Network storage | Password passes through to agent's 0600 cred file; controller NEVER persists it |
|
||||
| `agentapi.StatusError` | controller/internal/agentapi/client.go | `{Path, Code}` typed non-2xx GET error | Distinguishing HTTP statuses from transport errors (`errors.As`) | NEVER string-match agent error text — the capability probe keys on `Code==404` |
|
||||
| `SupportCache.Supports` / `Client.Supports` | controller/internal/agentapi/features.go | `(ctx, prober, Feature) SupportState` | Agent-capability gate for COUPLED features (route probe, TTL 5m) | 404 ⇒ No; transport/5xx ⇒ Unknown (NEVER refuse on Unknown). New coupled feature = new `featureProbes` row + gate call at the entry point + `MinAgent:` in the CHANGELOG header (publish-train-rules.md). Web layer: `Server.netFeatures` through the `netAgent` seam |
|
||||
| `agentapi.DiskVerdictFor` / `DiskVerdict.Label` / `DegradedAttributes` / `UncorrectableSectors` / `DiskPrior` / `TemperatureFailC` | controller/internal/agentapi/diskverdict.go | `(*SmartSummary, DiskPrior) DiskVerdict` | THE shared disk-health verdict (card chip + hourly check) — v0.169.0, 14-row ladder v0.215.0 | Pure — no clock, no I/O; history arrives as `DiskPrior`. nil/UNKNOWN → `DiskVerdictUnknown` (Nincs adat, NEVER alarms, row 1 is first for that reason). **Never trust `smart_status.passed`**: attrs 187/197/198 carry `thresh: 0`, so it cannot fail on unreadable sectors. A zero `DiskPrior` is the fail-safe (first sighting can only reach Figyelmeztetés). **Four labels, no fifth** — predicted failure is „Hiba". Do NOT recompute the verdict inline anywhere, and do NOT re-literal 60 °C — use `TemperatureFailC` |
|
||||
| `Server.resolveBackupTargetState` / `backupTargetView` | controller/internal/web/backup_target_offer.go | `(ctx)` → state / `*BackupTargetView` (nil = render nothing) | The whole-system backup-target answer: healthy · degraded-never-configured · **TargetAbsent** (configured, drive gone) · unknown | Test seams `Server.tiersFn` + `Server.disksFn` (nil → the real client). **`degradedMessageFor` is the ONE place that decides customer copy** — add a state there, never in a template. `backupTargetView` returns **nil** for healthy AND unknown so a template typo cannot decorate a working box. R-112: this state had NO consumer for two releases; the render is server-side on `backups.html`, and the seam test drives `backupsHandler` and asserts rendered HTML |
|
||||
| `Server.cachedDisks` / `RunDiskHealthCheck` | controller/internal/web/disk_health.go | `(ctx)` | Card fetch (60s TTL) / the hourly degradation check | Card uses the 60s TTL cache (anti-smartctl-storm); the CHECK fetches FRESH (`fetchDisks`). Test seams: `Server.disksFn` (source) + `Server.diskNotifyFn(notify.DiskAlert)` (sink). State is PERSISTED (v0.215.0) — a restart no longer re-baselines |
|
||||
| `diskAlertDecision` / `diskAlertKindFor` / `Server.priorFor` / `Server.cardPriorFor` | controller/internal/web/disk_health_state.go | pure + `(key) agentapi.DiskPrior` | Whether an observation emits, and which message shape | Compares against the **last ALERTED** verdict, not the last observed — that is what collapses a flap to one alert. Re-alert needs doubling **AND** 24h (an AND). **`priorFor` is for the CHECK, `cardPriorFor` for the CARD** — they differ by one observation and mixing them makes the chip read one level more severe than the email |
|
||||
| `diskRecord` / `writeDiskState` / `Server.loadDiskStateLocked` | controller/internal/web/disk_health_state.go | `disk-health-state.json` in `cfg.Paths.DataDir` | Persisted per-disk observation + alert history | Atomic tmp+rename (the `selfupdate.SaveState` shape, copied not imported). Missing file = normal; corrupt = LOG and fall back to no-prior, **never fatal**. Written ONCE per check run. Keyed by `diskKey`. **One record per disk, NOT a sample series** — history is Phase 2/3 in `metrics.MetricsStore` |
|
||||
|
||||
### Notifications / hub sync
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
|---|---|---|---|---|
|
||||
| `Notifier.PushEvent` | controller/internal/notify/notifier.go | `(eventType, severity, message, details)` | Hub events | Async goroutine, 3 attempts/3s backoff. NEW event types MUST be added to hub `allowedEventTypes` or POST /event 400s; hub only emails `warning`/`error` from this path |
|
||||
| `Notifier.PushEvent` | controller/internal/notify/notifier.go | `(eventType, severity, message, details)` | Hub events | Async goroutine, 3 attempts/3s backoff. NEW event types MUST be added to hub `allowedEventTypes` or POST /event 400s. **SEVERITY IS AN EXACT WIRE CONTRACT: `{"info","warning","error","critical"}` and nothing else.** The hub silently COERCES any other string to `"info"` (`hub/internal/api/handler.go`, the ingest severity switch) and `severityNotifies` (`hub/internal/notify/dispatcher.go`) emails only warning/error/critical — so a typo'd severity is stored and delivered to NOBODY, with no error anywhere. **`"warn"` is not a severity.** It shipped on `disk_health_degraded` (fixed v0.215.0, R-328) and is STILL live on `app_start_failed` (R-329) |
|
||||
| `notify.DiskAlert` / `DiskAlertKind` / `DiskAlertKind.Severity()` | controller/internal/notify/notifier.go | `NotifyDiskHealthDegraded(DiskAlert)` | The disk-health alert payload + its five Hungarian message shapes | The notifier owns customer copy — pass a `DiskAlert`, never a pre-formatted string, or Hungarian scatters across packages. `Severity()` is the ONE mapping kind→hub severity and is exported so any package can assert the contract instead of duplicating the literal |
|
||||
| `Notifier.Notify*` convenience methods | controller/internal/notify/notifier.go | typed wrappers (backup/DB/storage/channel/DR…) | Standard events | Add a typed wrapper rather than raw PushEvent calls |
|
||||
| `report.BuildReport` / `Pusher.Push` | controller/internal/report/builder.go + pusher.go | periodic hub report | Box→hub reporting | ACK carries `config_version` → `ConfigRefresher.Reconcile` |
|
||||
| `report.Trigger` (`NewTrigger`/`Fire`/`Run`) | controller/internal/report/trigger.go | `Fire()` after a hub-relevant user action | THE out-of-cycle report push (v0.139.0) — fire via `api.Router.reportPushNow` / `web.Server.reportTriggerNow`, both nil-safe | Coalesce-and-eventually-fire (trailing edge; quiet 2s, min spacing 15s). NEVER add retries (Pusher owns them); NEVER reuse the `internal/sync` REFUSE-debounce for hub pushes (a refused fire loses the update until the next cycle). Fire only AFTER a successful local commit |
|
||||
| `report.SetPendingLogTails` + `buildLogTailsSection` | controller/internal/report/logtail.go | ACK `log_tail_requests` → next report `log_tails` | THE pull-based ACK-flag pattern (hub asks, controller pushes next cycle) — copy for any new hub→box request | Consume-once drain at BuildReport; failed push re-arms from the hub's still-pending request; NEVER add a hub→controller push channel |
|
||||
| `metrics.FetchContainerLogTail` | controller/internal/metrics/logscanner.go | `(name, tailLines) (string, error)` | Raw per-container `docker logs --tail=N` | 15s timeout; caller caps/redacts (capTailLines) |
|
||||
| `ConfigRefresher.Reconcile` | controller/internal/report/config_refresh.go | `(ackVersion int)` | Pull-based config refresh | Re-pulls controller.yaml (re-merging local_api), then graceful self-restart; first-run = baseline, no restart |
|
||||
| `offsiteapply.SettleProvider` / `SettleFunc` / `Bridge.AwaitSettle` / `ReconcileWhenSettled` (R-71a, v0.162.0) | controller/internal/offsiteapply/offsiteapply.go + seams.go | `SettleState() (version, floor string, updateRunning, floorKnown bool)` | THE settle-gate: defers the offsite one-time-password consume past a managed day-0 floor-update (the F10 race). Wire the `SettleFunc` adapter over `updater.GetFloor()`/`IsUpdateRunning()` — **the updater's knowledge is the ONE floor source; never fetch the floor a second way**. Gate ONLY the bridge goroutine, and only when an updater exists (nil `Settle` = reconcile immediately). Bounds `settlePoll`/`settleFloorSubBound`/`settleOverallBound`; the floor is in-memory (report-ACK-derived, ~5–10 s), NOT persisted → unknown until the first ACK on any restart. Inject `Now`/`Sleep` in tests (no real sleeps). B′: at/above-floor GOes on the first poll, zero wait. Do NOT touch the consume/persist order or the 404 contract — ordering only |
|
||||
| `bootstrap.MaybeIngest` / `RefreshConfig` | controller/internal/bootstrap/bootstrap.go | bootstrap.json → controller.yaml | Day-0 + refresh | Overwrites controller.yaml, NEVER settings.json |
|
||||
| `api.GracefulSelfRestart` | controller/internal/api/selfrestart.go | `(logger)` | Controller self-restart | Detached exit; bootstrap unit re-runs the image |
|
||||
| `Settings.AddPendingEvent/DrainPendingEvents` | controller/internal/settings/settings.go | offline event queue | Events while hub unreachable | — |
|
||||
| `Manager.SetUnitNotify` + `UnitSpace` (R-158/R-167, v0.191.0) | controller/internal/backup/recovery_unit.go | `(func(stack string, err error, *UnitSpace))` | THE per-app Tier-1 recovery-unit capture failure alert — fires PER APP from `captureAllRecoveryUnits`, loop continues | **OPERATOR-TIER** (`recovery_unit_capture_failed`, in the hub's `operatorOnlyEvents`). **NEVER route it to `backup_failed`** — that type is in `DefaultEnabledEvents` and carries Hungarian copy, so it emails the CUSTOMER about a failure they cannot act on (D-c; R-158's own proposal said `backup_failed` and D-c overrides it). `UnitSpace` is **nil when the target filesystem is unreadable** and renders as *"unavailable"*, never as zeros — "0 GB free" and "we could not look" are opposite diagnoses. No controller-side cooldown: the hub owns it |
|
||||
| `Manager.beginRunSummary` / `noteFailure` / `noteAttempted` / `emitRunSummary` / `SetRunSummaryNotify` (R-182, v0.194.0) | controller/internal/backup/runsummary.go | `(kind, runID) func()` / `(app, leg, reason)` / `(RunSummary)` | **THE per-run operator digest.** One `backup_run_failures` event at the end of a run listing every failed app, its leg and its reason — emitted ONLY when something failed | **The RECORD and the NOTIFICATION are different things and must stay so.** The per-app `recovery_unit_capture_failed` event is the record (hub routes it *record-only*, stored + logged every time); this digest is the notification. Before R-182 one event was both, and did neither: nine arrived, two were mailed, seven vanished before `LogNotification`. **Lifetime is `admissionSet`'s exactly** — absent collector means "no run in flight", never a stale answer. **A refusal is noted ONCE, inside `admitApp` where the verdict is taken**, not at the three legs that consult it: R-181's one-verdict-covers-all-three contract makes per-leg noting produce "2 of 1 apps failed". **Deliberate skips (disconnected / decommissioned) must NEVER be noted** — they have their own alert and a nightly digest about an unplugged drive is an ignored digest. **A clean run emits NOTHING**; silence is safe only because the hub's deadline check (`monitor/deadline.go:396,417`) raises a missed backup from report freshness independently — if that is ever weakened this design loses its footing. **`run_id` is unique per real run** (so the hub's 1-h cooldown cannot collapse a manual run into the nightly one) and **deliberately EMPTY on the periodic refresh sweep**, which must stay under that cooldown or a polled status page becomes a mail flood |
|
||||
| `Manager.admitApp` / `beginAdmissionRun` / `decideAdmission` / `estimatedWriteBytes` (R-181, v0.193.0) | controller/internal/backup/admission.go | `(stackName) bool` / `() func()` | **THE reserve gate. Call it before ANY per-app backup write** — one verdict per app per run, covering the DB dump, the volume dump and the unit capture (all three write under one per-app root) | **Decided LAZILY at the app's first write, never once at run start** — app A's dump can put app B under the reserve, so a run-start verdict reads a disk that no longer exists. **Never re-decided between an app's own legs**: that is exactly the split R-181 closed (bulk written, capture refused). **Reset per run** via the closer `beginAdmissionRun` returns. **Must sit ahead of `DumpAppVolumesSafe`**, which stops the stack as its first act — a refusal decided inside it has already bounced the app. Fires **exactly one** `unitNotify` per refused app per run. Nil admission set (periodic status refresh) → decides fresh, which is still once per app per sweep. Wiring pinned by an **AST walk** in `TestAdmission_IsWiredIntoEveryProductionWriteLeg`, not `strings.Contains` |
|
||||
| `Manager.floorVerdict` + `FloorUsedPercent`/`FloorFreeGiB` / `ErrCaptureFloor` / `floorReason` (R-165 B2 v0.192.0, size term R-181 v0.193.0) | controller/internal/backup/recovery_unit.go | `(*UnitSpace, estGiB float64) (*UnitSpace, floorReason)` | The pure two-question predicate behind `admitApp`: is the filesystem already below the reserve (`floorHeadroom`), and would THIS app's write take it below (`floorSize`)? | **Headroom is about the FILESYSTEM, never a per-unit cap** — a size cap is R-163 rebuilt inside one volume; the size term bounds the *delta*, not the unit. **REFUSES, never deletes:** nothing here is generational (a unit is one fixed path per app, a DB dump one fixed name), so pruning could only destroy a DIFFERENT app's only local copy — **never repurpose `pruneStalePrimaryDirs`**, which removes ORPHANED dirs from an app that moved drives and has no notion of age. Two terms (97% / 1 GiB) in `fillwatch`'s shape, deliberately BEYOND its critical band (95% / 2 GiB) so the customer is always warned first — pinned by `TestFloorSitsBelowTheCriticalWarningBand`. **`estGiB == 0` degrades to headroom-only on purpose** — refusing an app with no history makes the FIRST backup the one that can never happen. A nil reading neither refuses nor warns (§8.4). Inject `unitSpaceFn` in tests rather than manufacturing occupancy on a real disk |
|
||||
| `fillwatch.Watcher` (`New`/`SetNotify`/`Check`) (R-167, v0.191.0) | controller/internal/fillwatch/fillwatch.go | `(statePath, logger, targetsFn, usageFn)` → `Check() error` | THE customer fill warning — warns BEFORE a filesystem fills, per FILESYSTEM (never per app: one full disk holding ten apps would fire ten times) | Emits the **pre-existing** `disk_warning`/`disk_critical` pair, which was allowlisted + copy'd + default-enabled with **no producer in any repo** until now — do NOT mint a new type beside it. **Two threshold terms, whichever trips first** (85% / 5 GiB; critical 95% / 2 GiB) because a percentage alone lies at both ends of this fleet's size range. **Edge-triggered on ESCALATION ONLY**, state persisted; de-escalation is silent and re-arms. Hysteresis dead zone between clear (75% / 7 GiB) and warn — pinned by `TestThresholdsKeepTheirHysteresisGap`. **A nil usage read is NEVER a warning** (§8.4). The hub has **no `customerMessages` entry** for either type on purpose — an entry would override the dynamic message and discard the drive label + free space |
|
||||
|
||||
### Scheduler / time / UI
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
|---|---|---|---|---|
|
||||
| `Scheduler.Every` / `Daily` | controller/internal/scheduler/scheduler.go | `(name, interval/"HH:MM", fn)` | ALL background jobs | Daily is Europe/Budapest, DST-safe (`nextDailyRun` avoids Add(24h)); register in main.go block (§5) |
|
||||
| `getBudapestLocation` | controller/internal/scheduler/scheduler.go | `() *time.Location` | Local-time math | web has its own `getTimezone` (§6) |
|
||||
| `Scheduler.UpdateDaily` | controller/internal/scheduler/scheduler.go | `(name, "HH:MM") bool` | Retime a daily job at runtime (no restart) | Per-job buffered `resched` chan + select case in `runDailyJob`; false (WARN) on invalid time / unknown-or-non-daily name; read `Schedule` under the mutex in the loop |
|
||||
| `backupwindow.*` (LegTimes / GateWindow / EffectiveWindow / ParseHHMM / FmtHHMM / Valid) | controller/internal/backupwindow/backupwindow.go | pure `string`↔`int` | Backup-window arithmetic (v0.168.0) | Offsets (W+60m/W+105m, gate W+2h..W+6h) are CONSTANTS — derived, never stored; wrap-safe modulo 1440; `EffectiveWindow(settings, yaml)` = settings>yaml>"02:30" |
|
||||
| `getBudapestLocation` | controller/internal/scheduler/scheduler.go | `() *time.Location` | Local-time math | web has its own `getTimezone` (§6); quiesce has its own `budapestLocation` (window gate) — 3rd copy, see §6 |
|
||||
| `Server.templateFuncMap` | controller/internal/web/funcmap.go | template.FuncMap | ALL template functions | `stateColor` outputs v2 suffixes `run/progress/warn/neutral/off`; stopped = NEUTRAL not red (operator-approved); `stateLabel` copy is frozen byte-identical (unit-tested) |
|
||||
| `timeAgoStr` | controller/internal/web/funcmap.go | `(s RFC3339 string) string` | Ago-format for STRING timestamps | Exists because `timeAgo(time.Time)` 500'd on strings (v0.93 bug) |
|
||||
| `Server.baseData` / `executeTemplate` | controller/internal/web/handlers.go + server.go | page-data plumbing | New pages | baseData injects nav/alerts/version; templates must pass `controller/scripts/template_id_gate.py` + `controller/scripts/emoji_gate.py` |
|
||||
| `Server.RequireAuth` / `CsrfProtect` / `csrfField` | controller/internal/web/auth.go + csrf.go | middleware | Any new authed route/form | csrfField emits the hidden input; setup wizard has its OWN csrf (§6) |
|
||||
| `LogBuffer` | controller/internal/web/logbuffer.go | ring buffer io.Writer | In-memory log capture for debug UI | — |
|
||||
| `LogBuffer` + `Lines(maxBytes)` | controller/internal/web/logbuffer.go | ring buffer io.Writer | In-memory log capture for the debug UI + the report `controller_log_tail` source | v0.116.0: ALWAYS constructed (any logging.level) — the logger is `MultiWriter(LevelFilterWriter(stdout, level), ring)`; `Lines` drops OLDEST to honor the byte budget |
|
||||
| `logx.Debugf/Infof/Warnf/Errorf` | controller/internal/logx/logx.go | `(l *log.Logger, format, args…)` | ALL NEW leveled log lines (the v0.116.0 sweep standard) | routing is the WRITER's job — Debugf always reaches the ring, stdout filters; nil logger = no-op; caller-attributed (Output calldepth 3) |
|
||||
| `web.LevelFilterWriter` | controller/internal/web/levelfilter.go | `NewLevelFilterWriter(w, minLevel)` | stdout leveling under the always-on ring | untagged lines parse INFO; always reports full length written |
|
||||
| `monitor.RunHealthCheck` / `EffectiveProtected` | controller/internal/monitor/healthcheck.go | system health report | Health + protected-container list | — |
|
||||
| `util.TruncateStr` | controller/internal/util/strings.go | `(s, maxLen) string` | Rune-safe truncation | The intended shared helper; stacks still uses its byte-based twin (§6) |
|
||||
|
||||
@@ -152,7 +216,10 @@
|
||||
| Settings mutator | controller/internal/settings/settings.go (any Set*/Add*) | Lock → mutate → `s.save()`; getters return copies; never expose internal slices |
|
||||
| Channel-health checker w/ born-down alerting | controller/internal/channelhealth/checker.go | classify → debounce N≥2 → `alerted` flag re-armed on reason change (F2) |
|
||||
| Platform split | controller/internal/system/mounts_linux.go + mounts_other.go | `_linux.go`/`_other.go` twins; other = permissive no-op stubs for dev on Windows |
|
||||
| Debounced trigger + status | controller/internal/sync/sync.go | `TriggerSync` 30s debounce, `Status()` snapshot struct, post-sync hook fan-out |
|
||||
| Debounced trigger + status (REFUSE-style — a too-soon fire is refused/lost) | controller/internal/sync/sync.go | `TriggerSync` 30s debounce, `Status()` snapshot struct, post-sync hook fan-out |
|
||||
| Coalescing trigger (trailing edge — a burst collapses but the LAST state always fires) | controller/internal/report/trigger.go | buffered-1 chan + non-blocking `Fire()` + single worker (quiet window → drain → min-interval → fire once); shape from hub `wgsync/reconciler.go` |
|
||||
| Detached job + status poll (single-flight, phase strings) | controller/internal/web/storage_init_job.go | acquire/release/set/**deep-copied** snapshot; phases mapped to Hungarian in the template; 1–3 s poll; terminal state **PROBED, not inferred**. Clones: `netstorage_job.go`, `samba_ensure_job.go` (v0.147.0). **Five of these now exist and agree on nothing — R-45 will unify them; prefer extending an existing one over a sixth** |
|
||||
| Streaming subprocess progress | controller/internal/backup/offbox_progress.go | `offboxStreamRunner` seam (stdout scanned line-by-line, stderr buffered, output tail-bounded) + a PURE line parser + a mutex-guarded published snapshot. Traps it encodes: a source reporting nothing is **normal** (restic sends 0 bytes for a whole incremental run) and the progress source may only update on unit completion — degrade bytes → files → current item + elapsed, never fake a percentage |
|
||||
| Post-start async verification | controller/internal/stacks/manager.go `logPostStartStatus` | goroutine + sleep, INFO log, never blocks/fails the operation |
|
||||
| Startup wiring order | controller/cmd/controller/main.go | init-only setters (`SetStackProvider` M2 contract: exactly once, before scheduler/HTTP), scheduler registration block |
|
||||
|
||||
@@ -167,6 +234,7 @@
|
||||
| `backup.Manager.DumpAppVolumes` on a running DB app | Inconsistent tar of live DB volume | `DumpAppVolumesSafe` (stop → dump → restart, both errors surfaced) |
|
||||
| `stacks.Manager.execCommand` / `composeExecCustomEnv` for NEW long-running calls | No context/timeout — a hung docker CLI blocks forever | `exec.CommandContext` + explicit timeout (copy `rsyncCopy` or appexport `composeExecEnv`) |
|
||||
| `config.LoadPermissive` | Skips validation — setup-mode only (customer.id/domain may be unset) | `config.Load` everywhere else |
|
||||
| `ExportDataMounts` / `ParseComposeHDDMounts` as **backup-classification** input | `ExportDataMounts` unions the `${USERDATA_PATH}` ROOT (export-capture logic, not per-bind); `ParseComposeHDDMounts` resolves absolutes AND drops the `:ro` flag — classification needs `${VAR}`-relative paths + read-only awareness | `ParseComposeClassifiableBinds` (controller/internal/stacks/classify_binds.go) |
|
||||
| `docker compose restart` (any wrapper) | Does not pick up new images or env | `RedeployFromEnv` / composeExec `up -d` |
|
||||
|
||||
## 4. Seams & interfaces (testing + cross-repo)
|
||||
@@ -175,7 +243,36 @@
|
||||
|---|---|---|---|
|
||||
| `diskAgent` | controller/internal/web/storage_handlers.go | `*agentapi.Client` | `mockAgent` in controller/internal/web/storage_handlers_test.go |
|
||||
| `netAgent` + `Server.netAgentFn/netProbeFn/netListFn` | controller/internal/web/netstorage_job.go (+ server.go fields) | `*agentapi.Client` / `runNetProbe` (linux re-exec) / `agent.ListNetStorage` | `fakeNetAgent` + fn injections in controller/internal/web/netstorage_job_test.go — the NAS add orchestration never shells/TLS-dials in tests |
|
||||
| `Server.agentLogsFn` (func seam) | controller/internal/web/server.go | nil → `agentClient().DebugLogs` (agent GET /debug/logs) | injected in controller/internal/web/observability_test.go (incl. the pre-0.83 typed-404 notice path) |
|
||||
| `escrowAgent` + `Server.escrowAgentFn/escrowStageFn/escrowStaleFn` | controller/internal/web/escrow_handlers.go (+ server.go fields) | `*agentapi.Client` / `PushOffboxPasswordForEscrow` / `report.EscrowAutoConfirmer.StaleBlob` (SetEscrowStale) | `fakeEscrowAgent` + fn injections in escrow_wizard_test.go — call-ORDER assertions (stage BEFORE trigger) + agent-never-called gates. The claim leg is the ONLY surface R crosses: no-store, never logged, never templated |
|
||||
| `offboxCeremonyWaitState` + `escrowCeremonyGraceWindow` | controller/internal/web/handlers.go | pure pick: (awaiting, timedOut) from `OffboxTarget.{EscrowState,CeremonyCompletedAt}` — the v0.138.0 "megerősítésre vár" card. Stamp SET on claim (escrow_handlers.go), CLEARED on the flip (main.go Flip + offbox_handlers.go manual confirm) | escrow_wait_state_test.go truth table (escrowed/unstamped/unparseable → plain CTA; boundary via `>=`) |
|
||||
| `Manager.sambaUpFn` / `sambaPasswdFn` / `sambaRunFn` / `sambaAddrFn` (func seams) | controller/internal/stacks/manager.go (fields) + samba.go | nil → `composeUp` / `docker exec smbpasswd` (STDIN) / `containerRunning("felhom-samba")` / `docker exec felhom-samba ip -4 -o addr show eth0` | injected in controller/internal/stacks/samba_test.go — the idempotency test asserts the up-seam is called **zero** times when config is unchanged; the passwd seam means no unit test ever handles a real secret or touches docker. **`sambaRunFn` has an EXPORTED setter (`SetSambaRunProbe`)** — internal/web's status-contract tests need a live-container world from another package. `sambaAddrFn` backs `SambaLANAddress()` (v0.151.0); its parse is separately pinned in samba_lanaddr_test.go and it returns "" on any failure — the page omits a line rather than printing a wrong address |
|
||||
| `Manager.SambaLANAddress()` | controller/internal/stacks/samba.go | `() string` — the guest's LAN IPv4 for the Megosztás connect card (v0.151.0, S-2) | Read from the SAMBA container's netns (`network_mode: host`), never `net.InterfaceAddrs()` — the controller is on a docker BRIDGE and would answer 172.x (the same trap `setup.DetectLocalIPs` needs `HOST_IP` for). **NEVER cache/persist it** — the guest holds it by DHCP (S-5); callers re-derive per render. `""` = omit the line |
|
||||
| `Server.sambaAddrFn` (func seam) | controller/internal/web/server.go (field) + sharing_handlers.go `sambaLANAddress()` | nil → `stackMgr.SambaLANAddress()` | The web-side half of the connect card. Tests inject a COUNTED fn — the fresh-per-render assertion is what stops anyone memoizing a DHCP lease |
|
||||
| `Manager.guestNetExecFn` (func seam) + `GuestGateway()` / `GuestNetSnapshot()` | controller/internal/stacks/manager.go (field) + guestnet.go | nil → `docker exec felhom-samba <args>` — ONE seam for all R-66 guest-netns reads (route/link/addr/resolv.conf); tests script canned outputs per argv | guestnet_test.go. **The netns door rule:** the controller's OWN netns is the docker bridge, so any in-process read (`net.Interfaces`, `/proc/net/route`, its own `/etc/resolv.conf` = 127.0.0.11) is the S-2 wrong answer — guest-net reads MUST go through the samba (`network_mode: host`) exec door. Megosztás off ⇒ door closed ⇒ "" / per-item error strings; NEVER substitute an in-process value. Same S-5 law as SambaLANAddress: live per render, never cached/persisted. Parsers (`parseDefaultRoute`, `parseGuestInterfaces`, `parseResolvConf`) are pure + separately pinned |
|
||||
| `buildFileBrowserPaths` + `fbPathDeps` (R-67, v0.160.0) | controller/internal/web/handlers.go | pure assembly of one FileBrowser sync pass: (mount lines, config source paths) from the registry, with per-kind gates | filebrowser_network_test.go. **Two storage classes, two DIFFERENT gates:** drives keep the drive-absent gate + userdata scoping + skeleton (byte-identical to pre-R-67 — tested); network shares bind the share ROOT `:rslave` with the STUB gate instead (`classifyFSPath`; stub ⇒ excluded from mounts AND sources — an exposed stub swallows uploads the real mount later shadows; idle autofs / unknown ⇒ include, fail open). NEVER call `EnsureUserdataSkeleton` toward a network path (red-proven); never force-wake an idle trigger in the sync (doctrine) |
|
||||
| `Settings.RefuseAsAppNamespace` (R-108, v0.187.0) | controller/internal/settings/settings.go | `(path) (refuse bool, hungarianReason string)` — may an app's DATA NAMESPACE live here? | **THE single predicate for every placement surface** (deploy POST `api/router.go`, per-app migrate list + `handleStorageMigrateApp`, `handleStorageDecommission` mode=migrate TARGET). **Network storage is refused** because an app's namespace root IS its backup root (`namespaceRoot` returns a non-system drive path as-is → `<HDD_PATH>/backups/primary/<stack>/`), and on a share that lands inside FileBrowser's share-ROOT `download:true` bind — which CANNOT be narrowed (R-67 `:rslave` = automount wake; and apps on a share store at `<share>/<app>`, so there is no `userdata/` to scope to and creating one would write Felhom convention onto a customer's NAS). **DISTINCT from `refuseNetworkLifecycle`** — that asks "may a DRIVE lifecycle op run on this path" and is applied to the op's SUBJECT; this asks "may an app live here" and is applied to a placement TARGET. Migrate needs BOTH. **FAILS CLOSED:** `/mnt/felhom-drives` holds both kinds, so a path prefix cannot classify — `Kind` exists only on a REGISTERED path, therefore an unregistered path under that root is un-classifiable and REFUSES. Empty path = SSD-resident = allowed; nil receiver refuses. network_app_namespace_test.go, 4 red-proofs |
|
||||
| `Server.guestGatewayFn` / `guestNetFn` (func seams) | controller/internal/web/server.go (fields) + sharing_handlers.go accessors | nil → `stackMgr.GuestGateway` / `stackMgr.GuestNetSnapshot` | network_card_test.go — the counted-fn freshness test (2 renders ⇒ 2 resolves) is what stops anyone memoizing a DHCP lease; the Hálózati név row is gated on `smb.Enabled` (red-proven: gate dropped ⇒ \\FELHOM rendered while samba is down) |
|
||||
| `sambaEnsureState.consumeIfRunning()` | controller/internal/web/samba_ensure_job.go | serve-once `snapshot()` for terminal `running` only | `/sharing/status` carries a job EDGE (`phase`) and a service LEVEL (`running`) in one envelope — never let a level reach the phase channel, and never re-serve a consumed edge: the client answers `phase=="running"` with `location.reload()`, so both mistakes produce an infinite page reload (S-1/S-4, DIAG-sharing-2026-07-20.md). `failed`/`needs_password`/in-flight are NOT consumed |
|
||||
| `infra.SambaHostInterface` | controller/internal/infra/samba.go | the guest LAN nic name (`eth0`) | Single source for smb.conf's `interfaces =`, the container's `FELHOM_IFACE`, and the LAN-address read — if they name different nics, the service and the address the page prints drift apart |
|
||||
| `Manager.sambaImgFn` (func seam) | controller/internal/stacks/manager.go (field) + samba.go | nil → `docker image inspect <infra.SambaImage>` | drives the 4b card's pulling-vs-starting decision, which MUST be taken before `compose up` (afterwards the image is always present) |
|
||||
| `Manager.offboxStreamRunner` + `SetOffboxStreamRunner` | controller/internal/backup/offbox_progress.go | nil → `defaultOffboxStreamRunner` (real `restic`, stdout scanned live) | streaming sibling of `offboxRunner`; fakes emit canned `--json` status lines in offbox_progress_test.go, so the whole progress path runs with no restic, network or repo |
|
||||
| `Manager.offsitePreDumpFn` + `SetOffsitePreDumpFn` (R-44, v0.148.0) | controller/internal/backup/offbox_reconstitute.go (seam) + offbox.go (call site) | nil → `runDBDumpsInternal` under the SAME running flag | THE dumps-before-capture ordering seam. Extracted so the order is observable without Docker/restic — an ordering guarantee no test can see is one refactor from silently reverting to the DIAG-immich-restore-2026-07-19 behaviour. Red-proof: moving the capture first yields `[capture dump]` |
|
||||
| `Manager.offboxFullPlaceCopier` + `SetOffboxFullPlaceCopier` (R-43) | controller/internal/backup/offbox_reconstitute.go | nil → `rsyncRestoreOverwrite` (`-a --itemize-changes`; **no** `--ignore-existing`, **no** `--delete`) | **TRAP: do NOT reuse `offboxPlaceCopier` here.** The two copiers have OPPOSITE semantics for an existing file — `--ignore-existing` is exactly what a full restore must not do, and conflating them is how a missing-only merge came to be labelled a restore. Never `rsyncMirror` (`--delete`) in any restore direction |
|
||||
| `Manager.safetyDumpFn` + `SetSafetyDumpFn` (R-43) | controller/internal/backup/offbox_reconstitute.go | nil → `DumpOne` | the pre-restore undo. Invariant: the `pre-restore-`-prefixed dump must be verified ON DISK before anything is stopped/overwritten/replayed; failure ⇒ refuse with zero changes. Red-proof requires removing BOTH guards (the `err != nil` return and the `os.Stat`) — removing one leaves the other holding |
|
||||
| `reimportDBDumpsFrom(ctx, stack, dumpDir)` | controller/internal/backup/restore_db.go | explicit-dir sibling of `reimportDBDumps` (which passes `AppDBDumpPath`) | offsite reconstitution replays from the SCRATCH unit: the live unit is deliberately never overwritten, so replaying from it would replay the current DB over itself and restore nothing |
|
||||
| The DB-only replay window (R-47, v0.153.0) | controller/internal/backup/{offbox_reconstitute,restore_unit}.go | both restore paths: stop → place/volumes → `StartStackServices(dbServices)` → replay → `StartStack` (full) | **THE ordering invariant.** Replaying while the whole stack is up lets the app's own schema management race the dump — measured at 2 s on 2026-07-19 (H4), replay aborted `already exists`. Fail-closed: a dump with NO identifiable DB service refuses BEFORE the first mutation. Every exit from the window (replay error, DB-only start error) MUST still do a best-effort full start, or a failed restore becomes an outage. `hasReplayableDump` excludes `pre-restore-` safety dumps — counting them would arm the window for an app with nothing to replay |
|
||||
| `Manager.OffsiteScratchPair` / `OffsitePairInfo` | controller/internal/backup/offbox_reconstitute.go | reads the restored scratch unit's manifest (`offsite_run_id` / `dumps_at`) + the R-44 sniff | the confirm-dialog honesty surface. All warn-level: a pre-v0.148 (unstamped) pair and an empty-looking dump are SURFACED, never blocked — a false positive that refused a legitimate restore would be worse than the skew |
|
||||
| `appbackup.DumpValidation.LooksEmpty` (R-44 sniff) | controller/internal/appbackup/dbdump.go | computed in ValidateDump's existing single pass; `userTableNames` is EXACT-match | size and table count are both useless as emptiness heuristics (the 2026-07-19 dump: 52MB, 60+ tables, zero users — all geodata). **TRAP: never widen to a substring match on "user"** — it would flag `user_metadata` / `album_user` / `user_audit` on every healthy single-user box. A row wider than the read buffer still counts as a row |
|
||||
| `Manager.execFn` (func seam) + `restartPolicyLookup` / `inspectRestartPolicyFn` (R-51, v0.156.0) | controller/internal/stacks/manager.go | nil → real `exec.Command` / `docker inspect -f {{.HostConfig.RestartPolicy.Name}}` | `scriptedDocker` in controller/internal/stacks/degraded_test.go drives the WHOLE production path (docker ps → aggregateState → docker inspect) — an aggregateState-only test proves the function, not the caller. Policy answers are cached per container+state and pruned to the live `docker ps` set; a FAILED inspect is deliberately never cached (a hiccup must not pin a container to "unknown") and reads as SUPERVISED, i.e. fail-closed — the opposite of `IsDownState`'s fail-open, because there the state is ambiguous while here a member is known dead |
|
||||
| `bootrecon.StackProvider` (R-52, v0.156.0) | controller/internal/bootrecon/bootrecon.go | `*stacks.Manager` (GetStacks/StartStack/RefreshStatus) | `fakeStacks` counts StartStack per app; the load-bearing assertion is the NEGATIVE — a zero-container stack (a UI Stop = `compose down` = containers removed) must record **0** starts, while a boot orphan (containers present, Exited) records exactly 1. `Reconciler.sleep` is injected so the 30 s gap costs nothing |
|
||||
| `bootReconcileFn` + `runBootReconcile` (package-main seam, v0.156.0) | controller/cmd/controller/main.go | `bootrecon.New(mgr, logger).Run` | controller/cmd/controller/bootrecon_wiring_test.go. **The wiring itself is asserted by an AST walk** over `func main()`, not a `strings.Contains` — the substring version passed its own red-proof because a commented-out call still contains the string. Comments are not callers |
|
||||
| `classifyRunStates` (pure fix-3 derivation, v0.164.0) | controller/cmd/controller/main.go | `([]stacks.Stack, quiesced, failedRestart map[string]bool, now time.Time)` → `(dead []web.DeadApp, states []notify.AppRunState)` | classify_runstates_test.go. **THE single fix-3 rule: down = `(IsDownState(st.State) || st.CrashLooping(now)) && !userStopped && !quiesced`.** C9-F2 (v0.183.0) added the crash-loop term: `restarting` is NOT in `IsDownState` and must not be — adding it alarms on every deploy and update fleet-wide — so a SUSTAINED restarting run (`stacks.crashLoopAfter` = 5 m, above the 120 s deploy timeout, Mealie's 60 s start_period AND R-97b's 180 s grace) becomes down instead. `now` is injected so the threshold is a testable contract. A deliberate UI stop (`compose down` → zero containers → StateStopped, I1) must not alarm — banner OR email — while faults (Exited/Degraded) alarm byte-identically; I2 (P2 census: all catalog services `unless-stopped`) is why a crash never rests at stopped. **Do NOT touch `IsDownState`** (other callers rely on stopped=down) and do NOT filter in `buildDeadAppAlerts`/`NotifyAppStartFailures` — one derivation point. If I1 or I2 changes, revisit the suppression |
|
||||
| `report.SetPendingControllerLog` / `SetControllerLogSource` | controller/internal/report/selftail.go | ACK-armed consume-once self-log pull (the logtail.go shape) | selftail_test.go; source = `logBuffer.Lines`, wired once in main.go |
|
||||
| `util.ParseVersion` / `util.Version.Compare` | controller/internal/util/version.go | THE one semver comparator (house rule: never a second) — selfupdate aliases it; agentapi's MinAgent comparison uses it | rejects pre-release/dev/latest (callers fall back, never trust); numeric compare (0.100 > 0.81) |
|
||||
| `agentapi.AgentVersionReporter` + `featureMinAgent` | controller/internal/agentapi/features.go | version-first Supports (v0.82.0 header channel); probe = fallback for header-less agents | a coupled feature adds BOTH a featureProbes row AND a featureMinAgent row; v0.116.0: `SupportsWithSource` also reports HOW the verdict was reached (version/probe-cache/probe) for the gate log line |
|
||||
| `netProbeReadBack` (package var) | controller/internal/web/netprobe.go | `os.ReadFile` | overridden in TestNetProbeChild (nonce-tamper + cleanup-fail rows); package var because the child is a RE-EXEC'd process in production |
|
||||
| `system.ClassifyPathFS(Timeout)` + `netProbeFSClass` / `Server.classifyFSPath` / `Router.classifyFSPath` | controller/internal/system/fsclass*.go (+ web/netprobe.go, web/server.go, api/router.go seams) | statfs f_type → network/autofs/stub/unknown in THIS namespace (RCA fix 2) | idle autofs = HEALTHY, never force-mount; unknown = fail OPEN; seams injected in netprobe_stub_test.go / networkstub_test.go / deploygate_test.go |
|
||||
| `quiesce.Backend` / `quiesce.Stacks` | controller/internal/quiesce/quiesce.go | adapter over `*agentapi.Client` / `*stacks.Manager` | `fakeBackend`/`fakeStacks` in controller/internal/quiesce/quiesce_test.go |
|
||||
| `channelhealth.Probe` (func) + `Sink` | controller/internal/channelhealth/checker.go | `Server.ProbeAgentChannel` / notifier adapter | `fakeSink` in controller/internal/channelhealth/checker_test.go |
|
||||
| `appbackup.StackDataProvider` | controller/internal/appbackup/appdata.go | `*stacks.Manager` (via `backup.SetStackProvider`) | `fakeRecoveryProvider` in controller/internal/backup/recovery_unit_test.go |
|
||||
@@ -188,6 +285,8 @@
|
||||
| `dumpVolumesSafe` (func seam) | controller/internal/backup/backup.go | nil → real `DumpAppVolumesSafe` | injected in controller/internal/backup/volume_dumps_test.go (gating tests without Docker) |
|
||||
| `generateSecret` (func seam) | controller/internal/backup/backup.go | `stacks.Manager.GenerateSecretForField` via `SetSecretGenerator` (main.go) | injected in controller/internal/backup/restore_secrets_gen_test.go |
|
||||
| `restoreFilesCopier` (func seam) | controller/internal/backup/backup.go | nil → real `rsyncRestoreMissing` | injected in controller/internal/backup/tier2_restore_test.go (orchestration without rsync) |
|
||||
| `tier2Mirror` (func seam) | controller/internal/backup/backup.go | nil → real `rsyncMirror` | both RunTier2 rsync legs; injected in controller/internal/backup/tier2_test.go (resolve→mirror without rsync) |
|
||||
| `migSeams.resolveNames` (func seam) | controller/internal/stacks/migrate.go | nil → real `ResolveAppDataDirNames` (compose-derived) | injected in controller/internal/stacks/migrate_fs3_test.go (F-S3 appdata dir-name resolution) |
|
||||
|
||||
Cross-repo edges:
|
||||
- `controller/internal/agentapi/client.go` ↔ **felhom-agent** local API (`/storage`, `/disks*`, `/backup*`, `/netstorage*`, `/guest/*`): pinned leaf SHA-256 + per-guest bearer token from bootstrap.json.
|
||||
@@ -200,7 +299,12 @@ Cross-repo edges:
|
||||
- **New REST endpoint**: path dispatch in `Router.ServeHTTP` (controller/internal/api/router.go); use `writeJSON` + `limitBody`.
|
||||
- **New background job**: `sched.Every`/`sched.Daily` registration block in controller/cmd/controller/main.go.
|
||||
- **New template function**: `Server.templateFuncMap` (controller/internal/web/funcmap.go) — obey v2 state-suffix vocabulary.
|
||||
- **New page/nav item**: `baseData` + sidebar in controller/internal/web/templates/ (nested sub-links pattern `.nav-links-nested`); must pass `controller/scripts/template_id_gate.py` + `controller/scripts/emoji_gate.py`.
|
||||
- **New page/nav item**: `baseData` + sidebar in controller/internal/web/templates/ (nested sub-links pattern `.nav-links-nested`); must pass `controller/scripts/template_id_gate.py` + `controller/scripts/emoji_gate.py` + `controller/scripts/native_confirm_gate.py` + `controller/scripts/offbox_rename_gate.py` + `controller/scripts/app_row_dedup_gate.py` + `controller/scripts/mojibake_gate.py`.
|
||||
- **Docker volume tar streaming (v0.125.0)**: `appexport.dockerExec` (seam, package var) + `withVolumeHelper`/`exportVolumeTar`/`importVolumeTar` — stream volume content via `docker cp` through a stopped helper container. NEVER `docker run -v <controller-local path>` — the daemon resolves `-v` host-side and strands the data when the controller is containerized (the v0.124.0 HIGH finding); `controller/scripts/docker_run_volume_path_gate.py` enforces (every `"-v"` allowlisted with its WHY).
|
||||
- **Guarded file download (v0.124.0)**: `handler_export_download.go` — the canonical shape for streaming a server-side file to the browser: accept a BASENAME only (shape regexp + no separators/`..`), `filepath.Join` then assert `filepath.Dir(path) == dir`, `io.Copy` (never ReadAll), `Content-Disposition: attachment`, remove after a successful stream, TTL sweep (`sweepFabDownloads(dir, now, maxAge, logger)` — now injected for tests). Red-proof the guard by loosening to prefix-matching (the `..` case must fail).
|
||||
- **Backups sub-page data**: `backupsCommonData(page, title, r)` + `backupsOffboxData(data)` (handlers.go) — the ONLY builders for the four `/backups*` pages; a new backups section extends these, never re-derives in a page handler. (The one-shot v0.124.0 move gate `backups_split_move_check.py` was retired in v0.126.0.)
|
||||
- **App-list row (v0.126.0)**: `app_list_row`/`app_list_row_end` in `controller/internal/web/templates/app_row.html` is THE canonical list pattern — icon+name(+secondary) left, caller action block right; open with `dict "Slug" ... "Name" ...` (optional `Secondary`/`RowClass`/`Href`/`FallbackIcon`), close with `app_list_row_end`. Do NOT hand-roll app rows — `controller/scripts/app_row_dedup_gate.py` enforces single-sourcing (the backups_apps expander header is the one allowlisted aligned copy). Infra display identity: `inframeta.go` map + `infraMeta` func (filebrowser is the only Linked stack).
|
||||
- **Consequential-action confirm (LIGHT)**: `felhomConfirm(el, question, onYes)` in layout.html (v0.123.0) — the trigger swaps in place to "kérdés + Igen/Mégse"; form buttons opt in with `data-confirm="…"` (delegated listener, `requestSubmit` keeps formaction/name-value). NEVER native `confirm()`/`prompt()` (OS-modals freeze browser automation — drill F-11; `native_confirm_gate.py` enforces). Heavy destructive flows keep the `.confirm-overlay` `openDialog` pattern.
|
||||
- **New hub event**: typed `Notify*` wrapper on Notifier + hub allowlist entry (cross-repo).
|
||||
- **New app integration**: `integrations.Manager.RegisterHandler` with `IntegrationKey(provider, target)`.
|
||||
- **New startup self-check**: append check fn in `selftest.Run` (controller/internal/selftest/selftest.go).
|
||||
@@ -217,7 +321,7 @@ Cross-repo edges:
|
||||
| dir-size ×6 | controller/internal/stacks/delete.go `getDirSizeBytes`/`getDirSizeHuman`; controller/internal/backup/tier2.go `dirSizeBytes` (du -sb); controller/internal/appexport/estimate.go `dirSize`+`duBytes`; controller/internal/appexport/export.go `calcDirSize`; controller/internal/web/handlers.go `dirSizeHuman` |
|
||||
| timeAgo switch body ×2 | controller/internal/web/funcmap.go `timeAgo` vs `timeAgoStr` (identical formatting logic) |
|
||||
| CSRF ×2 | controller/internal/web/csrf.go (session HMAC) vs controller/internal/setup/csrf.go (cookie double-submit) — intentional (pre-auth wizard) but unlabeled |
|
||||
| Budapest timezone loader ×2 | controller/internal/scheduler/scheduler.go `getBudapestLocation` vs controller/internal/web/funcmap.go `getTimezone` |
|
||||
| Budapest timezone loader ×3 | controller/internal/scheduler/scheduler.go `getBudapestLocation` vs controller/internal/web/funcmap.go `getTimezone` vs controller/internal/quiesce/quiesce.go `budapestLocation` (v0.168.0 window gate — Budapest wall-clock, kept local to avoid a scheduler↔quiesce import edge) |
|
||||
| JSON writers ×5, 3 envelope shapes | api `writeJSON`; web `writeDiskJSON`, `jsonResponse`/`jsonError`, `writeDebugJSON` |
|
||||
| Safe-name validators ×4 | controller/internal/web/validate.go `validStackName`; controller/internal/api/router.go `validStackParam` (same body — api↔web import cycle); controller/internal/backup/offbox.go `isSafeStackName`; controller/internal/appexport/validate.go `ValidateSegment` (strictest) |
|
||||
| DB wait/import ×2 | controller/internal/appbackup/dbdump.go `waitDBReady`/`ImportDump` vs controller/internal/appexport/restore.go `waitForDB`/`importDBDump` |
|
||||
|
||||
+13
-11
@@ -14,7 +14,9 @@ destructive section (7) is last and gated.
|
||||
bootstrap-managed. Public dashboard/API: **https://felhom.demo-felhom.eu** (no dashboard password →
|
||||
the API is open; drive it via the PUBLIC URL, not the container IP).
|
||||
- **Versions at writing:** controller **v0.60.0**, agent **v0.30.0**, hub v0.11.0.
|
||||
- `SSH=/c/Windows/System32/OpenSSH/ssh.exe`; host root via SSH alias `felhom-pve`; `export MSYS_NO_PATHCONV=1` for `pct exec`.
|
||||
- Run from DooPlex (192.168.0.180); host root via SSH alias `felhom-pve` — plain `ssh felhom-pve`.
|
||||
(Legacy Windows workstation: needed `SSH=/c/Windows/System32/OpenSSH/ssh.exe` and
|
||||
`export MSYS_NO_PATHCONV=1` for `pct exec`.)
|
||||
- **Findings log:** record every observation (✓/✗ + notes on UX friction, latency, confusing labels,
|
||||
error handling) in a new `REPORT-e2e-live-drive-<date>.md`. Each step says what "good" looks like and
|
||||
what to watch for.
|
||||
@@ -27,12 +29,12 @@ destructive section (7) is last and gated.
|
||||
|
||||
1. Controller healthy + version:
|
||||
- `curl -s https://felhom.demo-felhom.eu/api/health` → `{"ok":true,...}`.
|
||||
- `$SSH felhom-pve "pct exec 9201 -- docker ps --filter name=felhom-controller --format '{{.Image}} {{.Status}}'"` → `:0.60.0 Up ... (healthy)`.
|
||||
- `ssh felhom-pve "pct exec 9201 -- docker ps --filter name=felhom-controller --format '{{.Image}} {{.Status}}'"` → `:0.60.0 Up ... (healthy)`.
|
||||
- Dashboard loads (Hungarian UI), no error banners.
|
||||
2. Agent healthy + version: `$SSH felhom-pve "systemctl is-active felhom-agent; /usr/local/bin/felhom-agent --version"` → `active`, `0.30.0`.
|
||||
2. Agent healthy + version: `ssh felhom-pve "systemctl is-active felhom-agent; /usr/local/bin/felhom-agent --version"` → `active`, `0.30.0`.
|
||||
3. Headroom (deploys pull images — **bound to ≤3 small apps**):
|
||||
- Docker-data volume free: dashboard storage bars, or `$SSH felhom-pve "pct exec 9201 -- df -h /var/lib/docker /"`. Need comfortably above the v0.58 reserve (`max(5GB,10%)`) or deploys will be gated **507**.
|
||||
- RAM: `$SSH felhom-pve "pct exec 9201 -- free -h"`.
|
||||
- Docker-data volume free: dashboard storage bars, or `ssh felhom-pve "pct exec 9201 -- df -h /var/lib/docker /"`. Need comfortably above the v0.58 reserve (`max(5GB,10%)`) or deploys will be gated **507**.
|
||||
- RAM: `ssh felhom-pve "pct exec 9201 -- free -h"`.
|
||||
- Disk list sane: `curl -s https://felhom.demo-felhom.eu/api/disks` → felhom-usb (user-data, data_bearing), local/local-lvm (system), felhom-pbs (backup).
|
||||
4. Record current deployed apps (so cleanup is unambiguous): dashboard "Alkalmazások", or `curl -s https://felhom.demo-felhom.eu/api/stacks/rescan` then the stacks list. (actualbudget is expected already deployed.)
|
||||
- **Good:** all green/healthy; free space well above reserve. **Watch for:** any app stuck "Telepítés alatt" (deploying) from a prior run — note and resolve before starting.
|
||||
@@ -49,7 +51,7 @@ Pick **two small apps** not currently deployed (suggest: `vikunja`, `mealie` —
|
||||
- **Good:** progresses config→containers→health; ends `running`/healthy within ~120s; the card flips to deployed; no "Telepítés" button reappears mid-pull (in-memory Deployed=true during pull).
|
||||
- **Watch for:** stuck at a step, health-probe never going green (check the app's healthcheck tool exists), confusing Hungarian labels, the deploy gate returning **507** (insufficient Docker-data headroom — expected if low on space; note the banner wording).
|
||||
2. Deploy app #2; same checks.
|
||||
3. Confirm on disk the durable record is correct (CTRL-T2-1, happy case): `$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → `deployed: true` (only after success).
|
||||
3. Confirm on disk the durable record is correct (CTRL-T2-1, happy case): `ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → `deployed: true` (only after success).
|
||||
- **Good:** `deployed: true` on disk after a successful deploy. **Watch for:** secrets appearing in plaintext in app.yaml (they must be `enc:`-prefixed — H10/encryption check).
|
||||
|
||||
---
|
||||
@@ -59,10 +61,10 @@ Pick **two small apps** not currently deployed (suggest: `vikunja`, `mealie` —
|
||||
Goal: prove a crash during the image-pull window leaves the stack **NOT-deployed and redeployable**, not ghost-stuck.
|
||||
|
||||
1. Pick a **third app with a non-trivial image pull** (so the pull window is a few seconds — e.g. `paperless-ngx` if space allows, else `mealie`). Start the deploy (UI Telepítés or API POST), and **immediately** — while it is still pulling (status `deploying`, before `running`) — kill the controller:
|
||||
- `$SSH felhom-pve "pct exec 9201 -- docker kill felhom-controller"` **[operator: time this during the pull]**
|
||||
- `ssh felhom-pve "pct exec 9201 -- docker kill felhom-controller"` **[operator: time this during the pull]**
|
||||
- The bootstrap service (`felhom-controller-bootstrap.service`) restarts it within seconds. Confirm back up: `curl -s https://felhom.demo-felhom.eu/api/health`.
|
||||
2. After restart, check the stack state:
|
||||
- On disk: `$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → **`deployed: false`** (transitional — the fix).
|
||||
- On disk: `ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → **`deployed: false`** (transitional — the fix).
|
||||
- UI/API: `GET /api/stacks/<app>` → state `not_deployed` (the card shows **Telepítés**, not a ghost "deployed").
|
||||
3. **Redeploy** the same app — it must be **allowed** (no "already deployed; use update instead" refusal) and complete normally.
|
||||
- **Good:** post-crash the app reads not-deployed and redeploys cleanly. **PRE-FIX behaviour (must NOT occur):** app.yaml `deployed: true` with no containers, and redeploy refused — that's the ghost-stuck regression the fix removes.
|
||||
@@ -78,9 +80,9 @@ Goal: prove a crash during the image-pull window leaves the stack **NOT-deployed
|
||||
- **Good:** imports, recreates the stack, data restored; fail-closed data-key gate honored if the app has a data-encrypting key.
|
||||
3. **Negative — path traversal (CTRL-001):** craft a hostile `.fab` and confirm it is **rejected at parse**, not written.
|
||||
- Build a minimal bundle whose `manifest.json` has `"app_name":"../evil"` (and/or an `hdd_subdirs` / `volume_names` entry with `../`). Place it under a registered `exports/` dir on the host:
|
||||
`$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller sh -c 'ls /mnt/felhom-usb/exports/'"` to find the dir.
|
||||
`ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller sh -c 'ls /mnt/felhom-usb/exports/'"` to find the dir.
|
||||
- Attempt import of the hostile bundle.
|
||||
- **Good (the fix):** import **fails immediately** with a manifest/validation error; **no directory is created outside the stacks dir** (verify: `$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller ls -la /opt/docker/evil /etc/evil 2>/dev/null"` → nothing). **PRE-FIX (must NOT occur):** a dir/file written outside `/opt/docker/stacks/`.
|
||||
- **Good (the fix):** import **fails immediately** with a manifest/validation error; **no directory is created outside the stacks dir** (verify: `ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller ls -la /opt/docker/evil /etc/evil 2>/dev/null"` → nothing). **PRE-FIX (must NOT occur):** a dir/file written outside `/opt/docker/stacks/`.
|
||||
- **Watch for:** the error message clarity (does the UI explain why it was rejected?).
|
||||
|
||||
---
|
||||
@@ -105,7 +107,7 @@ Re-confirm the two refusals proven on 2026-06-13. **Do NOT send a matching confi
|
||||
- **Good:** `formatted:false`, `needs_confirmation:true`, **HTTP 409**; no mkfs.
|
||||
2. **Refusal B — wrong durable_id:** same call with `"confirmed":true,"durable_id":"byid:wwn-0xDEADBEEF-DOES-NOT-EXIST"`.
|
||||
- **Good:** `formatted:false`, refused **409**; a non-matching confirmation does not authorize a wipe.
|
||||
3. **Data-safety assertion:** `$SSH felhom-pve "findmnt /mnt/felhom-usb -o TARGET,SOURCE,FSTYPE; pct exec 9201 -- docker exec felhom-controller sh -c 'df -h /mnt/felhom-usb'"` → still mounted, used space unchanged.
|
||||
3. **Data-safety assertion:** `ssh felhom-pve "findmnt /mnt/felhom-usb -o TARGET,SOURCE,FSTYPE; pct exec 9201 -- docker exec felhom-controller sh -c 'df -h /mnt/felhom-usb'"` → still mounted, used space unchanged.
|
||||
4. **Happy-path destructive wipe** = **[HUMAN]** — never wipe a real/customer drive to test; covered by the agent unit test `retarget-mismatch-refused`. Only on a genuinely disposable blank device, supervised. **[DESTRUCTIVE — operator confirm]**
|
||||
|
||||
---
|
||||
|
||||
@@ -4,8 +4,13 @@ bin/
|
||||
*.dll
|
||||
*.so
|
||||
*.dylib
|
||||
controller
|
||||
controller.exe
|
||||
# ANCHORED (leading slash) on purpose: a bare `controller` also matches the DIRECTORY
|
||||
# cmd/controller/, so ripgrep silently skipped main.go and new files there needed `git add -f`.
|
||||
# Both directions produce inert-seam mistakes — a grep for a setter finds no caller and reads as
|
||||
# "this is unused", and a genuinely-new file never gets committed. Only the built binary at the
|
||||
# module root should be ignored here.
|
||||
/controller
|
||||
/controller.exe
|
||||
|
||||
# Test artifacts
|
||||
coverage.out
|
||||
|
||||
+1145
-53
File diff suppressed because it is too large
Load Diff
+3
-3
@@ -2,7 +2,7 @@
|
||||
# =============================================================================
|
||||
# felhom-controller — Docker image build script
|
||||
# =============================================================================
|
||||
# Location: /home/kisfenyo/build/felhom-controller/build.sh
|
||||
# Location: /mnt/5_hdd/felhom.eu/build/felhom-controller/build.sh (moved off the DooPlex SSD 2026-07-18)
|
||||
#
|
||||
# Copies source from the git repo, syncs app assets, and builds the image.
|
||||
# Build artifacts stay here — the git repo stays clean.
|
||||
@@ -16,9 +16,9 @@
|
||||
set -euo pipefail
|
||||
|
||||
# --- Configuration (edit these if your paths differ) ---
|
||||
REPO_DIR="/home/kisfenyo/git/felhom-controller"
|
||||
REPO_DIR="/mnt/5_hdd/felhom.eu/git/felhom-controller"
|
||||
CONTROLLER_SRC="${REPO_DIR}/controller"
|
||||
WEBSITE_ASSETS_DIR="/home/kisfenyo/git/felhom.eu/website/assets"
|
||||
WEBSITE_ASSETS_DIR="/mnt/5_hdd/felhom.eu/git/felhom.eu/website/assets"
|
||||
REGISTRY="gitea.dooplex.hu/admin"
|
||||
IMAGE="${REGISTRY}/felhom-controller"
|
||||
|
||||
|
||||
@@ -0,0 +1,495 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-166 §10 seam discipline — the recovery and the backfill are seams, and a seam that is never
|
||||
// called is the defect class this project has shipped four times: a correct component, green unit
|
||||
// tests that inject it directly, and no production caller.
|
||||
//
|
||||
// These walk main.go's AST. NOT strings.Contains — the sibling bootrecon test records the reason at
|
||||
// first hand: a commented-out call still satisfies a substring match, so the text version passed the
|
||||
// very red-proof it existed to fail. Comments are not code.
|
||||
|
||||
// mainBody returns func main()'s body from main.go, parsed.
|
||||
func mainBody(t *testing.T) *ast.BlockStmt {
|
||||
t.Helper()
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
for _, decl := range f.Decls {
|
||||
if fn, ok := decl.(*ast.FuncDecl); ok && fn.Name.Name == "main" && fn.Body != nil {
|
||||
return fn.Body
|
||||
}
|
||||
}
|
||||
t.Fatal("func main() not found in main.go")
|
||||
return nil
|
||||
}
|
||||
|
||||
// callsInMain returns, in source order, the names of every call in func main() whose function
|
||||
// expression is `x.Sel(...)` or `Sel(...)` — enough to identify the wiring calls by name.
|
||||
func callsInMain(t *testing.T, body *ast.BlockStmt) []string {
|
||||
t.Helper()
|
||||
var names []string
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
switch fun := call.Fun.(type) {
|
||||
case *ast.SelectorExpr:
|
||||
names = append(names, fun.Sel.Name)
|
||||
case *ast.Ident:
|
||||
names = append(names, fun.Name)
|
||||
}
|
||||
return true
|
||||
})
|
||||
return names
|
||||
}
|
||||
|
||||
func indexOfCall(names []string, want string) int {
|
||||
for i, n := range names {
|
||||
if n == want {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// TestMainWiresAppStopRecovery is the Group-I seam test. Comment out the `appStopGuard.Recover()`
|
||||
// line in main.go and this fails, where every behavioural test in internal/backup still passes.
|
||||
func TestMainWiresAppStopRecovery(t *testing.T) {
|
||||
names := callsInMain(t, mainBody(t))
|
||||
|
||||
if indexOfCall(names, "NewAppStopGuard") < 0 {
|
||||
t.Fatal("func main() no longer builds the R-166 app-stop guard — nothing writes or reads the marker")
|
||||
}
|
||||
if indexOfCall(names, "SetStarter") < 0 {
|
||||
t.Fatal("func main() no longer calls SetStarter on the app-stop guard — Recover would find the " +
|
||||
"marker and be unable to start anything, leaving every interrupted app down")
|
||||
}
|
||||
if indexOfCall(names, "Recover") < 0 {
|
||||
t.Fatal("func main() no longer calls Recover() on the app-stop guard — apps left stopped by an " +
|
||||
"interrupted backup stay down forever (the R-166 defect, un-fixed)")
|
||||
}
|
||||
if indexOfCall(names, "SetAppStopGuard") < 0 {
|
||||
t.Fatal("func main() no longer hands the recovered guard to the backup manager — the manager " +
|
||||
"would build a SECOND guard over the same file, i.e. one file with two owners")
|
||||
}
|
||||
if indexOfCall(names, "SetStopGuard") < 0 {
|
||||
t.Fatal("func main() no longer wires the exporter's stop guard — the .fab export path would be " +
|
||||
"the one uncovered stop-and-restart site, which is how a reader concludes the class is handled")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMainWiresDesiredStateBackfill pins the Part-1.5 call.
|
||||
func TestMainWiresDesiredStateBackfill(t *testing.T) {
|
||||
if indexOfCall(callsInMain(t, mainBody(t)), "BackfillDesiredState") < 0 {
|
||||
t.Fatal("func main() no longer calls BackfillDesiredState — every existing app would stay on " +
|
||||
"legacy inference until someone pressed a button on it")
|
||||
}
|
||||
}
|
||||
|
||||
// TestAppStopRecoveryPrecedesTheBootReconciler is §8.4's ORDERING requirement, and it is the reason
|
||||
// the recovery returns its result instead of pushing it through a notifier seam.
|
||||
//
|
||||
// The recovery must COMPLETE — not merely be reached — before `go runBootReconcile(...)` is
|
||||
// launched. If the boot reconciler ran first it would see an app the marker already explains, list
|
||||
// it as an unexplained boot orphan, and one fault would be reported as two.
|
||||
func TestAppStopRecoveryPrecedesTheBootReconciler(t *testing.T) {
|
||||
names := callsInMain(t, mainBody(t))
|
||||
|
||||
recover := indexOfCall(names, "Recover")
|
||||
bootrecon := indexOfCall(names, "runBootReconcile")
|
||||
backfill := indexOfCall(names, "BackfillDesiredState")
|
||||
|
||||
if recover < 0 || bootrecon < 0 || backfill < 0 {
|
||||
t.Fatalf("missing a call: Recover=%d runBootReconcile=%d BackfillDesiredState=%d", recover, bootrecon, backfill)
|
||||
}
|
||||
if recover >= bootrecon {
|
||||
t.Fatal("the app-stop Recover no longer runs BEFORE the boot reconciler is launched — an app " +
|
||||
"the marker explains would also be reported as an unexplained boot orphan (§8.4)")
|
||||
}
|
||||
if backfill >= bootrecon {
|
||||
t.Fatal("the desired-state backfill no longer runs BEFORE the boot reconciler — the reconciler " +
|
||||
"would decide from intent the backfill had not yet written")
|
||||
}
|
||||
if recover >= backfill {
|
||||
t.Fatal("the backfill no longer runs AFTER the app-stop recovery — an app the recovery just " +
|
||||
"restarted would still read as down and be left unrecorded")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMainReportsTheInterruptedOperation pins §2.4: the recovery's outcome reaches the operator.
|
||||
//
|
||||
// The reporting call is deliberately far from the recovery (the notifier does not exist yet at
|
||||
// recovery time), which is exactly the distance across which a wiring gets dropped.
|
||||
func TestMainReportsTheInterruptedOperation(t *testing.T) {
|
||||
body := mainBody(t)
|
||||
names := callsInMain(t, body)
|
||||
|
||||
if indexOfCall(names, "NotifyBackupFailed") < 0 {
|
||||
t.Fatal("func main() no longer reports an interrupted app-data operation to the operator — the " +
|
||||
"controller died mid-backup and nobody is told (§2.4)")
|
||||
}
|
||||
// It must be guarded, not unconditional: a box with nothing to recover must not email an operator
|
||||
// on every single boot.
|
||||
//
|
||||
// R-174 STRENGTHENED THIS. `!= nil` alone is no longer sufficient, because Recover now returns a
|
||||
// non-nil result for a recovery that merely REFUSED starts (an absent data drive) — the drive
|
||||
// gate working as designed. `NotifyBackupFailed` sends `backup_failed`, which is customer-enabled
|
||||
// by default (settings.DefaultEnabledEvents), so a nil-only guard would email the customer
|
||||
// "A biztonsági mentés sikertelen!" about an app nothing is wrong with. The guard must consult
|
||||
// Alarming().
|
||||
guardedByNil, guardedByAlarming := false, false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
ifst, ok := n.(*ast.IfStmt)
|
||||
if !ok || ifst.Cond == nil {
|
||||
return true
|
||||
}
|
||||
carries := false
|
||||
for _, name := range callsInMain(t, ifst.Body) {
|
||||
if name == "NotifyBackupFailed" {
|
||||
carries = true
|
||||
}
|
||||
}
|
||||
if !carries {
|
||||
return true
|
||||
}
|
||||
// Walk the whole condition: it may be `a != nil && a.Alarming()`.
|
||||
ast.Inspect(ifst.Cond, func(c ast.Node) bool {
|
||||
switch e := c.(type) {
|
||||
case *ast.BinaryExpr:
|
||||
if x, ok := e.X.(*ast.Ident); ok && x.Name == "appStopRecovery" && e.Op == token.NEQ {
|
||||
guardedByNil = true
|
||||
}
|
||||
case *ast.CallExpr:
|
||||
if sel, ok := e.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "Alarming" {
|
||||
if x, ok := sel.X.(*ast.Ident); ok && x.Name == "appStopRecovery" {
|
||||
guardedByAlarming = true
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
return true
|
||||
})
|
||||
if !guardedByNil {
|
||||
t.Fatal("the interrupted-operation alert is not guarded by `appStopRecovery != nil` — every " +
|
||||
"healthy boot would page the operator about a backup that was never interrupted")
|
||||
}
|
||||
if !guardedByAlarming {
|
||||
t.Fatal("the interrupted-operation alert is not guarded by appStopRecovery.Alarming() — a " +
|
||||
"recovery that only REFUSED starts (drive absent) would be reported through " +
|
||||
"NotifyBackupFailed, a customer-enabled event type, telling the customer their backup " +
|
||||
"failed when the drive gate was simply doing its job (R-174)")
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-171 seam: the boot drive gate must be WIRED in production -------------------------------
|
||||
|
||||
// TestMainWiresBootDriveGate is the Group-H seam test. An unwired drive gate is not a crash — it is
|
||||
// SILENTLY the pre-v0.190.0 behaviour, which started apps onto absent drives (observed live,
|
||||
// audits/DIAG-bootrecon-drive-absent-2026-08-02.md). Every behavioural test in internal/bootrecon
|
||||
// still passes with the wiring gone, which is exactly the hole this walks the AST to close.
|
||||
//
|
||||
// AST, not strings.Contains: a commented-out call still contains the string — the distinction that
|
||||
// made a previous version of this project's own seam test pass its red-proof (2026-07-21).
|
||||
func TestMainWiresBootDriveGate(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
|
||||
// (a) the settings handle the gate reads is assigned somewhere in main().
|
||||
assigned := false
|
||||
for _, name := range assignedIdentsIn(mainBody(t)) {
|
||||
if name == "bootDriveSettings" {
|
||||
assigned = true
|
||||
}
|
||||
}
|
||||
if !assigned {
|
||||
t.Fatal("func main() no longer assigns bootDriveSettings — the boot drive gate would read a " +
|
||||
"nil settings handle and could not see a disconnected drive")
|
||||
}
|
||||
|
||||
// (b) SetDriveGate is actually called where the reconciler is constructed.
|
||||
called := false
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
if sel, ok := call.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "SetDriveGate" {
|
||||
called = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !called {
|
||||
t.Fatal("main.go no longer calls SetDriveGate on the boot reconciler — the sweep would start " +
|
||||
"apps whose data drive is absent (R-171, a regression observed live on 2026-08-02)")
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-174 seam: the app-stop guard's starter must be GATED in production -----------------------
|
||||
|
||||
// TestMainWiresGatedAppStopStarter pins Part 0's production wiring. `SetStarter(stackMgr)` — the raw
|
||||
// manager, which is what shipped in v0.189.0 — compiles, passes every behavioural test in
|
||||
// internal/backup (they inject their own gating starter), and silently starts apps onto absent
|
||||
// drives at boot. The ONLY thing that distinguishes the fixed wiring from the broken one is the
|
||||
// argument at the call site, so that is what this reads.
|
||||
//
|
||||
// AST, not strings.Contains: a commented-out call still contains the string.
|
||||
func TestMainWiresGatedAppStopStarter(t *testing.T) {
|
||||
body := mainBody(t)
|
||||
|
||||
var arg ast.Expr
|
||||
found := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || sel.Sel.Name != "SetStarter" || len(call.Args) != 1 {
|
||||
return true
|
||||
}
|
||||
// Only the app-stop guard's SetStarter, not some other type's.
|
||||
if x, ok := sel.X.(*ast.Ident); !ok || x.Name != "appStopGuard" {
|
||||
return true
|
||||
}
|
||||
arg, found = call.Args[0], true
|
||||
return false
|
||||
})
|
||||
if !found {
|
||||
t.Fatal("func main() no longer calls appStopGuard.SetStarter — Recover would find the marker " +
|
||||
"and be unable to start anything")
|
||||
}
|
||||
|
||||
// The argument must be a gatedAppStopStarter composite literal. A bare identifier (`stackMgr`)
|
||||
// is precisely the v0.189.0 defect.
|
||||
lit, ok := arg.(*ast.CompositeLit)
|
||||
if !ok {
|
||||
t.Fatalf("appStopGuard.SetStarter is wired with %T, not a gatedAppStopStarter literal — an "+
|
||||
"un-gated starter restarts apps onto MISSING drives at boot (R-174, the R-171 defect one "+
|
||||
"path over)", arg)
|
||||
}
|
||||
id, ok := lit.Type.(*ast.Ident)
|
||||
if !ok || id.Name != "gatedAppStopStarter" {
|
||||
t.Fatalf("appStopGuard.SetStarter is wired with a %v literal, want gatedAppStopStarter", lit.Type)
|
||||
}
|
||||
|
||||
// And that gate must be a driveStartGate — the SAME predicate the boot sweep uses, so the two
|
||||
// cannot disagree about whether an app's drive is available.
|
||||
gated := false
|
||||
for _, el := range lit.Elts {
|
||||
kv, ok := el.(*ast.KeyValueExpr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
k, ok := kv.Key.(*ast.Ident)
|
||||
if !ok || k.Name != "gate" {
|
||||
continue
|
||||
}
|
||||
if gl, ok := kv.Value.(*ast.CompositeLit); ok {
|
||||
if gid, ok := gl.Type.(*ast.Ident); ok && gid.Name == "driveStartGate" {
|
||||
gated = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !gated {
|
||||
t.Fatal("the app-stop starter's gate is not a driveStartGate — the crash recovery and the " +
|
||||
"boot sweep would answer \"may this app start?\" from two different implementations, " +
|
||||
"which is the drift the extraction exists to prevent")
|
||||
}
|
||||
}
|
||||
|
||||
// TestBootDriveGateAndAppStopShareTheDrivePredicate pins the OTHER half of the same claim: the boot
|
||||
// sweep must keep delegating to driveStartGate rather than growing its own copy of the drive checks.
|
||||
//
|
||||
// This is the "a comment asserting an invariant needs a test pinning it" rule. The claim — that the
|
||||
// two gates cannot disagree — is true only while both call the same code.
|
||||
func TestBootDriveGateAndAppStopShareTheDrivePredicate(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
|
||||
var mayStart *ast.FuncDecl
|
||||
for _, decl := range f.Decls {
|
||||
fn, ok := decl.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "MayStart" || fn.Recv == nil || len(fn.Recv.List) != 1 {
|
||||
continue
|
||||
}
|
||||
if id, ok := fn.Recv.List[0].Type.(*ast.Ident); ok && id.Name == "bootDriveGate" {
|
||||
mayStart = fn
|
||||
}
|
||||
}
|
||||
if mayStart == nil {
|
||||
t.Fatal("bootDriveGate.MayStart not found in main.go")
|
||||
}
|
||||
|
||||
// It must call through to the shared predicate.
|
||||
delegates := false
|
||||
ast.Inspect(mayStart.Body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || sel.Sel.Name != "MayStart" {
|
||||
return true
|
||||
}
|
||||
if x, ok := sel.X.(*ast.SelectorExpr); ok && x.Sel.Name == "drive" {
|
||||
delegates = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !delegates {
|
||||
t.Fatal("bootDriveGate.MayStart no longer delegates to the shared driveStartGate — the boot " +
|
||||
"sweep and the app-stop crash recovery would each carry their own drive logic, and the " +
|
||||
"two can then disagree about whether an app may start (R-174)")
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-158 / R-167 seams: both new alerts must be WIRED in production ---------------------------
|
||||
|
||||
// TestMainWiresTheUnitCaptureAlert pins Part 1's seam. `SetUnitNotify` is nil-safe by design, so an
|
||||
// unwired seam is not a crash — it is SILENTLY the pre-v0.191.0 behaviour, in which a per-app Tier-1
|
||||
// capture failure is a `[WARN]` line and reaches no hub channel at all. Every behavioural test in
|
||||
// internal/backup injects its own callback and passes with the production wiring gone, which is
|
||||
// exactly the hole this closes. THIS PROJECT'S COUNT OF "BUILT BUT NEVER WIRED" REACHES FIVE WITH
|
||||
// R-158 — the defect being fixed here IS an instance of it.
|
||||
func TestMainWiresTheUnitCaptureAlert(t *testing.T) {
|
||||
names := callsInMain(t, mainBody(t))
|
||||
|
||||
if indexOfCall(names, "SetUnitNotify") < 0 {
|
||||
t.Fatal("func main() no longer calls backupMgr.SetUnitNotify — a per-app recovery-unit " +
|
||||
"capture failure would reach no hub channel, which is R-158 un-fixed (the seam built " +
|
||||
"and left disconnected, for the fifth time in this project)")
|
||||
}
|
||||
if indexOfCall(names, "NotifyRecoveryUnitCaptureFailed") < 0 {
|
||||
t.Fatal("main.go no longer calls NotifyRecoveryUnitCaptureFailed — the seam is wired to " +
|
||||
"something that pushes no event, which looks identical to a working alert from inside " +
|
||||
"internal/backup")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMainWiresTheFillWatcher pins Part 2's seam. Three separate things can be dropped and each one
|
||||
// silently reverts the customer to "nothing warns before a disk fills": the watcher can go
|
||||
// unconstructed, its notify can go unwired (the Watcher is nil-safe), or it can never be scheduled.
|
||||
func TestMainWiresTheFillWatcher(t *testing.T) {
|
||||
body := mainBody(t)
|
||||
names := callsInMain(t, body)
|
||||
|
||||
if indexOfCall(names, "New") < 0 || !assignsIdent(body, "fillWatcher") {
|
||||
t.Fatal("func main() no longer constructs the fill watcher — nothing warns the customer " +
|
||||
"before a filesystem fills (R-167, decision D-c's customer half)")
|
||||
}
|
||||
if indexOfCall(names, "SetNotify") < 0 {
|
||||
t.Fatal("func main() no longer calls SetNotify on the fill watcher — the Watcher is nil-safe, " +
|
||||
"so it would run the checks, update its state, log, and tell the CUSTOMER nothing")
|
||||
}
|
||||
|
||||
// It must actually be scheduled: a watcher nobody calls is a watcher that never fires.
|
||||
scheduled := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok || len(call.Args) == 0 {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || (sel.Sel.Name != "Daily" && sel.Sel.Name != "Every") {
|
||||
return true
|
||||
}
|
||||
lit, ok := call.Args[0].(*ast.BasicLit)
|
||||
if ok && strings.Contains(lit.Value, "fill-watch") {
|
||||
scheduled = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !scheduled {
|
||||
t.Fatal("the fill watcher is never registered on the scheduler — it would be constructed, " +
|
||||
"wired, and never run, which is indistinguishable from a filesystem that never fills")
|
||||
}
|
||||
|
||||
// It must ALSO run once at startup. Neither `Every` nor `Daily` fires on registration (both wait
|
||||
// for their first tick), so a schedule-only wiring means a box that BOOTS with a filesystem
|
||||
// already over the line stays silent for up to 24 hours — a real fault visible only after a
|
||||
// deadline elapses, which is the R-100 shape. The hub's own checkers leave already-breached keys
|
||||
// unseeded at init for exactly this reason.
|
||||
if indexOfCall(names, "After") < 0 {
|
||||
t.Fatal("nothing delays a startup fill check — see fillWatchStartupDelay")
|
||||
}
|
||||
startupRun := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
g, ok := n.(*ast.GoStmt)
|
||||
if !ok || g.Call == nil {
|
||||
return true
|
||||
}
|
||||
lit, ok := g.Call.Fun.(*ast.FuncLit)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
var sawDelay, sawCheck bool
|
||||
ast.Inspect(lit.Body, func(m ast.Node) bool {
|
||||
if id, ok := m.(*ast.Ident); ok && id.Name == "fillWatchStartupDelay" {
|
||||
sawDelay = true
|
||||
}
|
||||
if call, ok := m.(*ast.CallExpr); ok {
|
||||
if sel, ok := call.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "Check" {
|
||||
if x, ok := sel.X.(*ast.Ident); ok && x.Name == "fillWatcher" {
|
||||
sawCheck = true
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
if sawDelay && sawCheck {
|
||||
startupRun = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !startupRun {
|
||||
t.Fatal("the fill watcher never runs at STARTUP — Daily/Every both wait for their first " +
|
||||
"tick, so a box that boots with a full disk would not warn for up to 24 hours (the " +
|
||||
"R-100 shape: a real fault visible only after a deadline elapses)")
|
||||
}
|
||||
}
|
||||
|
||||
// assignsIdent reports whether a block assigns to the named identifier.
|
||||
func assignsIdent(body *ast.BlockStmt, want string) bool {
|
||||
for _, n := range assignedIdentsIn(body) {
|
||||
if n == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// assignedIdentsIn returns the names assigned to in a block (plain `=` and `:=`).
|
||||
func assignedIdentsIn(body *ast.BlockStmt) []string {
|
||||
var names []string
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
as, ok := n.(*ast.AssignStmt)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
for _, lhs := range as.Lhs {
|
||||
if id, ok := lhs.(*ast.Ident); ok {
|
||||
names = append(names, id.Name)
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
return names
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"io"
|
||||
"log"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/bootrecon"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// §9 rule 6 — the seam-discipline test. Two inert-seam defects shipped in the two days before this
|
||||
// task (controller v0.154.0 and agent v0.91.0), both the same shape: the component was correct, its
|
||||
// unit tests injected the seam directly, and the PRODUCTION CALLER was never made. Everything was
|
||||
// green and the feature did nothing. So R-52 gets its wiring asserted from package main, not only
|
||||
// from internal/bootrecon.
|
||||
|
||||
// TestRunBootReconcile_InvokesTheSweep pins the function main() actually calls: after the settle
|
||||
// window it runs the sweep exactly once, with the manager it was handed.
|
||||
func TestRunBootReconcile_InvokesTheSweep(t *testing.T) {
|
||||
orig := bootReconcileFn
|
||||
t.Cleanup(func() { bootReconcileFn = orig })
|
||||
origSettle := bootReconcileSettle
|
||||
t.Cleanup(func() { bootReconcileSettle = origSettle })
|
||||
bootReconcileSettle = time.Millisecond
|
||||
|
||||
calls := 0
|
||||
var gotMgr bootrecon.StackProvider
|
||||
bootReconcileFn = func(_ context.Context, mgr bootrecon.StackProvider, _ *log.Logger) bootrecon.Result {
|
||||
calls++
|
||||
gotMgr = mgr
|
||||
return bootrecon.Result{}
|
||||
}
|
||||
|
||||
fake := &wiringStacks{}
|
||||
runBootReconcile(context.Background(), fake, log.New(io.Discard, "", 0))
|
||||
|
||||
if calls != 1 {
|
||||
t.Fatalf("the boot sweep ran %d times, want exactly 1 (start-once, never a loop)", calls)
|
||||
}
|
||||
if gotMgr != bootrecon.StackProvider(fake) {
|
||||
t.Fatalf("the sweep was handed %v, want the stack manager main() owns", gotMgr)
|
||||
}
|
||||
}
|
||||
|
||||
// A controller shutting down during its own settle window must not start anything.
|
||||
func TestRunBootReconcile_CancelledDuringSettleDoesNothing(t *testing.T) {
|
||||
orig := bootReconcileFn
|
||||
t.Cleanup(func() { bootReconcileFn = orig })
|
||||
|
||||
calls := 0
|
||||
bootReconcileFn = func(context.Context, bootrecon.StackProvider, *log.Logger) bootrecon.Result {
|
||||
calls++
|
||||
return bootrecon.Result{}
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
runBootReconcile(ctx, &wiringStacks{}, log.New(io.Discard, "", 0))
|
||||
|
||||
if calls != 0 {
|
||||
t.Fatalf("the sweep ran %d times on a cancelled context, want 0", calls)
|
||||
}
|
||||
}
|
||||
|
||||
// The call site itself. A function-variable test can only prove the function is correct — it cannot
|
||||
// prove main() calls it, which is exactly the hole both inert-seam defects fell through. This walks
|
||||
// main.go's AST for a `go runBootReconcile(...)` inside func main(); delete or comment out that line
|
||||
// and this fails, where every behavioural test above would still pass.
|
||||
//
|
||||
// It is an AST walk and not a strings.Contains for a reason found while red-proofing it: a
|
||||
// commented-out call still satisfies a substring match, so the text version passed the very
|
||||
// red-proof it existed to fail. Comments are not code.
|
||||
func TestMainWiresBootReconcile(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
|
||||
found := false
|
||||
for _, decl := range f.Decls {
|
||||
fn, ok := decl.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "main" || fn.Body == nil {
|
||||
continue
|
||||
}
|
||||
ast.Inspect(fn.Body, func(n ast.Node) bool {
|
||||
gostmt, ok := n.(*ast.GoStmt)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
if ident, ok := gostmt.Call.Fun.(*ast.Ident); ok && ident.Name == "runBootReconcile" {
|
||||
found = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("func main() no longer starts the R-52 boot reconciliation with `go runBootReconcile(...)` " +
|
||||
"— the sweep is inert (the v0.154.0 / v0.91.0 defect class: a correct component nobody calls)")
|
||||
}
|
||||
}
|
||||
|
||||
// The settle window must stay inside the dead-app boot grace, or a successful recovery would alert.
|
||||
func TestBootReconcileFitsInsideTheBootGrace(t *testing.T) {
|
||||
worst := bootReconcileSettle + time.Duration(bootrecon.DefaultAttempts-1)*bootrecon.DefaultRetryDelay
|
||||
if worst >= deadAppBootGrace {
|
||||
t.Fatalf("worst-case sweep %s does not fit inside the %s boot grace — a successful "+
|
||||
"recovery would fire app_start_failed", worst, deadAppBootGrace)
|
||||
}
|
||||
}
|
||||
|
||||
type wiringStacks struct{}
|
||||
|
||||
func (w *wiringStacks) GetStacks() []stacks.Stack { return nil }
|
||||
func (w *wiringStacks) StartStack(string) error { return nil }
|
||||
func (w *wiringStacks) RefreshStatus() error { return nil }
|
||||
@@ -0,0 +1,366 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"log"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/bootrecon"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// R-157 mechanism A — the sweep that looked once.
|
||||
//
|
||||
// TIMING IS NOT TESTED BY SLEEPING (§10). The window's constants are package vars, so each test
|
||||
// shrinks them to sub-millisecond values: the CONTRACT under test is "how many samples, and what
|
||||
// ends the window", not "how long a second is". A test that waited real seconds would be slow,
|
||||
// flaky, and would still not prove the contract.
|
||||
|
||||
// windowStacks is a StackProvider whose fleet CHANGES over successive GetStacks() calls — which is
|
||||
// the whole point: the pre-v0.190.0 sweep sampled once and could not see a late settler.
|
||||
type windowStacks struct {
|
||||
// frames is the fleet as seen on each successive GetStacks() call; the last frame repeats.
|
||||
frames [][]stacks.Stack
|
||||
calls int
|
||||
starts map[string]int
|
||||
onStart func(*windowStacks, string)
|
||||
refreshes int
|
||||
refreshErr error
|
||||
// cycle makes the fleet NEVER settle: frames repeat forever instead of the last one sticking.
|
||||
// Required by the budget test — with frames that eventually stop changing, the window terminates
|
||||
// by SETTLING even with the budget removed, so the red-proof would not reach the hang it exists
|
||||
// to demonstrate.
|
||||
cycle bool
|
||||
}
|
||||
|
||||
func (w *windowStacks) GetStacks() []stacks.Stack {
|
||||
i := w.calls
|
||||
w.calls++
|
||||
if i >= len(w.frames) {
|
||||
if w.cycle {
|
||||
i = i % len(w.frames)
|
||||
} else {
|
||||
i = len(w.frames) - 1
|
||||
}
|
||||
}
|
||||
return w.frames[i]
|
||||
}
|
||||
|
||||
func (w *windowStacks) RefreshStatus() error {
|
||||
w.refreshes++
|
||||
if w.refreshErr != nil {
|
||||
return w.refreshErr
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (w *windowStacks) StartStack(name string) error {
|
||||
if w.starts == nil {
|
||||
w.starts = map[string]int{}
|
||||
}
|
||||
w.starts[name]++
|
||||
if w.onStart != nil {
|
||||
w.onStart(w, name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// shrinkWindow makes the window fast and deterministic, and restores the shipped values after.
|
||||
func shrinkWindow(t *testing.T, sample time.Duration, stableFor int, budget time.Duration) {
|
||||
t.Helper()
|
||||
os, ost, ob, osettle := bootReconcileSample, bootReconcileStableFor, bootReconcileBudget, bootReconcileSettle
|
||||
t.Cleanup(func() {
|
||||
bootReconcileSample, bootReconcileStableFor, bootReconcileBudget, bootReconcileSettle = os, ost, ob, osettle
|
||||
})
|
||||
bootReconcileSample, bootReconcileStableFor, bootReconcileBudget = sample, stableFor, budget
|
||||
bootReconcileSettle = time.Millisecond
|
||||
}
|
||||
|
||||
// captureSweep replaces the sweep with a recorder and returns the fleet it was handed.
|
||||
func captureSweep(t *testing.T) *[][]stacks.Stack {
|
||||
t.Helper()
|
||||
orig := bootReconcileFn
|
||||
t.Cleanup(func() { bootReconcileFn = orig })
|
||||
var seen [][]stacks.Stack
|
||||
bootReconcileFn = func(_ context.Context, mgr bootrecon.StackProvider, _ *log.Logger) bootrecon.Result {
|
||||
seen = append(seen, mgr.GetStacks())
|
||||
return bootrecon.Result{}
|
||||
}
|
||||
return &seen
|
||||
}
|
||||
|
||||
func upStack(name string) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name, Deployed: true, State: stacks.StateRunning,
|
||||
Containers: []stacks.ContainerInfo{{Name: name, State: stacks.StateRunning}},
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
// settlingLate is the R-157-A shape: at T+5s the app is still `starting` with its containers coming
|
||||
// up, and it only comes to rest in a DOWN state later.
|
||||
func settlingLate(name string) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name, Deployed: true, State: stacks.StateStarting,
|
||||
Containers: []stacks.ContainerInfo{{Name: name, State: stacks.StateStarting}},
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
func settledDown(name string) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name, Deployed: true, State: stacks.StateExited,
|
||||
Containers: []stacks.ContainerInfo{{Name: name, State: stacks.StateExited}},
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group A / Scenario B — a late settler IS swept -----------------------------------------------
|
||||
|
||||
func TestBootWindow_LateSettlerIsSweptOnASettledFleet(t *testing.T) {
|
||||
// The fleet is still moving for the first frames and settles only later. The sweep must run
|
||||
// AFTER it settles and must be handed the SETTLED fleet — because the pre-v0.190.0 defect was a
|
||||
// candidate set derived from a fleet that had not finished moving.
|
||||
//
|
||||
// RED-PROOF: restore the single-sweep shape (delete the sampling loop so runBootReconcile calls
|
||||
// bootReconcileFn straight after the settle delay) and this test fails — the sweep is handed the
|
||||
// `starting` frame, in which the app is not a down-state candidate at all.
|
||||
// Demonstrated in REPORT.md §4.
|
||||
shrinkWindow(t, time.Millisecond, 2, 500*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
|
||||
w := &windowStacks{frames: [][]stacks.Stack{
|
||||
{settlingLate("immich")}, // T+5s: still coming up
|
||||
{settlingLate("immich")},
|
||||
{settledDown("immich")}, // settles into a down state only now
|
||||
{settledDown("immich")},
|
||||
{settledDown("immich")},
|
||||
}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("the sweep ran %d times, want exactly 1 — the window samples, it does not sweep per sample", len(*seen))
|
||||
}
|
||||
got := (*seen)[0]
|
||||
if len(got) != 1 || got[0].State != stacks.StateExited {
|
||||
t.Fatalf("the sweep was handed state=%v, want the SETTLED (exited) fleet — a candidate set "+
|
||||
"derived from a still-moving fleet is exactly the R-157 mechanism-A defect", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_SweepRunsExactlyOnceEvenOnAQuietBoot(t *testing.T) {
|
||||
shrinkWindow(t, time.Millisecond, 2, 500*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("bookstack")}}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("sweeps=%d, want exactly 1 on a quiet boot", len(*seen))
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group B / Scenario C — the window TERMINATES -------------------------------------------------
|
||||
|
||||
func TestBootWindow_BudgetEndsAForeverChangingFleet(t *testing.T) {
|
||||
// A fleet that never stops changing must not sample forever. The budget ends it, the sweep runs
|
||||
// once anyway (a churning box is exactly the box that needs it), and the log SAYS the budget
|
||||
// ended it — "settled and found nothing" and "ran out of time" are different facts.
|
||||
//
|
||||
// RED-PROOF: remove the `time.Since(started) < bootReconcileBudget` loop condition and this test
|
||||
// hangs — the unbounded-loop shape §5 bans. Demonstrated in REPORT.md §4 (observed as a timeout).
|
||||
shrinkWindow(t, time.Millisecond, 3, 30*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
var buf strings.Builder
|
||||
|
||||
// Every frame differs, so `stable` can never reach stableFor.
|
||||
frames := make([][]stacks.Stack, 0, 200)
|
||||
for i := 0; i < 200; i++ {
|
||||
s := upStack("immich")
|
||||
s.Containers = make([]stacks.ContainerInfo, i%7) // container count changes every sample
|
||||
frames = append(frames, []stacks.Stack{s})
|
||||
}
|
||||
w := &windowStacks{frames: frames, cycle: true}
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
runBootReconcile(context.Background(), w, log.New(&buf, "", 0))
|
||||
close(done)
|
||||
}()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("runBootReconcile did not terminate on a forever-changing fleet — this is the " +
|
||||
"unbounded restart-loop shape the package's own boundary forbids")
|
||||
}
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("sweeps=%d, want exactly 1 after the budget expired", len(*seen))
|
||||
}
|
||||
if out := buf.String(); !strings.Contains(out, "budget") {
|
||||
t.Fatalf("the log does not say the BUDGET ended the window, so a churning boot reads like a "+
|
||||
"quiet one:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_SettledPathSaysSettled(t *testing.T) {
|
||||
shrinkWindow(t, time.Millisecond, 2, 500*time.Millisecond)
|
||||
captureSweep(t)
|
||||
var buf strings.Builder
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("docmost")}}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(&buf, "", 0))
|
||||
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "settled") {
|
||||
t.Fatalf("a settled window must say so — otherwise it is indistinguishable from a budget "+
|
||||
"expiry:\n%s", out)
|
||||
}
|
||||
if strings.Contains(out, "budget") {
|
||||
t.Fatalf("a settled window must NOT claim the budget ended it:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_CancelledContextStopsImmediately(t *testing.T) {
|
||||
shrinkWindow(t, time.Millisecond, 3, time.Second)
|
||||
seen := captureSweep(t)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
runBootReconcile(ctx, &windowStacks{frames: [][]stacks.Stack{{upStack("x")}}}, log.New(io.Discard, "", 0))
|
||||
if len(*seen) != 0 {
|
||||
t.Fatalf("the sweep ran %d times on a cancelled context, want 0", len(*seen))
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group C / Scenario D — a customer's Stop survives the WIDENED window -------------------------
|
||||
|
||||
func TestBootWindow_CustomerStoppedAppSurvivesEveryPass(t *testing.T) {
|
||||
// THE REGRESSION THIS TASK COULD INTRODUCE. A longer window means more chances to resurrect an
|
||||
// app the customer deliberately stopped. It must survive the whole window — this drives the REAL
|
||||
// bootrecon sweep (not the captured stub), so the desired-state check is genuinely exercised.
|
||||
//
|
||||
// RED-PROOF: drop the DesiredStateStopped branch from isBootOrphan (make it fall through to the
|
||||
// running case) and this test fails with a start count of 1. Demonstrated in REPORT.md §4.
|
||||
shrinkWindow(t, time.Millisecond, 2, 200*time.Millisecond)
|
||||
|
||||
stopped := stacks.Stack{
|
||||
Name: "nextcloud", Deployed: true, State: stacks.StateStopped, Containers: nil,
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateStopped},
|
||||
}
|
||||
// The fleet churns around it, so the window runs many passes before settling.
|
||||
frames := [][]stacks.Stack{
|
||||
{stopped, settlingLate("immich")},
|
||||
{stopped, settlingLate("immich")},
|
||||
{stopped, settledDown("immich")},
|
||||
{stopped, upStack("immich")},
|
||||
{stopped, upStack("immich")},
|
||||
{stopped, upStack("immich")},
|
||||
}
|
||||
w := &windowStacks{frames: frames, onStart: func(w *windowStacks, _ string) {}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if n := w.starts["nextcloud"]; n != 0 {
|
||||
t.Fatalf("the customer-stopped app was started %d time(s) by the widened window — this is the "+
|
||||
"regression a longer window makes possible and it is the worst outcome available here", n)
|
||||
}
|
||||
}
|
||||
|
||||
// --- §8.3 — a late recovery is REPORTED, never hidden ---------------------------------------------
|
||||
|
||||
func TestRecordLateRecovery_WarnsWhenTheGraceHasAlreadyExpired(t *testing.T) {
|
||||
var buf strings.Builder
|
||||
lg := log.New(&buf, "", 0)
|
||||
// started far enough back that settle + elapsed exceeds the 90 s grace
|
||||
recordLateRecovery(lg, time.Now().Add(-(deadAppBootGrace + 10*time.Second)), bootrecon.Result{Recovered: []string{"immich"}})
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "LATE RECOVERY") || !strings.Contains(out, "immich") {
|
||||
t.Fatalf("a recovery past the dead-app grace must be reported by name — otherwise a stale "+
|
||||
"alarm stands with no counter-evidence (§8.3):\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecordLateRecovery_SilentInsideTheGrace(t *testing.T) {
|
||||
var buf strings.Builder
|
||||
recordLateRecovery(log.New(&buf, "", 0), time.Now(), bootrecon.Result{Recovered: []string{"immich"}})
|
||||
if buf.Len() != 0 {
|
||||
t.Fatalf("a recovery INSIDE the grace must stay silent — that is what makes a successful "+
|
||||
"recovery invisible to the customer:\n%s", buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecordLateRecovery_SilentWhenNothingRecovered(t *testing.T) {
|
||||
var buf strings.Builder
|
||||
recordLateRecovery(log.New(&buf, "", 0), time.Now().Add(-time.Hour), bootrecon.Result{})
|
||||
if buf.Len() != 0 {
|
||||
t.Fatalf("nothing was recovered, so there is nothing late to report:\n%s", buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
// --- The window's constants must fit the grace they are justified against -------------------------
|
||||
|
||||
func TestBootWindow_CommonCaseFitsInsideTheDeadAppGrace(t *testing.T) {
|
||||
// The comment on the window constants justifies them against deadAppBootGrace. A comment
|
||||
// asserting an invariant needs a test pinning it, or it is a wish.
|
||||
common := bootReconcileSettle + bootReconcileBudget + bootrecon.DefaultRetryDelay
|
||||
if common > deadAppBootGrace {
|
||||
t.Fatalf("settle(%s) + budget(%s) + one retry(%s) = %s exceeds the %s dead-app grace — the "+
|
||||
"COMMON case must stay silent, or every slow boot alerts",
|
||||
bootReconcileSettle, bootReconcileBudget, bootrecon.DefaultRetryDelay, common, deadAppBootGrace)
|
||||
}
|
||||
if bootReconcileSample <= 0 || bootReconcileStableFor < 2 {
|
||||
t.Fatalf("sample=%s stableFor=%d — one sample cannot distinguish 'settled' from 'sampled "+
|
||||
"between two docker events'", bootReconcileSample, bootReconcileStableFor)
|
||||
}
|
||||
}
|
||||
|
||||
// --- the sample must observe REALITY, not the Manager's cache ------------------------------------
|
||||
|
||||
func TestBootWindow_EverySampleRefreshesTheStatus(t *testing.T) {
|
||||
// FOUND BY LIVE VALIDATION, not review. GetStacks() returns the Manager's in-memory map, which
|
||||
// the scheduler refreshes on its own 10 s cadence. Sampling every 5 s WITHOUT refreshing means two
|
||||
// consecutive samples can be identical because the cache did not update — so the window declares
|
||||
// "settled" on stale data and sweeps on a picture of the box from up to 10 s ago. On 9201 a
|
||||
// container removed ~5 s before the window closed was still in the sampled fleet, and the sweep
|
||||
// logged "no boot-orphaned apps" for an app that had none.
|
||||
//
|
||||
// RED-PROOF: delete the `_ = mgr.RefreshStatus()` line from sampleBootFleet and this test fails
|
||||
// with refreshes=0. Demonstrated in REPORT.md §4.
|
||||
shrinkWindow(t, time.Millisecond, 3, 500*time.Millisecond)
|
||||
captureSweep(t)
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("immich")}}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if w.refreshes < 3 {
|
||||
t.Fatalf("the window refreshed %d time(s) for %d samples — every sample must observe reality, "+
|
||||
"or 'settled' can mean 'the cache did not update'", w.refreshes, w.calls)
|
||||
}
|
||||
// calls includes ONE extra GetStacks from the captured sweep itself, which does not sample.
|
||||
if w.refreshes != w.calls-1 {
|
||||
t.Fatalf("refreshes=%d but samples=%d — each sample must refresh exactly once before reading",
|
||||
w.refreshes, w.calls-1)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_RefreshErrorDoesNotStopTheWindow(t *testing.T) {
|
||||
// A boot window that cannot reach docker is exactly when a stale verdict is most dangerous, but
|
||||
// giving up entirely would leave the sweep un-run. Degrade, do not abort.
|
||||
shrinkWindow(t, time.Millisecond, 2, 200*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("immich")}}, refreshErr: errRefresh{}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("sweeps=%d, want 1 — a refresh error must not abort the window", len(*seen))
|
||||
}
|
||||
}
|
||||
|
||||
type errRefresh struct{}
|
||||
|
||||
func (errRefresh) Error() string { return "docker unreachable" }
|
||||
@@ -0,0 +1,125 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/web"
|
||||
)
|
||||
|
||||
// v0.164.0: classifyRunStates is the single fix-3 derivation point. A deliberate user stop
|
||||
// (StateStopped) must NOT alarm — it is excluded from both the banner dead-list and the notifier
|
||||
// Down-set — while every genuine fault (StateExited / StateDegraded) keeps alerting byte-identically.
|
||||
// Invariants behind the suppression are documented at classifyRunStates (I1: compose down ⇒ zero
|
||||
// containers ⇒ StateStopped; I2: P2 census — all catalog services unless-stopped ⇒ faults never rest
|
||||
// at stopped).
|
||||
|
||||
func stack(name string, st stacks.ContainerState, deployed, deploying bool) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name,
|
||||
Meta: stacks.Metadata{DisplayName: name},
|
||||
State: st,
|
||||
Deployed: deployed,
|
||||
Deploying: deploying,
|
||||
}
|
||||
}
|
||||
|
||||
func downByName(states []notify.AppRunState) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
for _, s := range states {
|
||||
m[s.Name] = s.Down
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func deadNames(dead []web.DeadApp) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
for _, d := range dead {
|
||||
m[d.Name] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Group A (Scenario A) — suppression. Over a [running, stopped, exited, degraded] fixture, the dead
|
||||
// list is EXACTLY {exited, degraded} and the Down flags are {false, false, true, true}: the stopped
|
||||
// app is silent, the two faults still alarm.
|
||||
//
|
||||
// COMPANION red-proof: revert the filter to bare `stacks.IsDownState(st.State)` (drop the
|
||||
// `&& st.State != stacks.StateStopped` guard) → stopped reports Down=true and enters the dead list →
|
||||
// both the dead-set and the Down-flag assertions below fail. (Verified by hand-editing the seam.)
|
||||
func TestClassifyRunStates_StoppedIsSuppressed(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("radarr", stacks.StateRunning, true, false),
|
||||
stack("cwa", stacks.StateStopped, true, false),
|
||||
stack("immich", stacks.StateExited, true, false),
|
||||
stack("nextcloud", stacks.StateDegraded, true, false),
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
|
||||
gotDead := deadNames(dead)
|
||||
if len(gotDead) != 2 || !gotDead["immich"] || !gotDead["nextcloud"] {
|
||||
t.Fatalf("dead list must be exactly {immich(exited), nextcloud(degraded)}, got %+v", dead)
|
||||
}
|
||||
if gotDead["cwa"] {
|
||||
t.Errorf("a deliberately stopped app must NOT be in the dead list (no banner)")
|
||||
}
|
||||
if gotDead["radarr"] {
|
||||
t.Errorf("a running app must never be in the dead list")
|
||||
}
|
||||
|
||||
down := downByName(states)
|
||||
want := map[string]bool{"radarr": false, "cwa": false, "immich": true, "nextcloud": true}
|
||||
if len(down) != len(want) {
|
||||
t.Fatalf("every deployed app must have a run state, got %+v", down)
|
||||
}
|
||||
for name, w := range want {
|
||||
if down[name] != w {
|
||||
t.Errorf("Down[%s] = %v, want %v (stopped ⇒ false ⇒ no app_start_failed event)", name, down[name], w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Group B (Scenario B) — fault parity. With only exited + degraded present, BOTH surface in the dead
|
||||
// list AND both report Down=true — byte-identical to v0.163.1 for every non-stopped down state. The
|
||||
// suppression touches stopped and nothing else.
|
||||
func TestClassifyRunStates_FaultParity(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("immich", stacks.StateExited, true, false),
|
||||
stack("nextcloud", stacks.StateDegraded, true, false),
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
|
||||
gotDead := deadNames(dead)
|
||||
if len(gotDead) != 2 || !gotDead["immich"] || !gotDead["nextcloud"] {
|
||||
t.Fatalf("both faults must appear in the dead list, got %+v", dead)
|
||||
}
|
||||
down := downByName(states)
|
||||
if !down["immich"] || !down["nextcloud"] {
|
||||
t.Fatalf("both faults must report Down=true, got %+v", down)
|
||||
}
|
||||
// State strings must ride through to the banner unchanged (banner shows "(exited)"/"(degraded)").
|
||||
byName := map[string]string{}
|
||||
for _, d := range dead {
|
||||
byName[d.Name] = d.State
|
||||
}
|
||||
if byName["immich"] != string(stacks.StateExited) || byName["nextcloud"] != string(stacks.StateDegraded) {
|
||||
t.Errorf("dead-app State must carry the raw aggregate state, got %+v", byName)
|
||||
}
|
||||
}
|
||||
|
||||
// Deploying and undeployed stacks are skipped entirely (unchanged fix-3 behavior).
|
||||
func TestClassifyRunStates_SkipsDeployingAndUndeployed(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("mid", stacks.StateDeploying, true, true), // mid-deploy → skipped
|
||||
stack("gone", stacks.StateExited, false, false), // not deployed → skipped
|
||||
}
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
if len(dead) != 0 || len(states) != 0 {
|
||||
t.Fatalf("deploying and undeployed stacks must be skipped, got dead=%+v states=%+v", dead, states)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// C9-F2 — a SUSTAINED `restarting` is a crash loop and must alarm; a BRIEF one must not.
|
||||
//
|
||||
// The defect: `IsDownState` excludes `restarting` as "self-recovering", but for the catalog's
|
||||
// standard `restart: unless-stopped` Docker retries forever, so a crash loop sat in `restarting`
|
||||
// indefinitely and was counted as working. Campaign 9 watched docmost loop for nine minutes
|
||||
// (restartcount 18) while the F-OBS heartbeat printed "4 deployed app(s) evaluated, 0 currently down".
|
||||
//
|
||||
// The whole design tension is that B must keep passing while A does: an alarm that fires on every
|
||||
// deploy is one the operator learns to ignore.
|
||||
|
||||
// restartingSince builds a deployed stack that has been restarting since `since`.
|
||||
func restartingSince(name string, since time.Time) stacks.Stack {
|
||||
s := stack(name, stacks.StateRestarting, true, false)
|
||||
s.RestartingSince = since
|
||||
return s
|
||||
}
|
||||
|
||||
// SCENARIO A — a crash loop alarms. A stack restarting for longer than the threshold enters BOTH the
|
||||
// banner dead-list and the notifier Down-set, so app_start_failed can fire.
|
||||
//
|
||||
// RED-PROOF (observed): drop `|| crashLooping` from the `down` expression in classifyRunStates →
|
||||
//
|
||||
// crashloop_classify_test.go:52: docmost is NOT in the Down-set — a crash loop is silent (this is C9-F2)
|
||||
// crashloop_classify_test.go:55: docmost is NOT in the banner dead-list
|
||||
func TestClassifyRunStates_SustainedRestartingAlarms(t *testing.T) {
|
||||
now := time.Now()
|
||||
sts := []stacks.Stack{
|
||||
stack("paperless-ngx", stacks.StateRunning, true, false),
|
||||
restartingSince("docmost", now.Add(-9*time.Minute)), // the Campaign 9 observation, exactly
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, now)
|
||||
|
||||
if !downByName(states)["docmost"] {
|
||||
t.Errorf("docmost is NOT in the Down-set — a crash loop is silent (this is C9-F2)")
|
||||
}
|
||||
if !deadNames(dead)["docmost"] {
|
||||
t.Errorf("docmost is NOT in the banner dead-list")
|
||||
}
|
||||
if downByName(states)["paperless-ngx"] {
|
||||
t.Errorf("a healthy app was dragged down with it")
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO B — a normal deploy or update does NOT alarm. `docker compose up -d` passes through
|
||||
// restarting; alarming there would page the operator on every routine operation, fleet-wide.
|
||||
//
|
||||
// This is the test that must fail against the naive fix. RED-PROOF (observed): add StateRestarting
|
||||
// to IsDownState instead of using the threshold →
|
||||
//
|
||||
// crashloop_classify_test.go:78: a BRIEFLY restarting app alarms — every deploy and update would page the operator
|
||||
func TestClassifyRunStates_BriefRestartingIsSilent(t *testing.T) {
|
||||
now := time.Now()
|
||||
sts := []stacks.Stack{
|
||||
restartingSince("mealie", now.Add(-30*time.Second)), // mid-deploy
|
||||
restartingSince("ghost", now.Add(-2*time.Minute)), // slow image pull, still normal
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, now)
|
||||
|
||||
for _, name := range []string{"mealie", "ghost"} {
|
||||
if downByName(states)[name] {
|
||||
t.Errorf("a BRIEFLY restarting app alarms (%s) — every deploy and update would page the operator", name)
|
||||
}
|
||||
}
|
||||
if len(dead) != 0 {
|
||||
t.Errorf("banner dead-list should be empty during normal restarts, got %v", deadNames(dead))
|
||||
}
|
||||
}
|
||||
|
||||
// The boundary itself, asserted from both sides so the threshold cannot drift silently.
|
||||
func TestCrashLooping_ThresholdBoundary(t *testing.T) {
|
||||
now := time.Now()
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
age time.Duration
|
||||
want bool
|
||||
}{
|
||||
{"just under the threshold", 4*time.Minute + 59*time.Second, false},
|
||||
{"exactly at the threshold", 5 * time.Minute, true},
|
||||
{"well past it", 30 * time.Minute, true},
|
||||
} {
|
||||
s := restartingSince("app", now.Add(-tc.age))
|
||||
if got := s.CrashLooping(now); got != tc.want {
|
||||
t.Errorf("%s: CrashLooping(age=%s) = %v, want %v", tc.name, tc.age, got, tc.want)
|
||||
}
|
||||
}
|
||||
|
||||
// A stack that is not restarting is never a crash loop, however old the stamp.
|
||||
s := stack("app", stacks.StateRunning, true, false)
|
||||
s.RestartingSince = now.Add(-time.Hour)
|
||||
if s.CrashLooping(now) {
|
||||
t.Error("a RUNNING stack reported as crash-looping — the state test is missing")
|
||||
}
|
||||
|
||||
// A zero stamp is "not yet observed restarting", never a crash loop — this is what makes the
|
||||
// first scan after a controller restart silent instead of alarming on everything at once.
|
||||
z := stack("app", stacks.StateRestarting, true, false)
|
||||
if z.CrashLooping(now) {
|
||||
t.Error("a zero RestartingSince reported as crash-looping — a controller restart would alarm fleet-wide")
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO C — R-97b's quiesce suppression still wins inside its window. A stack the backup stopped
|
||||
// and is restarting must stay silent while suppressed, even if its restarting run is old enough to
|
||||
// qualify. The window EXPIRES, so a genuinely dead app still alarms afterwards — proven by the
|
||||
// second half of this test.
|
||||
//
|
||||
// RED-PROOF (observed): drop `&& !quiesced[st.Name]` from the `down` expression →
|
||||
//
|
||||
// crashloop_classify_test.go:129: a quiesced stack alarms — every backup would page the customer
|
||||
func TestClassifyRunStates_QuiesceSuppressionBeatsCrashLoop(t *testing.T) {
|
||||
now := time.Now()
|
||||
sts := []stacks.Stack{restartingSince("docmost", now.Add(-9*time.Minute))}
|
||||
|
||||
// Inside the R-97b window.
|
||||
_, states := classifyRunStates(sts, map[string]bool{"docmost": true}, nil, now)
|
||||
if downByName(states)["docmost"] {
|
||||
t.Errorf("a quiesced stack alarms — every backup would page the customer")
|
||||
}
|
||||
|
||||
// Window expired (the stack is no longer reported as suppressed): the same stack must now alarm.
|
||||
dead, states := classifyRunStates(sts, nil, nil, now)
|
||||
if !downByName(states)["docmost"] {
|
||||
t.Errorf("suppression outlived its window — a genuinely dead app stayed silent (R-97b's own warning)")
|
||||
}
|
||||
if !deadNames(dead)["docmost"] {
|
||||
t.Errorf("suppression outlived its window for the banner too")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"log"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// F-OBS (Campaign 8): on a default `info`-level box there was NO positive observable that
|
||||
// `deadapp-check` had run. Its per-cycle scheduler line goes through Scheduler.dbg(), which is gated
|
||||
// on logging.level==debug and therefore never PRODUCED on a default box — so it could not even reach
|
||||
// the always-DEBUG ring — and a 30 s interval also puts the job on the scheduler's quiet path.
|
||||
//
|
||||
// "No alarms" was therefore indistinguishable from "the detector never ran", which is exactly the
|
||||
// fallacy this project now has a standing rule against, and it undermines confidence in the
|
||||
// F-CRIT-1 fix in the field.
|
||||
//
|
||||
// Scenario F — the observable must appear AT INFO LEVEL. These tests assert the emitted LINE, not
|
||||
// merely that a function was called; asserting the call would reproduce the original mistake.
|
||||
|
||||
// RED-PROOF: delete the logger.Printf in noteDeadAppScan (or drop the whole call from the job
|
||||
// closure) → every case below sees an empty buffer and this fails with
|
||||
// "no observable emitted at scan 20 — silence is indistinguishable from not running".
|
||||
func TestNoteDeadAppScan_EmitsAtInfoLevel(t *testing.T) {
|
||||
var buf bytes.Buffer
|
||||
lg := log.New(&buf, "", 0)
|
||||
|
||||
noteDeadAppScan(lg, deadAppHeartbeatEvery, 7, 2)
|
||||
|
||||
out := buf.String()
|
||||
if out == "" {
|
||||
t.Fatalf("no observable emitted at scan %d — silence is indistinguishable from not running", deadAppHeartbeatEvery)
|
||||
}
|
||||
if !strings.Contains(out, "[INFO]") {
|
||||
t.Errorf("the observable is not at INFO level, so a default `logging.level: info` box would never see it:\n%s", out)
|
||||
}
|
||||
if !strings.Contains(out, "[deadapp]") {
|
||||
t.Errorf("the observable does not identify the check that produced it:\n%s", out)
|
||||
}
|
||||
// it must carry WHAT IT SAW, not just "I ran" — an operator needs to distinguish
|
||||
// "running and everything is up" from "running and 2 apps are down".
|
||||
for _, want := range []string{"scans since boot", "evaluated", "currently down"} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("the observable omits %q — it proves the check ran but not what it found:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// It must NOT be a line per run. At a 30 s cadence that is 2880 lines/day, which is precisely why
|
||||
// the original author chose silence — so a fix that floods is not a fix.
|
||||
//
|
||||
// RED-PROOF: change the guard to `scans%1 != 0` (i.e. emit every run) → this fails with
|
||||
// "emitted 60 lines across 60 scans — that is the flood that made silence attractive".
|
||||
func TestNoteDeadAppScan_IsASummaryNotAFlood(t *testing.T) {
|
||||
var buf bytes.Buffer
|
||||
lg := log.New(&buf, "", 0)
|
||||
|
||||
const scans = 60
|
||||
for i := 1; i <= scans; i++ {
|
||||
noteDeadAppScan(lg, i, 3, 0)
|
||||
}
|
||||
|
||||
got := strings.Count(buf.String(), "[deadapp] check alive")
|
||||
want := scans / deadAppHeartbeatEvery
|
||||
if got == scans {
|
||||
t.Fatalf("emitted %d lines across %d scans — that is the flood that made silence attractive", got, scans)
|
||||
}
|
||||
if got != want {
|
||||
t.Errorf("emitted %d heartbeat lines across %d scans, want %d (one per %d)", got, scans, want, deadAppHeartbeatEvery)
|
||||
}
|
||||
}
|
||||
|
||||
// The cadence must be frequent enough that a STALLED detector is obvious well inside the 180 s alarm
|
||||
// grace this check feeds. 20 scans x 30 s = 10 min; if someone widens it to hours the observable
|
||||
// stops being useful as a liveness signal, and this is the tripwire.
|
||||
func TestDeadAppHeartbeatEvery_StaysUsefulAsALivenessSignal(t *testing.T) {
|
||||
const scanInterval = 30 // seconds, matching sched.Every("deadapp-check", 30*time.Second, ...)
|
||||
periodSec := deadAppHeartbeatEvery * scanInterval
|
||||
if periodSec > 15*60 {
|
||||
t.Errorf("heartbeat period is %ds (>15min) — too sparse to notice a stalled detector", periodSec)
|
||||
}
|
||||
if deadAppHeartbeatEvery < 2 {
|
||||
t.Errorf("heartbeat every %d scans is a per-run flood", deadAppHeartbeatEvery)
|
||||
}
|
||||
}
|
||||
|
||||
// Off-cadence scans stay quiet, and a nil logger is tolerated (the job closure must never panic).
|
||||
func TestNoteDeadAppScan_QuietOffCadenceAndNilSafe(t *testing.T) {
|
||||
var buf bytes.Buffer
|
||||
lg := log.New(&buf, "", 0)
|
||||
noteDeadAppScan(lg, deadAppHeartbeatEvery-1, 1, 0)
|
||||
if buf.Len() != 0 {
|
||||
t.Errorf("emitted off-cadence:\n%s", buf.String())
|
||||
}
|
||||
noteDeadAppScan(nil, deadAppHeartbeatEvery, 1, 0) // must not panic
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// F-CRIT-1 cause 2 (Campaign 8): `classifyRunStates` whitelisted StateStopped on invariant I1
|
||||
// ("StateStopped means the USER stopped it"). The quiesce loop broke I1 by stopping stacks via the
|
||||
// same `docker compose down` path, so a stack quiesce stopped and then FAILED to restart was also
|
||||
// StateStopped — and was whitelisted into total silence. Live evidence: a customer app dead
|
||||
// indefinitely, no banner, no event, no email, while the dead-app scanner ran 11 times over it.
|
||||
//
|
||||
// The two cases are byte-identical on the Docker side. The ONLY thing that separates them is that
|
||||
// the quiesce loop knows it tried to restart and could not — `failedRestart` is that knowledge.
|
||||
|
||||
// Scenario A (cause 2) — a stack quiesce failed to restart MUST alarm, despite being StateStopped.
|
||||
//
|
||||
// RED-PROOF: restore the unconditional whitelist (`down := IsDownState(st.State) &&
|
||||
// st.State != stacks.StateStopped && !quiesced[st.Name]`) → immich reports Down=false and stays out
|
||||
// of the dead list, and this fails with "a stack that FAILED to restart is silent".
|
||||
func TestClassifyRunStates_FailedRestartAlarmsDespiteStateStopped(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("bookstack", stacks.StateRunning, true, false),
|
||||
stack("immich", stacks.StateStopped, true, false), // quiesce stopped it; restart FAILED
|
||||
}
|
||||
failed := map[string]bool{"immich": true}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, failed, time.Now())
|
||||
|
||||
if !downByName(states)["immich"] {
|
||||
t.Error("a stack that FAILED to restart is silent (Down=false) — this is F-CRIT-1")
|
||||
}
|
||||
if !deadNames(dead)["immich"] {
|
||||
t.Error("a stack that FAILED to restart is absent from the dashboard dead-list — this is F-CRIT-1")
|
||||
}
|
||||
if downByName(states)["bookstack"] {
|
||||
t.Error("a healthy running stack was marked down")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B — a DELIBERATE user stop must still be silent. This pins v0.164.0 and is what stops
|
||||
// the fix above from becoming a regression.
|
||||
//
|
||||
// RED-PROOF: make the whitelist unconditional in the other direction (drop the `&& !failedRestart`
|
||||
// term, i.e. treat every StateStopped as a failed restart) → cwa alarms and this fails with
|
||||
// "a deliberate user stop alarmed".
|
||||
func TestClassifyRunStates_UserStopStillSilent(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("cwa", stacks.StateStopped, true, false), // the user stopped this from the UI
|
||||
stack("immich", stacks.StateStopped, true, false),
|
||||
}
|
||||
// only immich failed to restart; cwa was never touched by a quiesce
|
||||
failed := map[string]bool{"immich": true}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, failed, time.Now())
|
||||
down := downByName(states)
|
||||
|
||||
if down["cwa"] || deadNames(dead)["cwa"] {
|
||||
t.Error("a deliberate user stop alarmed — that is the v0.164.0 regression this must not reintroduce")
|
||||
}
|
||||
if !down["immich"] {
|
||||
t.Error("the failed restart went silent")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B, stronger form — with NO failed restarts at all, behaviour is byte-identical to
|
||||
// v0.164.0: every StateStopped is silent.
|
||||
func TestClassifyRunStates_NoFailedRestartsIsV0164Behaviour(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("radarr", stacks.StateRunning, true, false),
|
||||
stack("cwa", stacks.StateStopped, true, false),
|
||||
stack("immich", stacks.StateExited, true, false),
|
||||
stack("nextcloud", stacks.StateDegraded, true, false),
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
down := downByName(states)
|
||||
|
||||
if down["cwa"] {
|
||||
t.Error("stopped alarmed with no failed restarts — v0.164.0 behaviour broken")
|
||||
}
|
||||
if !down["immich"] || !down["nextcloud"] {
|
||||
t.Error("a genuine fault (exited/degraded) stopped alarming")
|
||||
}
|
||||
if got := len(deadNames(dead)); got != 2 {
|
||||
t.Errorf("dead list has %d entries, want exactly {immich, nextcloud}", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C — during the R-97b grace window the stack is suppressed even if its restart failed.
|
||||
// The grace exists so a slow-starting app is not called dead; it EXPIRES, and the alarm follows.
|
||||
//
|
||||
// RED-PROOF: drop the `&& !quiesced[st.Name]` term → the app alarms mid-restart on every normal
|
||||
// backup, which is the false-alarm R-97b was built to remove.
|
||||
func TestClassifyRunStates_GraceWindowStillSuppresses(t *testing.T) {
|
||||
sts := []stacks.Stack{stack("immich", stacks.StateStopped, true, false)}
|
||||
quiesced := map[string]bool{"immich": true} // still inside quiesceAlarmGrace
|
||||
failed := map[string]bool{"immich": true} // and we already know the restart failed
|
||||
|
||||
dead, states := classifyRunStates(sts, quiesced, failed, time.Now())
|
||||
|
||||
if downByName(states)["immich"] {
|
||||
t.Error("alarmed while still inside the grace window — R-97b Scenario E broken")
|
||||
}
|
||||
if len(dead) != 0 {
|
||||
t.Errorf("dead list not empty during grace: %v", deadNames(dead))
|
||||
}
|
||||
}
|
||||
|
||||
// An undeployed or mid-deploy stack is never classified, failed restart or not.
|
||||
func TestClassifyRunStates_UndeployedIgnored(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("ghost", stacks.StateStopped, false, false),
|
||||
stack("deploying", stacks.StateStopped, true, true),
|
||||
}
|
||||
dead, states := classifyRunStates(sts, nil, map[string]bool{"ghost": true, "deploying": true}, time.Now())
|
||||
if len(dead) != 0 || len(states) != 0 {
|
||||
t.Errorf("undeployed/deploying stacks were classified: dead=%v states=%v", deadNames(dead), states)
|
||||
}
|
||||
}
|
||||
+1390
-92
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,30 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
|
||||
)
|
||||
|
||||
// R-97a — REACHABILITY, not behaviour.
|
||||
//
|
||||
// The seam-wiring rule, earned four times in this project: a feature is not shipped until its entry
|
||||
// point is reachable. `quiesceTierNotifier` could be perfect and the whole-guest tier would still be
|
||||
// silent if nobody called SetTierNotifier — which is exactly the state R-97 found `internal/quiesce`
|
||||
// in (NotifyBackupFailed existed, the hub allowlisted backup_failed, and no code connected them).
|
||||
//
|
||||
// This asserts the adapter SATISFIES the interface the loop requires. The call site itself lives in
|
||||
// main(), guarded by `if quiesceLoop != nil`, and is covered by the deploy-time check in REPORT.md.
|
||||
func TestQuiesceTierNotifierIsWired(t *testing.T) {
|
||||
var _ quiesce.TierNotifier = quiesceTierNotifier{}
|
||||
|
||||
// And it must not panic on a nil notifier — main() constructs it with a real one, but a future
|
||||
// refactor that reorders startup must fail loudly here rather than at 03:00 on a customer box.
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
t.Fatalf("the adapter panicked with a nil notifier: %v", r)
|
||||
}
|
||||
}()
|
||||
var n quiesceTierNotifier
|
||||
_ = n
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
package main
|
||||
|
||||
import "gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
|
||||
|
||||
// COMPILE-TIME WITNESSES for OPTIONAL interfaces satisfied by a RUNTIME type assertion.
|
||||
//
|
||||
// Moved here from a _test.go file (R-88 Part 2) on purpose: a witness in a test fires on `go test`
|
||||
// and `go vet`, but NOT on `go build` alone. The failure it guards against — a signature change that
|
||||
// silently breaks an interface nobody checks at compile time — is exactly the kind that gets pushed
|
||||
// by a build-only step.
|
||||
//
|
||||
// THE INCIDENT THIS PREVENTS, which already happened once: when `TieredBackend.DueFor` gained a
|
||||
// return value during R-88 Part 2, `quiesceBackend` stopped satisfying the interface and the whole
|
||||
// repo still BUILT AND VETTED CLEAN, because `resolveDueTiers` only ever asserts it at runtime
|
||||
// (`l.backend.(TieredBackend)`). A failed assertion silently falls back to the untargeted
|
||||
// single-tier path — so every box would have quietly lost R-82's multi-tier backups, with no error
|
||||
// anywhere. It was caught by accident, not by the toolchain.
|
||||
//
|
||||
// THIS DOES NOT MAKE THE INTERFACE REQUIRED. Optionality is deliberate: it is what lets a new
|
||||
// controller meet an old agent, and what `resolveDueTiers` degrades through on purpose. The witness
|
||||
// pins the IMPLEMENTATION, not the CONTRACT.
|
||||
var _ quiesce.TieredBackend = quiesceBackend{}
|
||||
@@ -95,7 +95,7 @@ monitoring:
|
||||
hub:
|
||||
enabled: true # Enable central reporting
|
||||
url: "https://hub.felhom.eu" # Hub API endpoint
|
||||
api_key: "094091de545ce28795c47ac2158fc30750db5c24a621c49329b001ee8db57fb8" # Shared secret for authentication
|
||||
api_key: "<hub-issued-per-customer-key>" # From the hub-generated config; never commit a real key
|
||||
push_interval: "15m" # How often to push reports
|
||||
|
||||
# --- Self-update ---
|
||||
|
||||
@@ -5,6 +5,7 @@ go 1.24.0
|
||||
require (
|
||||
github.com/emersion/go-sasl v0.0.0-20241020182733-b788ff22d5a6
|
||||
github.com/emersion/go-smtp v0.24.0
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e
|
||||
golang.org/x/crypto v0.31.0
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
modernc.org/sqlite v1.45.0
|
||||
|
||||
@@ -16,6 +16,8 @@ github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOF
|
||||
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e h1:MRM5ITcdelLK2j1vwZ3Je0FKVCfqOLp5zO6trqMLYs0=
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e/go.mod h1:XV66xRDqSt+GTGFMVlhk3ULuV0y9ZmzeVGR4mloJI3M=
|
||||
golang.org/x/crypto v0.31.0 h1:ihbySMvVjLAeSH1IbfcRTkD/iNscyz8rGzjF/E5hV6U=
|
||||
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
|
||||
golang.org/x/exp v0.0.0-20251023183803-a4bb9ffd2546 h1:mgKeJMpvi0yx/sU5GsxQ7p6s2wtOnGAHZWCHUM4KGzY=
|
||||
@@ -27,6 +29,8 @@ golang.org/x/sync v0.17.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI=
|
||||
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.37.0 h1:fdNQudmxPjkdUTPnLn5mdQv7Zwvbvpaxqs831goi9kQ=
|
||||
golang.org/x/sys v0.37.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
|
||||
golang.org/x/term v0.27.0 h1:WP60Sv1nlK1T6SupCHbXzSaN0b9wUmsPoRS9b61A23Q=
|
||||
golang.org/x/term v0.27.0/go.mod h1:iMsnZpn0cago0GOrHO2+Y7u7JPn5AylBrcoWkElMTSM=
|
||||
golang.org/x/tools v0.38.0 h1:Hx2Xv8hISq8Lm16jvBZ2VQf+RLmbd7wVUsALibYI/IQ=
|
||||
golang.org/x/tools v0.38.0/go.mod h1:yEsQ/d/YK8cjh0L6rZlY8tgtlKiBNTL14pGDJPJpYQs=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
# felhom-samba — the LAN SMB-sharing infra image for felhom-controller (R-7 slice 1).
|
||||
#
|
||||
# DUMB BY DESIGN: /etc/samba/smb.conf is bind-mounted READ-ONLY by the controller, which
|
||||
# owns all rendering. This image templates nothing and bakes NO share name and NO password.
|
||||
# The three-daemon discovery stack is the spike verdict
|
||||
# (felhom.eu/documentation/audits/SPIKE-lan-discovery-2026-07-18.md, S4/S4b):
|
||||
# - smbd : the SMB/CIFS server (445)
|
||||
# - nmbd : NetBIOS name service — REQUIRED alongside wsdd. wsdd-only makes the box visible
|
||||
# in Explorer but the double-click fails 0x80070035 (no flat-name resolution);
|
||||
# nmbd is what makes \\<NAME> resolve + mount (S4b, proven live).
|
||||
# - wsdd : WS-Discovery, so the box appears in Windows Explorer's Network view.
|
||||
# - avahi : mDNS/Bonjour (v1.1.0) — THE macOS path. Windows and macOS do not share a
|
||||
# discovery mechanism, and nmbd does not cover the Mac: captured live on
|
||||
# 2026-07-20, macOS broadcasts a correct NBNS query for FELHOM<20>, the box
|
||||
# answers correctly in 140us (flags 0x8580, RCODE=0, the right address), and
|
||||
# macOS REFUSES TO ACT ON IT — no TCP follows. NetBIOS feeds legacy browsing
|
||||
# there, not smb:// URL resolution. With mDNS, `smb://<NAME>.local` connects
|
||||
# immediately — PROVEN live from a Mac on 2026-07-20.
|
||||
# NOT proven: automatic appearance in the Finder sidebar. The _smb._tcp record
|
||||
# is published and answers browse queries on the wire, but the test Mac's
|
||||
# sidebar stayed empty (it had no Network/Bonjour section shown at all, which
|
||||
# is a Finder Settings -> Sidebar toggle). Treat sidebar discovery as an OPEN
|
||||
# question, not a shipped feature.
|
||||
# Evidence: felhom.eu/documentation/audits/DIAG-sharing-2026-07-20.md.
|
||||
FROM alpine:3.21@sha256:48b0309ca019d89d40f670aa1bc06e426dc0931948452e8491e3d65087abc07d
|
||||
|
||||
# samba = smbd + nmbd + smbpasswd/testparm (meta-package proven installable in the spike);
|
||||
# wsdd = WS-Discovery daemon; tini = a proper PID1 to reap nmbd/wsdd/avahi and forward signals;
|
||||
# avahi + dbus = mDNS/Bonjour (avahi-daemon talks to the system bus, so dbus is not optional).
|
||||
RUN apk add --no-cache samba wsdd tini avahi dbus \
|
||||
&& rm -rf /var/cache/apk/* \
|
||||
&& rm -f /etc/samba/smb.conf \
|
||||
&& rm -f /etc/avahi/services/*.service
|
||||
|
||||
# passdb on a named volume → the household SMB password survives container recreation
|
||||
# (share add/remove re-renders + `compose up -d`, which recreates the container).
|
||||
VOLUME ["/var/lib/samba"]
|
||||
|
||||
COPY entrypoint.sh /entrypoint.sh
|
||||
RUN chmod +x /entrypoint.sh
|
||||
|
||||
ENTRYPOINT ["/sbin/tini", "--", "/entrypoint.sh"]
|
||||
@@ -0,0 +1,83 @@
|
||||
#!/bin/sh
|
||||
# felhom-samba entrypoint (R-7 slice 1). A dumb supervisor: smb.conf is bind-mounted
|
||||
# READ-ONLY by the controller, so nothing here templates config or bakes a secret. It
|
||||
# only ensures the household unix user exists (uid:gid 1000) and launches the three
|
||||
# discovery daemons. Verdict source: SPIKE-lan-discovery-2026-07-18 (S4/S4b).
|
||||
set -e
|
||||
|
||||
FELHOM_UID="${FELHOM_UID:-1000}"
|
||||
FELHOM_GID="${FELHOM_GID:-1000}"
|
||||
SERVER_NAME="${FELHOM_SERVER_NAME:-FELHOM}"
|
||||
IFACE="${FELHOM_IFACE:-eth0}"
|
||||
|
||||
# Household group/user at uid:gid 1000 — files written over SMB then match the app +
|
||||
# backup ownership convention (smb.conf sets `force user = felhom` per share).
|
||||
if ! getent group "$FELHOM_GID" >/dev/null 2>&1; then
|
||||
addgroup -g "$FELHOM_GID" felhom 2>/dev/null || true
|
||||
fi
|
||||
GRP_NAME="$(getent group "$FELHOM_GID" 2>/dev/null | cut -d: -f1)"
|
||||
[ -z "$GRP_NAME" ] && GRP_NAME=felhom
|
||||
if ! getent passwd "$FELHOM_UID" >/dev/null 2>&1; then
|
||||
adduser -D -H -u "$FELHOM_UID" -G "$GRP_NAME" -s /sbin/nologin felhom 2>/dev/null || true
|
||||
fi
|
||||
|
||||
mkdir -p /var/lib/samba/private /run/samba
|
||||
|
||||
# --- mDNS / Bonjour (v1.1.0) -------------------------------------------------------------
|
||||
# THE macOS path. Templated from SERVER_NAME rather than baked, so renaming the server in the
|
||||
# UI re-advertises under the new name on the next container recreate — a baked name would
|
||||
# leave the box answering to something the customer no longer sees anywhere.
|
||||
#
|
||||
# A STATIC service file, deliberately, rather than smbd's own `multicast dns register`: it
|
||||
# needs no line in smb.conf (which is bind-mounted READ-ONLY and owned by the controller's
|
||||
# renderer) and it lets us publish _device-info._tcp so the Finder shows a sensible icon
|
||||
# instead of a generic globe.
|
||||
mkdir -p /etc/avahi/services /run/dbus
|
||||
cat > /etc/avahi/avahi-daemon.conf <<CONF
|
||||
[server]
|
||||
host-name=${SERVER_NAME}
|
||||
use-ipv4=yes
|
||||
use-ipv6=no
|
||||
allow-interfaces=${IFACE}
|
||||
ratelimit-interval-usec=1000000
|
||||
ratelimit-burst=1000
|
||||
|
||||
[wide-area]
|
||||
enable-wide-area=no
|
||||
|
||||
[publish]
|
||||
publish-addresses=yes
|
||||
publish-hinfo=no
|
||||
publish-workstation=no
|
||||
CONF
|
||||
|
||||
cat > /etc/avahi/services/smb.service <<CONF
|
||||
<?xml version="1.0" standalone='no'?><!DOCTYPE service-group SYSTEM "avahi-service.dtd">
|
||||
<service-group>
|
||||
<name replace-wildcards="yes">%h</name>
|
||||
<service>
|
||||
<type>_smb._tcp</type>
|
||||
<port>445</port>
|
||||
</service>
|
||||
<service>
|
||||
<type>_device-info._tcp</type>
|
||||
<port>0</port>
|
||||
<txt-record>model=RackMac</txt-record>
|
||||
</service>
|
||||
</service-group>
|
||||
CONF
|
||||
|
||||
echo "[felhom-samba] launching nmbd + wsdd + avahi + smbd (server=${SERVER_NAME} iface=${IFACE} uid=${FELHOM_UID})"
|
||||
|
||||
# nmbd: NetBIOS flat-name resolution so \\<NAME> resolves and mounts on WINDOWS (the S4b fix).
|
||||
# It does NOT serve macOS — see the Dockerfile header for the captured proof.
|
||||
nmbd --daemon --no-process-group
|
||||
# wsdd: WS-Discovery so the box appears in Windows Explorer's Network view.
|
||||
wsdd -i "$IFACE" -4 -H 4 -s -n "$SERVER_NAME" -w WORKGROUP &
|
||||
# dbus + avahi: mDNS, so `smb://<NAME>.local` resolves and the box appears in the Finder sidebar.
|
||||
# Non-fatal on failure: sharing over an address still works, and refusing to start smbd because
|
||||
# a discovery daemon did not come up would turn a convenience gap into an outage.
|
||||
dbus-daemon --system --fork 2>/dev/null || echo "[felhom-samba] WARN: dbus failed to start — mDNS disabled"
|
||||
avahi-daemon --daemonize --no-drop-root 2>/dev/null || echo "[felhom-samba] WARN: avahi failed to start — mDNS disabled"
|
||||
# smbd in the foreground = the container's main process.
|
||||
exec smbd --foreground --no-process-group
|
||||
@@ -0,0 +1,130 @@
|
||||
package agentapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
)
|
||||
|
||||
// R-82 Slice B — the per-tier backup surface (agent >= v0.97.0).
|
||||
//
|
||||
// Every method here is ADDITIVE. The untargeted BackupDue/StartBackup/BackupStatus keep their exact
|
||||
// pre-R-82 meaning and are still the single-tier path used against an older agent.
|
||||
|
||||
// ErrTiersUnsupported reports that this agent does not serve GET /backup/tiers — it predates R-82.
|
||||
// It is the DESIGNED capability probe (the route 404s), not a fault. The caller MUST degrade to the
|
||||
// untargeted single-tier path and still take a backup; concluding "nothing to do" from it would
|
||||
// silently stop backups during a fleet rollout.
|
||||
var ErrTiersUnsupported = errors.New("agentapi: agent does not serve /backup/tiers (pre-R-82)")
|
||||
|
||||
// BackupTierInfo is one advertised tier.
|
||||
type BackupTierInfo struct {
|
||||
Target string `json:"target"`
|
||||
CadenceSeconds int64 `json:"cadence_seconds"`
|
||||
Primary bool `json:"primary"`
|
||||
}
|
||||
|
||||
// TiersResponse mirrors the agent's GET /backup/tiers payload.
|
||||
type TiersResponse struct {
|
||||
VMID int `json:"vmid"`
|
||||
Tiers []BackupTierInfo `json:"tiers"`
|
||||
}
|
||||
|
||||
// BackupTiers lists the agent's backup tiers, primary first.
|
||||
// Returns ErrTiersUnsupported (wrapped) on a pre-R-82 agent — key on it with errors.Is.
|
||||
func (c *Client) BackupTiers(ctx context.Context) (TiersResponse, error) {
|
||||
var out TiersResponse
|
||||
body, err := c.get(ctx, "/backup/tiers")
|
||||
if err != nil {
|
||||
var se *StatusError
|
||||
if errors.As(err, &se) && se.Code == http.StatusNotFound {
|
||||
return out, ErrTiersUnsupported
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/tiers: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// targetQuery renders the ?target= suffix. An EMPTY target yields an empty string, so the caller
|
||||
// hits the untargeted route byte-for-byte — that is what keeps the pre-R-82 contract intact when
|
||||
// this client talks to an older agent.
|
||||
func targetQuery(target string) string {
|
||||
if target == "" {
|
||||
return ""
|
||||
}
|
||||
return "?target=" + url.QueryEscape(target)
|
||||
}
|
||||
|
||||
// BackupDueFor reports whether THIS TIER is due. A fresh backup on another tier must not satisfy it
|
||||
// — that filtering happens agent-side (latestSuccessfulBackupForTarget); this just asks per tier.
|
||||
func (c *Client) BackupDueFor(ctx context.Context, target string) (DueResponse, error) {
|
||||
var out DueResponse
|
||||
body, err := c.get(ctx, "/backup/due"+targetQuery(target))
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/due (target %q): %w", target, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// StartBackupFor enqueues a backup of this guest ON THE GIVEN TIER.
|
||||
func (c *Client) StartBackupFor(ctx context.Context, target string) (BackupResponse, error) {
|
||||
var out BackupResponse
|
||||
body, err := c.post(ctx, "/backup"+targetQuery(target), struct{}{})
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode POST /backup (target %q): %w", target, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// BackupStatusFor reports THIS TIER's current/last job phase. Jobs are keyed per tier agent-side,
|
||||
// so polling the wrong target would report a different tier's progress.
|
||||
func (c *Client) BackupStatusFor(ctx context.Context, target string) (StatusResponse, error) {
|
||||
var out StatusResponse
|
||||
body, err := c.get(ctx, "/backup/status"+targetQuery(target))
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/status (target %q): %w", target, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// SetBackupTargetResponse mirrors POST /backup/target (agent >= v0.113.0).
|
||||
type SetBackupTargetResponse struct {
|
||||
Target string `json:"target"`
|
||||
Where string `json:"where"`
|
||||
// RestartRequired is always true on success: the agent builds its tiers once at daemon start, so
|
||||
// the move needs a restart. The agent deliberately does NOT restart itself — restarting with a
|
||||
// backup in flight cancels the wait and records a spurious tier failure for a backup that actually
|
||||
// succeeded. The RESTART IS THE OPERATOR'S, behind an immediate in-flight check.
|
||||
RestartRequired bool `json:"restart_required"`
|
||||
}
|
||||
|
||||
// SetBackupTarget moves the primary whole-guest backup tier onto the drive at raw host mount `where`.
|
||||
// Creates the storage and grants the agent access as one ordered operation.
|
||||
func (c *Client) SetBackupTarget(ctx context.Context, where string) (SetBackupTargetResponse, error) {
|
||||
var out SetBackupTargetResponse
|
||||
// vmid is deliberately omitted: the agent derives the guest from the token and scopedFromBody
|
||||
// treats an absent vmid as "use the token's" — the same shape as AssignDisk/GuestAttach.
|
||||
body, err := c.post(ctx, "/backup/target", map[string]string{"where": where})
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/target: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -16,9 +16,14 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/logx"
|
||||
)
|
||||
|
||||
// Client talks to one agent local-API endpoint with a pinned leaf + bearer token.
|
||||
@@ -28,6 +33,47 @@ type Client struct {
|
||||
hc *http.Client
|
||||
// features caches capability-probe verdicts for Supports (features.go).
|
||||
features SupportCache
|
||||
// verMu guards lastAgentVersion — the most recent STRICTLY-VALIDATED X-Felhom-Agent-Version
|
||||
// seen on any agent response (v0.82.0 version channel). "" = never seen (pre-0.82 agent) →
|
||||
// Supports falls back to the route probe.
|
||||
verMu sync.Mutex
|
||||
lastAgentVersion string
|
||||
// logger is the optional per-call DEBUG trace sink (v0.116.0 observability — the
|
||||
// capture ring holds these even at logging.level=info). nil = silent (unchanged).
|
||||
logger *log.Logger
|
||||
}
|
||||
|
||||
// SetLogger wires the optional per-call DEBUG trace logger (method, path, status,
|
||||
// duration + agent-version changes — never bodies or tokens).
|
||||
func (c *Client) SetLogger(l *log.Logger) { c.logger = l }
|
||||
|
||||
// reAgentVersion is the bare-semver shape the publish pipeline enforces (publish-agent.sh) — the
|
||||
// ONLY header values trusted for capability comparison. Anything else (garbage, "dev", suffixes)
|
||||
// is ignored and the probe fallback stays in charge.
|
||||
var reAgentVersion = regexp.MustCompile(`^[0-9]+\.[0-9]+\.[0-9]+$`)
|
||||
|
||||
// noteAgentVersion records a response's version header (passive capture — called on EVERY response
|
||||
// path). Invalid/absent headers never overwrite a previously-seen valid version.
|
||||
func (c *Client) noteAgentVersion(resp *http.Response) {
|
||||
v := strings.TrimSpace(resp.Header.Get("X-Felhom-Agent-Version"))
|
||||
if v == "" || !reAgentVersion.MatchString(v) {
|
||||
return
|
||||
}
|
||||
c.verMu.Lock()
|
||||
prev := c.lastAgentVersion
|
||||
c.lastAgentVersion = v
|
||||
c.verMu.Unlock()
|
||||
if prev != v {
|
||||
logx.Debugf(c.logger, "[agentapi] agent version seen: %s (was %q)", v, prev)
|
||||
}
|
||||
}
|
||||
|
||||
// AgentVersion returns the last strictly-validated agent version seen on this client's traffic
|
||||
// ("" = unknown — header-less agent or no traffic yet). This is the Supports comparison source.
|
||||
func (c *Client) AgentVersion() string {
|
||||
c.verMu.Lock()
|
||||
defer c.verMu.Unlock()
|
||||
return c.lastAgentVersion
|
||||
}
|
||||
|
||||
// MountInfo mirrors the agent's GET /storage mount entry (doc 03 §6).
|
||||
@@ -128,6 +174,14 @@ type DueResponse struct {
|
||||
Due bool `json:"due"`
|
||||
Reason string `json:"reason"`
|
||||
AgeSecs *int64 `json:"age_seconds"`
|
||||
// AgeState (R-88 Part 2, agent >= v0.105.0) says WHY AgeSecs is nil: "absent" (a positive
|
||||
// determination that no backup has ever landed) or "unknown" (the agent could not tell —
|
||||
// unreadable storage, unparseable timestamp). "known" accompanies a real age.
|
||||
//
|
||||
// EMPTY MEANS LEGACY — an agent older than v0.105.0 simply omits the field. It does NOT mean
|
||||
// "unknown", and the distinction is load-bearing: see quiesce.ageStateFromWire. Never
|
||||
// discriminate on Reason instead; those strings are operator copy and will drift.
|
||||
AgeState string `json:"age_state"`
|
||||
}
|
||||
|
||||
// BackupResponse mirrors the agent's POST /backup payload.
|
||||
@@ -269,6 +323,12 @@ type DiskInfo struct {
|
||||
// opposed to merely present on the host (F9) — the signal whose absence let the HDD look available
|
||||
// when it wasn't attached. LEGACY (per-drive mp model); the intermediary model uses BoundUnderParent.
|
||||
GuestAttached bool `json:"guest_attached"`
|
||||
// BackupTarget (E-2, agent >= v0.112.0) reports that this drive backs the PRIMARY whole-guest
|
||||
// backup tier. The agent is the only component that can answer: our own
|
||||
// settings.StoragePath.BackupTarget is customer INTENT, and on a box migrated by hand (E-1) that
|
||||
// intent was never recorded while the drive really IS the target. Absent on an older agent →
|
||||
// false, which degrades to the pre-E-2 behaviour (a generic disconnect alarm, never a wrong one).
|
||||
BackupTarget bool `json:"backup_target,omitempty"`
|
||||
// GuestPath is the drive's STABLE in-guest path in the intermediary-mount model
|
||||
// (/mnt/felhom-drives/<name>) — what the controller registers + repoints HDD_PATH to. Distinct from
|
||||
// MountPath (the raw /mnt/<name> host PVE mount the agent ops on). "" for non-user-data drives.
|
||||
@@ -276,6 +336,10 @@ type DiskInfo struct {
|
||||
// BoundUnderParent reports whether the drive's felhom-data is currently bound under the shared parent
|
||||
// (live + usable in the guest). The controller's drive-absent gate keys on this + State.
|
||||
BoundUnderParent bool `json:"bound_under_parent"`
|
||||
// Smart is the per-disk SMART health (agent v0.94.0+), nil when the device exposes no SMART or the
|
||||
// agent predates the field — the disk-health card + 6h degradation check feature-detect on this and
|
||||
// render "Nincs adat" (never alarm) when nil. See DiskVerdictFor.
|
||||
Smart *SmartSummary `json:"smart,omitempty"`
|
||||
}
|
||||
|
||||
// FSUUID returns the raw filesystem UUID from a "uuid:<…>" DurableID, or "" if this disk's identity
|
||||
@@ -360,6 +424,11 @@ type DiskCandidate struct {
|
||||
Mountable bool `json:"mountable"`
|
||||
MountSource string `json:"mount_source,omitempty"`
|
||||
DurableID string `json:"durable_id,omitempty"`
|
||||
// AlreadyMounted marks a candidate the CONTROLLER contributed from its own mount table (R-280),
|
||||
// not one the agent scanned. The agent NEVER sets it. Its action is REGISTER the existing
|
||||
// mountpoint — sending it down the device-attach path would try to mount an in-guest path as if
|
||||
// it were a raw device. See web/attach_sources.go for why the agent's scan cannot supply these.
|
||||
AlreadyMounted bool `json:"already_mounted,omitempty"`
|
||||
}
|
||||
|
||||
// CandidatesResult mirrors GET /disks/candidates: disks free to enroll, split into initialize (all
|
||||
@@ -408,6 +477,101 @@ func (c *Client) GuestReboot(ctx context.Context) error {
|
||||
return err
|
||||
}
|
||||
|
||||
// ---- v0.143.0: guest RAM resize (R-24, agent ≥ 0.90.0) ------------------------------------
|
||||
|
||||
// GuestMemoryInfo mirrors the agent's GET /guest/memory (every field MB, agent-computed). The
|
||||
// min/max/floor are the CURRENTLY-enforced bounds — the UI renders them but the agent re-checks fresh.
|
||||
type GuestMemoryInfo struct {
|
||||
VMID int `json:"vmid"`
|
||||
AllocatedMB int64 `json:"allocated_mb"`
|
||||
UsageMB int64 `json:"usage_mb"`
|
||||
HostTotalMB int64 `json:"host_total_mb"`
|
||||
MinMB int64 `json:"min_mb"`
|
||||
MaxMB int64 `json:"max_mb"`
|
||||
FloorMB int64 `json:"floor_mb"`
|
||||
Running bool `json:"running"`
|
||||
}
|
||||
|
||||
// MemoryResizeResult mirrors the agent's POST /guest/memory success body.
|
||||
type MemoryResizeResult struct {
|
||||
VMID int `json:"vmid"`
|
||||
OldMB int64 `json:"old_mb"`
|
||||
NewMB int64 `json:"new_mb"`
|
||||
Unchanged bool `json:"unchanged"`
|
||||
}
|
||||
|
||||
// MemoryRefusedError carries the agent's machine refusal code (below_min | above_max |
|
||||
// below_usage_floor) plus the fresh bounds, so the web layer maps it to a Hungarian message and
|
||||
// re-renders the range honestly — the agent's English message is never shown raw.
|
||||
type MemoryRefusedError struct {
|
||||
Code string
|
||||
Bounds GuestMemoryInfo
|
||||
Msg string
|
||||
}
|
||||
|
||||
func (e *MemoryRefusedError) Error() string {
|
||||
return "agentapi: memory resize refused (" + e.Code + "): " + e.Msg
|
||||
}
|
||||
|
||||
// GuestMemory reads the guest's current allocation, live usage, and the enforced bounds. A pre-0.90
|
||||
// agent has no such route → the get helper returns *StatusError{404} (the capability probe signal
|
||||
// and the UI's "needs an update" path).
|
||||
func (c *Client) GuestMemory(ctx context.Context) (GuestMemoryInfo, error) {
|
||||
var out GuestMemoryInfo
|
||||
data, err := c.get(ctx, "/guest/memory")
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(data, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /guest/memory: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ResizeMemory requests a bounded resize. The agent enforces every bound; a ruled refusal (412)
|
||||
// returns *MemoryRefusedError carrying the code + fresh bounds; a non-coded failure (e.g. the 502
|
||||
// verify-after-apply) returns a plain error; success returns old→new.
|
||||
func (c *Client) ResizeMemory(ctx context.Context, memoryMB int64) (MemoryResizeResult, error) {
|
||||
var out MemoryResizeResult
|
||||
env, status, err := c.postWithStatus(ctx, "/guest/memory", map[string]int64{"memory_mb": memoryMB})
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if status == http.StatusOK && env.OK {
|
||||
if len(env.Data) > 0 {
|
||||
_ = json.Unmarshal(env.Data, &out)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
// Refusal — the data carries {code, ...fresh bounds}.
|
||||
var ref struct {
|
||||
Code string `json:"code"`
|
||||
AllocatedMB int64 `json:"allocated_mb"`
|
||||
UsageMB int64 `json:"usage_mb"`
|
||||
HostTotalMB int64 `json:"host_total_mb"`
|
||||
MinMB int64 `json:"min_mb"`
|
||||
MaxMB int64 `json:"max_mb"`
|
||||
FloorMB int64 `json:"floor_mb"`
|
||||
}
|
||||
if len(env.Data) > 0 {
|
||||
_ = json.Unmarshal(env.Data, &ref)
|
||||
}
|
||||
if ref.Code != "" {
|
||||
return out, &MemoryRefusedError{
|
||||
Code: ref.Code,
|
||||
Bounds: GuestMemoryInfo{
|
||||
AllocatedMB: ref.AllocatedMB, UsageMB: ref.UsageMB, HostTotalMB: ref.HostTotalMB,
|
||||
MinMB: ref.MinMB, MaxMB: ref.MaxMB, FloorMB: ref.FloorMB,
|
||||
},
|
||||
Msg: truncateErr(env.Error, 300),
|
||||
}
|
||||
}
|
||||
if rerr := refusalError("/guest/memory", status, env); rerr != nil {
|
||||
return out, rerr
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// SwapResult mirrors the agent's 202 from POST /controller/swap (agentic controller update, Phase 1).
|
||||
type SwapResult struct {
|
||||
Status string `json:"status"` // "swapping"
|
||||
@@ -491,6 +655,7 @@ func (c *Client) WipeStagedEscrowSecret(ctx context.Context) error {
|
||||
return fmt.Errorf("agentapi: DELETE /escrow/stage-secret: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
c.noteAgentVersion(resp) // v0.82.0 version channel: passive capture on EVERY response
|
||||
raw, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
var env apiResponse
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
@@ -614,6 +779,30 @@ func (c *Client) FormatDisk(ctx context.Context, device, fstype string, confirme
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// FormatStatusResult mirrors GET /disks/format/status (F20-BUG3): the most-recent / in-flight format
|
||||
// job on the host. Phase ∈ idle | running | done | failed. The drive-init flow polls this to follow a
|
||||
// mkfs that outran the 15 s client timeout — the agent runs the mkfs DETACHED and keeps the record, so
|
||||
// the client can learn the real outcome instead of assuming failure (F6).
|
||||
type FormatStatusResult struct {
|
||||
Phase string `json:"phase"`
|
||||
Device string `json:"device"`
|
||||
FSType string `json:"fstype"`
|
||||
Error string `json:"error"`
|
||||
}
|
||||
|
||||
// FormatStatus fetches the agent's most-recent format-job record.
|
||||
func (c *Client) FormatStatus(ctx context.Context) (FormatStatusResult, error) {
|
||||
var out FormatStatusResult
|
||||
body, err := c.get(ctx, "/disks/format/status")
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /disks/format/status: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ---- NAS network storage (Part A2 → agent A1 /netstorage/*) ------------------------------
|
||||
//
|
||||
// A NAS share is a DISTINCT storage class from a drive: the controller proxies add/list/remove to the
|
||||
@@ -766,11 +955,15 @@ func (c *Client) postWithStatus(ctx context.Context, path string, body any) (api
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+c.token)
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
start := time.Now()
|
||||
resp, err := c.hc.Do(req)
|
||||
if err != nil {
|
||||
logx.Debugf(c.logger, "[agentapi] POST %s failed after %dms: %v", path, time.Since(start).Milliseconds(), err)
|
||||
return env, 0, fmt.Errorf("agentapi: POST %s: %w", path, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
c.noteAgentVersion(resp) // v0.82.0 version channel: passive capture on EVERY response
|
||||
logx.Debugf(c.logger, "[agentapi] POST %s -> %d (%dms)", path, resp.StatusCode, time.Since(start).Milliseconds())
|
||||
raw, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
return env, resp.StatusCode, fmt.Errorf("agentapi: POST %s: HTTP %d, bad envelope: %w", path, resp.StatusCode, err)
|
||||
@@ -778,6 +971,37 @@ func (c *Client) postWithStatus(ctx context.Context, path string, body any) (api
|
||||
return env, resp.StatusCode, nil
|
||||
}
|
||||
|
||||
// ---- v0.116.0: agent debug-log ring (the Debug page agent tab) ----------------------------
|
||||
|
||||
// AgentLogEntry mirrors the agent's GET /debug/logs entry (agent ≥ 0.83.0).
|
||||
type AgentLogEntry struct {
|
||||
Timestamp time.Time `json:"timestamp"`
|
||||
Level string `json:"level"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
// AgentLogsResponse mirrors the agent's GET /debug/logs data payload.
|
||||
type AgentLogsResponse struct {
|
||||
VMID int `json:"vmid"`
|
||||
Entries []AgentLogEntry `json:"entries"`
|
||||
Total int `json:"total"`
|
||||
}
|
||||
|
||||
// DebugLogs fetches the agent's always-DEBUG capture ring. Against a pre-0.83
|
||||
// agent the route is absent → a typed *StatusError with Code 404 (the caller
|
||||
// renders the "available after the agent's next update" notice — S6).
|
||||
func (c *Client) DebugLogs(ctx context.Context) (AgentLogsResponse, error) {
|
||||
var out AgentLogsResponse
|
||||
data, err := c.get(ctx, "/debug/logs")
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(data, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: parsing debug logs: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ---- slice 9: host metrics (the customer host-health view) -------------------------------
|
||||
|
||||
// HostMetrics mirrors the agent's GET /host/metrics `host` block (shared HostMetrics wire shape).
|
||||
@@ -802,14 +1026,32 @@ type ThinPoolFill struct {
|
||||
MetadataUsedFraction *float64 `json:"metadata_used_fraction"`
|
||||
}
|
||||
|
||||
// SmartSummary mirrors the agent's per-disk SMART health (only the fields the UI renders). Pointers
|
||||
// are null when the device type does not expose that attribute.
|
||||
// SmartSummary mirrors the agent's per-disk SMART health. Pointers are null when the device type
|
||||
// does not expose that attribute (a null is "unknown / not-applicable", distinct from a real zero).
|
||||
// The SATA set (reallocated/pending/offline-uncorrectable) and the NVMe set
|
||||
// (critical_warning/media_errors/percentage_used) are both carried; a device populates only its own.
|
||||
type SmartSummary struct {
|
||||
Health string `json:"health"` // PASSED | FAILING | UNKNOWN
|
||||
TemperatureC *int `json:"temperature_c"`
|
||||
PercentageUsed *int `json:"percentage_used"` // NVMe wear (%); null for SATA/USB
|
||||
Health string `json:"health"` // PASSED | FAILING | UNKNOWN
|
||||
ModelName *string `json:"model_name,omitempty"` // smartctl device model (agent v0.95.0+); nil on older agents
|
||||
TemperatureC *int `json:"temperature_c"`
|
||||
PowerOnHours *int `json:"power_on_hours"`
|
||||
// SATA attributes.
|
||||
ReallocatedSectors *int `json:"reallocated_sectors"`
|
||||
PendingSectors *int `json:"pending_sectors"`
|
||||
OfflineUncorrectable *int `json:"offline_uncorrectable"`
|
||||
// NVMe attributes.
|
||||
CriticalWarning *int `json:"critical_warning"`
|
||||
MediaErrors *int `json:"media_errors"`
|
||||
PercentageUsed *int `json:"percentage_used"` // NVMe wear (%); null for SATA/USB
|
||||
}
|
||||
|
||||
// SMART health vocabulary (mirrors the agent's).
|
||||
const (
|
||||
SmartPassed = "PASSED"
|
||||
SmartFailing = "FAILING"
|
||||
SmartUnknown = "UNKNOWN"
|
||||
)
|
||||
|
||||
// StorageTarget mirrors the agent's GET /host/metrics storage_targets entry (the per-storage
|
||||
// capacity + health the monitoring view renders). It is a SUBSET of the agent's wire shape — only
|
||||
// the fields the UI reads; unknown JSON keys are ignored.
|
||||
@@ -858,13 +1100,28 @@ func (c *Client) HostMetrics(ctx context.Context) (HostMetricsResponse, error) {
|
||||
// StatusError is a non-2xx agent HTTP status surfaced as a TYPED error (same text the old
|
||||
// fmt.Errorf produced). errors.As-able — the capability probe (features.go) keys on Code 404 to
|
||||
// distinguish "this agent predates the route" from every other failure. Never match the string.
|
||||
// StatusError is a non-2xx response from the agent, carrying the STATUS CODE so callers can react
|
||||
// to specific ones rather than string-matching an error message.
|
||||
//
|
||||
// F-A1: this exists on the POST path because HTTP 409 from `POST /backup` is not a failure — it is
|
||||
// the agent's R-85 single-flight gate correctly refusing while a restore-test holds it. Treating
|
||||
// that refusal as a tier failure armed the breaker and emailed the operator about a backup that was
|
||||
// never actually broken. The controller now needs to tell 409 apart from a real error, and a typed
|
||||
// code is the only honest way to do that.
|
||||
type StatusError struct {
|
||||
Path string
|
||||
Code int
|
||||
// Method is the HTTP method. Empty means GET, so the message stays byte-identical for the
|
||||
// pre-existing GET call sites.
|
||||
Method string
|
||||
Path string
|
||||
Code int
|
||||
}
|
||||
|
||||
func (e *StatusError) Error() string {
|
||||
return fmt.Sprintf("agentapi: GET %s: HTTP %d", e.Path, e.Code)
|
||||
m := e.Method
|
||||
if m == "" {
|
||||
m = http.MethodGet
|
||||
}
|
||||
return fmt.Sprintf("agentapi: %s %s: HTTP %d", m, e.Path, e.Code)
|
||||
}
|
||||
|
||||
// get issues an authenticated GET and unwraps the {ok,data,error} envelope.
|
||||
@@ -874,11 +1131,15 @@ func (c *Client) get(ctx context.Context, path string) (json.RawMessage, error)
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+c.token)
|
||||
start := time.Now()
|
||||
resp, err := c.hc.Do(req)
|
||||
if err != nil {
|
||||
logx.Debugf(c.logger, "[agentapi] GET %s failed after %dms: %v", path, time.Since(start).Milliseconds(), err)
|
||||
return nil, fmt.Errorf("agentapi: GET %s: %w", path, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
c.noteAgentVersion(resp) // v0.82.0 version channel: passive capture on EVERY response
|
||||
logx.Debugf(c.logger, "[agentapi] GET %s -> %d (%dms)", path, resp.StatusCode, time.Since(start).Milliseconds())
|
||||
raw, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return nil, &StatusError{Path: path, Code: resp.StatusCode}
|
||||
@@ -906,14 +1167,20 @@ func (c *Client) post(ctx context.Context, path string, body any) (json.RawMessa
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+c.token)
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
start := time.Now()
|
||||
resp, err := c.hc.Do(req)
|
||||
if err != nil {
|
||||
logx.Debugf(c.logger, "[agentapi] POST %s failed after %dms: %v", path, time.Since(start).Milliseconds(), err)
|
||||
return nil, fmt.Errorf("agentapi: POST %s: %w", path, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
c.noteAgentVersion(resp) // v0.82.0 version channel: passive capture on EVERY response
|
||||
logx.Debugf(c.logger, "[agentapi] POST %s -> %d (%dms)", path, resp.StatusCode, time.Since(start).Milliseconds())
|
||||
raw, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
if resp.StatusCode != http.StatusOK && resp.StatusCode != http.StatusAccepted {
|
||||
return nil, fmt.Errorf("agentapi: POST %s: HTTP %d", path, resp.StatusCode)
|
||||
// Typed, not fmt.Errorf: callers must be able to distinguish 409 (the agent's single-flight
|
||||
// gate refusing — contention, not failure) from a genuine 5xx. See StatusError.
|
||||
return nil, &StatusError{Method: http.MethodPost, Path: path, Code: resp.StatusCode}
|
||||
}
|
||||
var env apiResponse
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
|
||||
@@ -0,0 +1,201 @@
|
||||
package agentapi
|
||||
|
||||
// DiskVerdict is the customer-facing disk-health verdict derived from a SmartSummary (v0.169.0).
|
||||
// It is the SHARED source of truth for both the "Lemezek állapota" dashboard card and the periodic
|
||||
// degradation check — one pure function so the chip and the alert can never disagree.
|
||||
type DiskVerdict int
|
||||
|
||||
const (
|
||||
// DiskVerdictUnknown — no SMART data (nil / UNKNOWN / old agent). Renders "Nincs adat"; NEVER
|
||||
// alarms and NEVER participates in degradation transitions (excluded both directions).
|
||||
DiskVerdictUnknown DiskVerdict = iota
|
||||
DiskVerdictOK // "Rendben" — clean
|
||||
DiskVerdictWarn // "Figyelmeztetés" — a wear/relocation counter is non-zero, below the Hiba bar
|
||||
DiskVerdictFail // "Hiba" — FAILING, or failing-but-not-self-reported (v0.215.0)
|
||||
)
|
||||
|
||||
// Thresholds. A number without a reason becomes permanent by default, so each carries its provenance.
|
||||
// The evidence is committed at felhom.eu/documentation/audits/DIAG-smart-passed-trap-2026-08-14.md
|
||||
// and its two fixtures (ST3000VX010 S/N Z6A07P2G, /dev/sdg on DooPlex, 11-13 Aug 2026).
|
||||
const (
|
||||
// percentageUsedWarn / percentageUsedFail — NVMe wear (%). 100 means the vendor's rated endurance
|
||||
// is spent; that is a declaration, not a trend, so it is Hiba.
|
||||
percentageUsedWarn = 90
|
||||
percentageUsedFail = 100
|
||||
|
||||
// uncorrectableFailCount — unreadable sectors too numerous to be a blip.
|
||||
//
|
||||
// PROVENANCE: on the one real failing drive observed, the benign excursion peaked at 16 and
|
||||
// cleared COMPLETELY within an hour (11 Aug 12:28 -> 13:28); the terminal run passed 64 at
|
||||
// 13 Aug 11:28 and never came back below it. 64 sits above the one observed transient and below
|
||||
// the observed terminal run. This is a judgement from ONE drive: it is a static BACKSTOP behind
|
||||
// the sustain rule, not the primary signal, and Phase 3 is expected to replace it with
|
||||
// growth-rate detection once the box keeps history.
|
||||
uncorrectableFailCount = 64
|
||||
|
||||
// temperatureWarnC / TemperatureFailC — adopted UNCHANGED from the operator's existing Prometheus
|
||||
// bands on DooPlex, so the two systems cannot disagree about the same drive.
|
||||
temperatureWarnC = 55
|
||||
// TemperatureFailC is exported because the alert-copy layer must pick the "overheated" message
|
||||
// shape from the SAME number the verdict fired on. A second literal elsewhere would be free to
|
||||
// drift, and the drift would show up as a customer told the wrong reason.
|
||||
TemperatureFailC = 60
|
||||
)
|
||||
|
||||
// DiskPrior is what the previous check observed for THIS SAME disk. It is the only history the
|
||||
// verdict consults, and it is passed in rather than read so the function stays pure — the caller
|
||||
// (internal/web) owns loading it from the persisted per-disk state.
|
||||
//
|
||||
// Plain value type: no methods, no I/O. A zero DiskPrior means "nothing known", which is the correct
|
||||
// fail-safe — a first-ever observation can only reach Figyelmeztetés from counters, never Hiba.
|
||||
type DiskPrior struct {
|
||||
// SawUncorrectable reports whether unreadable sectors (pending OR offline-uncorrectable) were
|
||||
// present at the previous check. It is what turns a one-off excursion into a sustained fault.
|
||||
SawUncorrectable bool
|
||||
}
|
||||
|
||||
// DiskVerdictFor maps a SmartSummary plus the previous observation to a verdict. Rules are evaluated
|
||||
// TOP-DOWN and the FIRST match wins (v0.215.0):
|
||||
//
|
||||
// 1. nil / "" / UNKNOWN -> Nincs adat
|
||||
// 2. Health == FAILING -> Hiba (drive self-reports)
|
||||
// 3. temperature_c >= 60 -> Hiba
|
||||
// 4. critical_warning > 0 (NVMe's own flag: a declaration) -> Hiba
|
||||
// 5. percentage_used >= 100 -> Hiba
|
||||
// 6. unreadable > 0 AND prior.SawUncorrectable -> Hiba (SUSTAINED)
|
||||
// 7. unreadable > 0 AND reallocated > 0 -> Hiba (accumulating + remapping)
|
||||
// 8. unreadable >= 64 -> Hiba (too large to be a blip)
|
||||
// 9. unreadable > 0 -> Figyelmeztetés (first sighting)
|
||||
// 10. reallocated > 0 -> Figyelmeztetés
|
||||
// 11. media_errors > 0 -> Figyelmeztetés
|
||||
// 12. percentage_used >= 90 -> Figyelmeztetés
|
||||
// 13. temperature_c >= 55 -> Figyelmeztetés
|
||||
// 14. otherwise -> Rendben
|
||||
//
|
||||
// WHY rows 2-8 exist at all: smart_status.passed CANNOT fail on unreadable sectors. Attributes 187,
|
||||
// 197 and 198 all carry thresh 0, and a normalized SMART value floors at 1, so it can never drop to
|
||||
// or below the threshold. The real drive stayed PASSED at 352 pending sectors with 1001 reported
|
||||
// uncorrectable reads. A verdict built on the drive's own self-assessment is blind to this whole
|
||||
// class of failure, which is why rows 3-8 read the raw counters instead.
|
||||
//
|
||||
// WHY row 6 sits ABOVE row 8: sustain is the PRIMARY rule and the count is the backstop. On the real
|
||||
// drive sustain fires a full day earlier (12 Aug) than the count threshold (13 Aug). Row 8 exists for
|
||||
// a box that was powered off or restarted across the sustain window and so has no prior.
|
||||
//
|
||||
// Pure: no clock, no I/O, no logging. Everything it needs arrives as an argument.
|
||||
func DiskVerdictFor(s *SmartSummary, prior DiskPrior) DiskVerdict {
|
||||
// 1 — no data. Never alarms.
|
||||
if s == nil || s.Health == "" || s.Health == SmartUnknown {
|
||||
return DiskVerdictUnknown
|
||||
}
|
||||
// 2 — the drive admits failure.
|
||||
if s.Health == SmartFailing {
|
||||
return DiskVerdictFail
|
||||
}
|
||||
// Health == PASSED (or any non-empty non-FAILING value we treat as passing): inspect the counters,
|
||||
// because the overall verdict is structurally unable to report this class of fault.
|
||||
switch {
|
||||
case atLeast(s.TemperatureC, TemperatureFailC): // 3
|
||||
return DiskVerdictFail
|
||||
case positive(s.CriticalWarning): // 4
|
||||
return DiskVerdictFail
|
||||
case atLeast(s.PercentageUsed, percentageUsedFail): // 5
|
||||
return DiskVerdictFail
|
||||
}
|
||||
unreadable := UncorrectableSectors(s)
|
||||
switch {
|
||||
case unreadable > 0 && prior.SawUncorrectable: // 6 — sustained across two consecutive checks
|
||||
return DiskVerdictFail
|
||||
case unreadable > 0 && positive(s.ReallocatedSectors): // 7 — accumulating and remapping together
|
||||
return DiskVerdictFail
|
||||
case unreadable >= uncorrectableFailCount: // 8 — too large to be a blip
|
||||
return DiskVerdictFail
|
||||
case unreadable > 0: // 9 — first sighting, below the bar
|
||||
return DiskVerdictWarn
|
||||
case positive(s.ReallocatedSectors): // 10
|
||||
return DiskVerdictWarn
|
||||
case positive(s.MediaErrors): // 11
|
||||
return DiskVerdictWarn
|
||||
case atLeast(s.PercentageUsed, percentageUsedWarn): // 12
|
||||
return DiskVerdictWarn
|
||||
case atLeast(s.TemperatureC, temperatureWarnC): // 13
|
||||
return DiskVerdictWarn
|
||||
}
|
||||
return DiskVerdictOK // 14
|
||||
}
|
||||
|
||||
// UncorrectableSectors is the disk's unreadable-sector count: max(pending, offline_uncorrectable).
|
||||
// The two attributes track the same physical defect and on the real drive moved in lockstep, so the
|
||||
// larger is the honest figure. 0 when neither is reported (an old agent or a device without them).
|
||||
// Exported because the alert copy quotes this number and the persisted state remembers it.
|
||||
func UncorrectableSectors(s *SmartSummary) int {
|
||||
if s == nil {
|
||||
return 0
|
||||
}
|
||||
n := 0
|
||||
if s.PendingSectors != nil && *s.PendingSectors > n {
|
||||
n = *s.PendingSectors
|
||||
}
|
||||
if s.OfflineUncorrectable != nil && *s.OfflineUncorrectable > n {
|
||||
n = *s.OfflineUncorrectable
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// Label is the exact Hungarian customer copy for the verdict (shared by the card chip and the email).
|
||||
//
|
||||
// There are FOUR labels and there will not be a fifth: a predicted failure is "Hiba", the same word a
|
||||
// self-reported failure gets. A fourth word sharing a root with "Figyelmeztetés" would make the MORE
|
||||
// severe state read as the milder one (settled operator decision, v0.215.0).
|
||||
func (v DiskVerdict) Label() string {
|
||||
switch v {
|
||||
case DiskVerdictOK:
|
||||
return "Rendben"
|
||||
case DiskVerdictWarn:
|
||||
return "Figyelmeztetés"
|
||||
case DiskVerdictFail:
|
||||
return "Hiba"
|
||||
default:
|
||||
return "Nincs adat"
|
||||
}
|
||||
}
|
||||
|
||||
// DegradedAttributes returns the human-readable Hungarian names of the attribute(s) behind a
|
||||
// degraded verdict, for the alert body.
|
||||
//
|
||||
// v0.215.0: this now also names the attributes behind a Hiba REACHED FROM COUNTERS (truth-table rows
|
||||
// 3 and 6-8), not only a Figyelmeztetés — the alert message needs to say what is wrong, and those
|
||||
// rows do have a triggering counter. It returns nil ONLY for row 2 (the drive self-reports FAILING,
|
||||
// a whole-disk verdict with no single triggering counter) and, naturally, for Nincs adat / Rendben.
|
||||
func DegradedAttributes(s *SmartSummary) []string {
|
||||
if s == nil || s.Health == "" || s.Health == SmartUnknown || s.Health == SmartFailing {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
if positive(s.ReallocatedSectors) {
|
||||
out = append(out, "áthelyezett szektorok")
|
||||
}
|
||||
if positive(s.PendingSectors) {
|
||||
out = append(out, "függőben lévő szektorok")
|
||||
}
|
||||
if positive(s.OfflineUncorrectable) {
|
||||
out = append(out, "javíthatatlan szektorok")
|
||||
}
|
||||
if positive(s.CriticalWarning) {
|
||||
out = append(out, "kritikus figyelmeztetés")
|
||||
}
|
||||
if positive(s.MediaErrors) {
|
||||
out = append(out, "adathordozó-hibák")
|
||||
}
|
||||
if atLeast(s.PercentageUsed, percentageUsedWarn) {
|
||||
out = append(out, "elhasználódás")
|
||||
}
|
||||
// Newly able to trigger a verdict on its own (rows 3 and 13), so it must be nameable.
|
||||
if atLeast(s.TemperatureC, temperatureWarnC) {
|
||||
out = append(out, "hőmérséklet")
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func positive(p *int) bool { return p != nil && *p > 0 }
|
||||
func atLeast(p *int, n int) bool { return p != nil && *p >= n }
|
||||
@@ -0,0 +1,202 @@
|
||||
package agentapi
|
||||
|
||||
import "testing"
|
||||
|
||||
// The v0.215.0 severity ladder, verdict half. The event half (emission, damping, cooldown,
|
||||
// persistence) lives in internal/web — this file pins ONLY what the pure function decides.
|
||||
//
|
||||
// Every value used here is taken from the committed evidence:
|
||||
// felhom.eu/documentation/audits/fixtures/smart-ST3000VX010-failing-2026-08-14.json
|
||||
// (ST3000VX010-2E3166, S/N Z6A07P2G, /dev/sdg on DooPlex).
|
||||
|
||||
// realDrive is the failing drive AS CAPTURED on 2026-08-14: PASSED, 352 pending, 352 offline
|
||||
// uncorrectable, 0 reallocated, 40 °C. The whole point of the fixture is that Health is PASSED.
|
||||
func realDrive() *SmartSummary {
|
||||
return &SmartSummary{
|
||||
Health: SmartPassed,
|
||||
PendingSectors: ip(352),
|
||||
OfflineUncorrectable: ip(352),
|
||||
ReallocatedSectors: ip(0),
|
||||
TemperatureC: ip(40),
|
||||
}
|
||||
}
|
||||
|
||||
// Group A (verdict half) — Scenario A. The real drive on its SECOND observation reaches Hiba, and
|
||||
// the chip label is exactly "Hiba".
|
||||
//
|
||||
// Red-proof: delete truth-table row 6 (the `prior.SawUncorrectable` case) from DiskVerdictFor →
|
||||
// the drive still reaches Fail via row 8 (352 >= 64), so this test alone does NOT prove row 6.
|
||||
// TestLadder_SustainIsWhatFires below is the one that isolates it.
|
||||
func TestLadder_RealDrive_ReachesHiba(t *testing.T) {
|
||||
got := DiskVerdictFor(realDrive(), DiskPrior{SawUncorrectable: true})
|
||||
if got != DiskVerdictFail {
|
||||
t.Fatalf("real failing drive verdict = %d (%s), want Fail/Hiba", got, got.Label())
|
||||
}
|
||||
if got.Label() != "Hiba" {
|
||||
t.Errorf("label = %q, want %q", got.Label(), "Hiba")
|
||||
}
|
||||
// The trap this whole change exists for: the drive's own verdict says everything is fine.
|
||||
if realDrive().Health != SmartPassed {
|
||||
t.Fatal("fixture drift: the real drive's Health must be PASSED — that IS the defect")
|
||||
}
|
||||
}
|
||||
|
||||
// Groups B + C (verdict half) — Scenarios B and C. The SAME SmartSummary yields Figyelmeztetés on a
|
||||
// first sighting and Hiba once it is sustained. This is the pair that isolates row 6: the counters
|
||||
// are identical and only `prior` differs, so nothing else in the table can be producing the change.
|
||||
//
|
||||
// The values are the 11 August excursion (8 sectors), which cleared completely within an hour — a
|
||||
// count deliberately far below the 64 backstop so row 8 cannot mask row 6.
|
||||
//
|
||||
// Red-proof: remove the `prior.SawUncorrectable` clause from row 6 → the sustained case stays Warn.
|
||||
func TestLadder_SustainIsWhatFires(t *testing.T) {
|
||||
excursion := func() *SmartSummary {
|
||||
return &SmartSummary{Health: SmartPassed, PendingSectors: ip(8), OfflineUncorrectable: ip(8), ReallocatedSectors: ip(0)}
|
||||
}
|
||||
if got := DiskVerdictFor(excursion(), DiskPrior{}); got != DiskVerdictWarn {
|
||||
t.Errorf("first sighting of 8 sectors = %d (%s), want Warn/Figyelmeztetés — a single "+
|
||||
"excursion that clears by itself is normal and must NOT reach Hiba", got, got.Label())
|
||||
}
|
||||
if got := DiskVerdictFor(excursion(), DiskPrior{SawUncorrectable: true}); got != DiskVerdictFail {
|
||||
t.Errorf("SAME 8 sectors, now sustained = %d (%s), want Fail/Hiba", got, got.Label())
|
||||
}
|
||||
if got := DiskVerdictFor(excursion(), DiskPrior{}).Label(); got != "Figyelmeztetés" {
|
||||
t.Errorf("first-sighting label = %q, want Figyelmeztetés", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Row 8, the backstop — for a box that was powered off or restarted across the sustain window and so
|
||||
// has NO prior. 63 stays Warn, 64 reaches Hiba. The boundary is inclusive, which is what
|
||||
// `uncorrectableFailCount` claims and what the real drive did at 13 Aug 11:28 (exactly 64).
|
||||
//
|
||||
// Red-proof: change `>=` to `>` in row 8 → the "exactly 64" case reads Warn.
|
||||
func TestLadder_CountBackstopBoundary(t *testing.T) {
|
||||
cases := []struct {
|
||||
pending int
|
||||
want DiskVerdict
|
||||
}{
|
||||
{63, DiskVerdictWarn},
|
||||
{64, DiskVerdictFail},
|
||||
{352, DiskVerdictFail},
|
||||
}
|
||||
for _, c := range cases {
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(c.pending)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != c.want {
|
||||
t.Errorf("%d pending sectors, no prior = %d (%s), want %d", c.pending, got, got.Label(), c.want)
|
||||
}
|
||||
}
|
||||
// Row 7 — unreadable AND remapping together is Hiba even at a low count with no prior.
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(8), ReallocatedSectors: ip(1)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != DiskVerdictFail {
|
||||
t.Errorf("row 7 (unreadable + reallocated) = %d, want Fail", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Group I — Scenario I, heat. 61 → Hiba, 56 → Figyelmeztetés, 54 → Rendben, with all counters clean.
|
||||
//
|
||||
// Red-proof: remove rows 3 and 13 → all three read Rendben.
|
||||
func TestLadder_Temperature(t *testing.T) {
|
||||
cases := []struct {
|
||||
temp int
|
||||
want DiskVerdict
|
||||
}{
|
||||
{54, DiskVerdictOK},
|
||||
{55, DiskVerdictWarn}, // inclusive boundary
|
||||
{56, DiskVerdictWarn},
|
||||
{59, DiskVerdictWarn},
|
||||
{60, DiskVerdictFail}, // inclusive boundary
|
||||
{61, DiskVerdictFail},
|
||||
}
|
||||
for _, c := range cases {
|
||||
s := &SmartSummary{Health: SmartPassed, TemperatureC: ip(c.temp), PendingSectors: ip(0), ReallocatedSectors: ip(0)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != c.want {
|
||||
t.Errorf("%d °C = %d (%s), want %d", c.temp, got, got.Label(), c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Group J (verdict half) — Scenario J. No data never alarms, and a prior must not manufacture one:
|
||||
// a nil/UNKNOWN SmartSummary reads Nincs adat EVEN WITH SawUncorrectable set. Row 1 is first in the
|
||||
// table for exactly this reason.
|
||||
//
|
||||
// Red-proof: move row 1 below row 6 → the UNKNOWN-with-prior case reads Hiba, i.e. a disk whose
|
||||
// SMART briefly became unreadable would be reported as failing.
|
||||
func TestLadder_UnknownNeverAlarms(t *testing.T) {
|
||||
for _, s := range []*SmartSummary{nil, {Health: ""}, {Health: SmartUnknown}} {
|
||||
if got := DiskVerdictFor(s, DiskPrior{SawUncorrectable: true}); got != DiskVerdictUnknown {
|
||||
t.Errorf("no-data disk with a prior = %d (%s), want Unknown/Nincs adat", got, got.Label())
|
||||
}
|
||||
}
|
||||
if got := DiskVerdictFor(&SmartSummary{Health: SmartUnknown}, DiskPrior{}).Label(); got != "Nincs adat" {
|
||||
t.Errorf("label = %q, want Nincs adat", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The zero DiskPrior must be the SAFE default: a caller that forgets to load history can only
|
||||
// under-report (Figyelmeztetés), never over-report (Hiba) on a first sighting. This pins the
|
||||
// fail-safe direction the persisted-state loader relies on when its file is missing or corrupt.
|
||||
func TestLadder_ZeroPriorIsFailSafe(t *testing.T) {
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(8)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != DiskVerdictWarn {
|
||||
t.Fatalf("zero prior must degrade to Warn, not Fail; got %d (%s)", got, got.Label())
|
||||
}
|
||||
}
|
||||
|
||||
// UncorrectableSectors is max(pending, offline) — the number the alert copy quotes and the persisted
|
||||
// state remembers. A wrong answer here puts a wrong count in a customer's email.
|
||||
func TestUncorrectableSectors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in *SmartSummary
|
||||
want int
|
||||
}{
|
||||
{"nil summary", nil, 0},
|
||||
{"neither reported (old agent)", &SmartSummary{Health: SmartPassed}, 0},
|
||||
{"both zero", &SmartSummary{PendingSectors: ip(0), OfflineUncorrectable: ip(0)}, 0},
|
||||
{"pending only", &SmartSummary{PendingSectors: ip(8)}, 8},
|
||||
{"offline only", &SmartSummary{OfflineUncorrectable: ip(24)}, 24},
|
||||
{"pending larger", &SmartSummary{PendingSectors: ip(40), OfflineUncorrectable: ip(24)}, 40},
|
||||
{"offline larger", &SmartSummary{PendingSectors: ip(24), OfflineUncorrectable: ip(40)}, 40},
|
||||
{"the real drive", realDrive(), 352},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := UncorrectableSectors(c.in); got != c.want {
|
||||
t.Errorf("%s: UncorrectableSectors = %d, want %d", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// DegradedAttributes must NAME the counters behind a Hiba reached from counters (v0.215.0) — the
|
||||
// alert body is built from this and an empty list produces a message that says nothing is wrong.
|
||||
// It still returns nil for row 2 (drive-reported FAILING), which has no single triggering counter.
|
||||
//
|
||||
// Red-proof: restore the pre-v0.215.0 body (nil for anything at Fail) → the real-drive case returns
|
||||
// an empty list.
|
||||
func TestDegradedAttributes_NamesFailCounters(t *testing.T) {
|
||||
got := DegradedAttributes(realDrive())
|
||||
if len(got) == 0 {
|
||||
t.Fatal("a Hiba reached from counters must name its attributes, got none")
|
||||
}
|
||||
found := map[string]bool{}
|
||||
for _, a := range got {
|
||||
found[a] = true
|
||||
}
|
||||
for _, want := range []string{"függőben lévő szektorok", "javíthatatlan szektorok"} {
|
||||
if !found[want] {
|
||||
t.Errorf("missing attribute %q in %v", want, got)
|
||||
}
|
||||
}
|
||||
// Row 2 — the drive self-reports FAILING: no single triggering counter, so nil.
|
||||
if a := DegradedAttributes(&SmartSummary{Health: SmartFailing, PendingSectors: ip(5)}); a != nil {
|
||||
t.Errorf("FAILING (row 2) must return nil attributes, got %v", a)
|
||||
}
|
||||
// Nincs adat must never produce attribute names either.
|
||||
if a := DegradedAttributes(&SmartSummary{Health: SmartUnknown}); a != nil {
|
||||
t.Errorf("UNKNOWN must return nil attributes, got %v", a)
|
||||
}
|
||||
// Temperature is newly able to trigger on its own, so it must be nameable.
|
||||
hot := DegradedAttributes(&SmartSummary{Health: SmartPassed, TemperatureC: ip(61)})
|
||||
if len(hot) != 1 || hot[0] != "hőmérséklet" {
|
||||
t.Errorf("hot disk attributes = %v, want [hőmérséklet]", hot)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
package agentapi
|
||||
|
||||
import "testing"
|
||||
|
||||
func ip(v int) *int { return &v }
|
||||
|
||||
// Verdict table (Part 2, extended v0.215.0). Red-proof: change the PercentageUsed boundary from
|
||||
// `>= 90` to `> 90` in DiskVerdictFor → the "NVMe percentage_used exactly 90 → Figyelmeztetés" case
|
||||
// fails.
|
||||
//
|
||||
// v0.215.0 moved ONE pre-existing case deliberately: critical_warning>0 was Figyelmeztetés and is
|
||||
// now Hiba (truth-table row 4). It is NVMe's own critical flag — a declaration by the device, not a
|
||||
// counter that might drift back — so it belongs with the self-reported failures, not below them.
|
||||
func TestDiskVerdictFor(t *testing.T) {
|
||||
noPrior := DiskPrior{}
|
||||
cases := []struct {
|
||||
name string
|
||||
in *SmartSummary
|
||||
prior DiskPrior
|
||||
want DiskVerdict
|
||||
}{
|
||||
{"nil → unknown", nil, noPrior, DiskVerdictUnknown},
|
||||
{"empty health → unknown", &SmartSummary{Health: ""}, noPrior, DiskVerdictUnknown},
|
||||
{"UNKNOWN → unknown", &SmartSummary{Health: SmartUnknown}, noPrior, DiskVerdictUnknown},
|
||||
{"FAILING → fail", &SmartSummary{Health: SmartFailing}, noPrior, DiskVerdictFail},
|
||||
{"FAILING beats counters", &SmartSummary{Health: SmartFailing, ReallocatedSectors: ip(0)}, noPrior, DiskVerdictFail},
|
||||
{"PASSED clean → ok", &SmartSummary{Health: SmartPassed, ReallocatedSectors: ip(0), PendingSectors: ip(0), TemperatureC: ip(30)}, noPrior, DiskVerdictOK},
|
||||
{"PASSED nil counters → ok", &SmartSummary{Health: SmartPassed}, noPrior, DiskVerdictOK},
|
||||
{"reallocated>0 alone → warn", &SmartSummary{Health: SmartPassed, ReallocatedSectors: ip(1)}, noPrior, DiskVerdictWarn},
|
||||
{"pending>0 first sighting → warn", &SmartSummary{Health: SmartPassed, PendingSectors: ip(5)}, noPrior, DiskVerdictWarn},
|
||||
{"offline_unc>0 first sighting → warn", &SmartSummary{Health: SmartPassed, OfflineUncorrectable: ip(2)}, noPrior, DiskVerdictWarn},
|
||||
{"critical_warning>0 → fail (row 4)", &SmartSummary{Health: SmartPassed, CriticalWarning: ip(1)}, noPrior, DiskVerdictFail},
|
||||
{"media_errors>0 → warn", &SmartSummary{Health: SmartPassed, MediaErrors: ip(3)}, noPrior, DiskVerdictWarn},
|
||||
{"percentage_used 89 → ok", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(89)}, noPrior, DiskVerdictOK},
|
||||
{"percentage_used exactly 90 → warn", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(90)}, noPrior, DiskVerdictWarn},
|
||||
{"percentage_used 95 → warn", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(95)}, noPrior, DiskVerdictWarn},
|
||||
{"percentage_used exactly 100 → fail (row 5)", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(100)}, noPrior, DiskVerdictFail},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := DiskVerdictFor(c.in, c.prior); got != c.want {
|
||||
t.Errorf("%s: DiskVerdictFor = %d, want %d", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiskVerdict_Label(t *testing.T) {
|
||||
want := map[DiskVerdict]string{
|
||||
DiskVerdictUnknown: "Nincs adat",
|
||||
DiskVerdictOK: "Rendben",
|
||||
DiskVerdictWarn: "Figyelmeztetés",
|
||||
DiskVerdictFail: "Hiba",
|
||||
}
|
||||
for v, w := range want {
|
||||
if got := v.Label(); got != w {
|
||||
t.Errorf("verdict %d Label = %q, want %q", v, got, w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A warn lists every triggering attribute at once (Scenario "multiple attributes degrade" → ONE event).
|
||||
func TestDegradedAttributes_ListsAll(t *testing.T) {
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(5), ReallocatedSectors: ip(2), PercentageUsed: ip(91)}
|
||||
got := DegradedAttributes(s)
|
||||
if len(got) != 3 {
|
||||
t.Fatalf("want 3 attributes, got %d: %v", len(got), got)
|
||||
}
|
||||
// clean disk → none
|
||||
if a := DegradedAttributes(&SmartSummary{Health: SmartPassed}); len(a) != 0 {
|
||||
t.Errorf("clean disk should list no attributes, got %v", a)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,282 @@
|
||||
package agentapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
)
|
||||
|
||||
// Controller-driven escrow ceremony client methods (v0.127.0, agent ≥ v0.88.0). The claim call
|
||||
// is the ONLY place the recovery code R crosses this client — its response body must never be
|
||||
// logged (the shared helpers log path/status/duration only, never bodies) and the caller hands
|
||||
// R straight to the wizard's claim XHR, nowhere else.
|
||||
|
||||
// EscrowPreflightItem mirrors one agent preflight checklist row.
|
||||
type EscrowPreflightItem struct {
|
||||
ID string `json:"id"`
|
||||
OK bool `json:"ok"`
|
||||
Detail string `json:"detail"`
|
||||
}
|
||||
|
||||
// EscrowPreflightResponse mirrors GET /escrow/preflight.
|
||||
type EscrowPreflightResponse struct {
|
||||
VMID int `json:"vmid"`
|
||||
OK bool `json:"ok"`
|
||||
Items []EscrowPreflightItem `json:"items"`
|
||||
}
|
||||
|
||||
// EscrowPreflight fetches the agent's ceremony prerequisite checklist.
|
||||
func (c *Client) EscrowPreflight(ctx context.Context) (EscrowPreflightResponse, error) {
|
||||
var out EscrowPreflightResponse
|
||||
body, err := c.get(ctx, "/escrow/preflight")
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /escrow/preflight: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// EscrowCeremonyStartResponse mirrors the POST /escrow/ceremony 202 payload.
|
||||
type EscrowCeremonyStartResponse struct {
|
||||
JobID string `json:"job_id"`
|
||||
Phase string `json:"phase"`
|
||||
}
|
||||
|
||||
// EscrowCeremonyStart triggers the agent's detached root ceremony job. Status-aware: the HTTP
|
||||
// status is returned so the caller can map the agent's 409 (a ceremony already running) to its
|
||||
// own house-style refusal.
|
||||
func (c *Client) EscrowCeremonyStart(ctx context.Context) (EscrowCeremonyStartResponse, int, error) {
|
||||
var out EscrowCeremonyStartResponse
|
||||
env, status, err := c.postWithStatus(ctx, "/escrow/ceremony", struct{}{})
|
||||
if err != nil {
|
||||
return out, status, err
|
||||
}
|
||||
if err := refusalError("/escrow/ceremony", status, env); err != nil {
|
||||
return out, status, err
|
||||
}
|
||||
if err := json.Unmarshal(env.Data, &out); err != nil {
|
||||
return out, status, fmt.Errorf("agentapi: decode /escrow/ceremony: %w", err)
|
||||
}
|
||||
return out, status, nil
|
||||
}
|
||||
|
||||
// EscrowCeremonyStatusResponse mirrors GET /escrow/ceremony/status — the NON-SECRET job view
|
||||
// (R is structurally absent from the agent's payload).
|
||||
type EscrowCeremonyStatusResponse struct {
|
||||
Phase string `json:"phase"` // none | running | done | failed | unclaimed_void
|
||||
JobID string `json:"job_id"`
|
||||
KeyFingerprint string `json:"key_fingerprint"`
|
||||
EntropyBits float64 `json:"entropy_bits"`
|
||||
ResticPwSealed bool `json:"restic_pw_sealed"`
|
||||
Uploaded bool `json:"uploaded"`
|
||||
Claimable bool `json:"claimable"`
|
||||
Claimed bool `json:"claimed"`
|
||||
ClaimExpiresInSec int `json:"claim_expires_in_sec"`
|
||||
Detail string `json:"detail"`
|
||||
}
|
||||
|
||||
// EscrowCeremonyStatus polls the ceremony job.
|
||||
func (c *Client) EscrowCeremonyStatus(ctx context.Context) (EscrowCeremonyStatusResponse, error) {
|
||||
var out EscrowCeremonyStatusResponse
|
||||
body, err := c.get(ctx, "/escrow/ceremony/status")
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /escrow/ceremony/status: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// EscrowCeremonyClaim performs the ONE-SHOT R claim. Returns the recovery code + the agent's
|
||||
// HTTP status (410 = already claimed / expired — the wizard's void state). The code must never
|
||||
// be logged, persisted, or placed anywhere but the claim XHR response; the error path carries
|
||||
// the agent's reason text, never the code.
|
||||
func (c *Client) EscrowCeremonyClaim(ctx context.Context) (string, int, error) {
|
||||
env, status, err := c.postWithStatus(ctx, "/escrow/ceremony/claim", struct{}{})
|
||||
if err != nil {
|
||||
return "", status, err
|
||||
}
|
||||
if status == http.StatusGone {
|
||||
return "", status, fmt.Errorf("agentapi: POST /escrow/ceremony/claim: gone (claimed or expired)")
|
||||
}
|
||||
if err := refusalError("/escrow/ceremony/claim", status, env); err != nil {
|
||||
return "", status, err
|
||||
}
|
||||
var out struct {
|
||||
RecoveryCode string `json:"recovery_code"`
|
||||
}
|
||||
if err := json.Unmarshal(env.Data, &out); err != nil {
|
||||
return "", status, fmt.Errorf("agentapi: decode /escrow/ceremony/claim: %w", err)
|
||||
}
|
||||
if out.RecoveryCode == "" {
|
||||
return "", status, fmt.Errorf("agentapi: /escrow/ceremony/claim returned no code")
|
||||
}
|
||||
return out.RecoveryCode, status, nil
|
||||
}
|
||||
|
||||
// RecoverOffsiteRepoPassword asks the agent to open this host's hub-held sealed bundle with the
|
||||
// customer's recovery code and return ONLY the offsite restic repository password, plus its sha256
|
||||
// (R-199, agent >= v0.125.0).
|
||||
//
|
||||
// R CROSSES HERE, AND NOWHERE ELSE IN THIS DIRECTION. It travels in the request body over the pinned
|
||||
// local-API channel (the operator's 2026-08-04 acceptance) and is not retained by this client. The
|
||||
// shared POST helper logs path/status/duration and never bodies — do not add a body log, on either
|
||||
// the request or the response side: the request carries R and the response carries the password.
|
||||
func (c *Client) RecoverOffsiteRepoPassword(ctx context.Context, recoveryCode string) (password, sha256hex string, err error) {
|
||||
env, status, perr := c.postWithStatus(ctx, "/escrow/recover-offsite-password",
|
||||
map[string]string{"recovery_code": recoveryCode})
|
||||
if perr != nil {
|
||||
return "", "", perr
|
||||
}
|
||||
// R-224: this route's refusal keeps its STATUS as a value. `refusalError` flattens status into a
|
||||
// sentence, and a sentence is not something a caller can branch on — which is exactly how a failed
|
||||
// fetch and a wrong recovery code came to produce one customer-facing message.
|
||||
if status < 200 || status > 299 || !env.OK {
|
||||
return "", "", &RecoveryRefusal{Status: status, Reason: truncateErr(env.Error, 300)}
|
||||
}
|
||||
var out struct {
|
||||
ResticRepoPassword string `json:"restic_repo_password"`
|
||||
ResticPwSHA256 string `json:"restic_pw_sha256"`
|
||||
}
|
||||
if uerr := json.Unmarshal(env.Data, &out); uerr != nil {
|
||||
return "", "", fmt.Errorf("agentapi: decode /escrow/recover-offsite-password: %w", uerr)
|
||||
}
|
||||
if out.ResticRepoPassword == "" || out.ResticPwSHA256 == "" {
|
||||
return "", "", fmt.Errorf("agentapi: the agent returned an empty recovery result")
|
||||
}
|
||||
return out.ResticRepoPassword, out.ResticPwSHA256, nil
|
||||
}
|
||||
|
||||
// ── R-224 — CLASSIFYING A FAILED UNLOCK ─────────────────────────────────────────────────────────
|
||||
//
|
||||
// CAMPAIGN-11 measured what happens without this. On 2026-08-05, with a CORRECT current recovery
|
||||
// code: the hub firewalled off returned the customer "this code does not open your package" in
|
||||
// 0.0556 s, and this agent stopped returned the same in 0.0299 s — against ~1.0 s for a genuine
|
||||
// unseal. Neither attempted one. The failure path had exactly two branches, both of them statements
|
||||
// about the customer's code, and `rerr` was never inspected.
|
||||
//
|
||||
// The rule this type exists to enforce: **the customer is blamed only after a real attempt refused
|
||||
// their code.** Everything else — including anything we cannot classify — says something else.
|
||||
|
||||
// RecoveryRefusal is the agent's refusal of an unlock, carrying the STATUS as a value so callers
|
||||
// classify on it rather than on the sentence. The message keeps `refusalError`'s shape so operator
|
||||
// logs read as they did.
|
||||
type RecoveryRefusal struct {
|
||||
Status int
|
||||
Reason string
|
||||
}
|
||||
|
||||
func (e *RecoveryRefusal) Error() string {
|
||||
reason := e.Reason
|
||||
if reason == "" {
|
||||
reason = "(no reason in agent response)"
|
||||
}
|
||||
return fmt.Sprintf("agentapi: POST /escrow/recover-offsite-password: HTTP %d: %s", e.Status, reason)
|
||||
}
|
||||
|
||||
// RecoveryFailure is what went wrong, as far as it can be known.
|
||||
type RecoveryFailure int
|
||||
|
||||
const (
|
||||
// RecoveryUnknown — the cause could not be determined. **The safe default**, and deliberately the
|
||||
// zero value: a new status, a transport shape nobody anticipated, or an agent too old to
|
||||
// distinguish fetch from refusal all land here, and none of them may blame the customer.
|
||||
RecoveryUnknown RecoveryFailure = iota
|
||||
// RecoveryHubUnreachable — the agent answered, and it could not FETCH the sealed package: the hub
|
||||
// refused, was unreachable, or recovery is not configured on this agent. **The code was not used.**
|
||||
RecoveryHubUnreachable
|
||||
// RecoveryAskedAndRefused — the bundle was fetched and the code did not open it. The ONLY class
|
||||
// from which the customer may be told to check their typing.
|
||||
RecoveryAskedAndRefused
|
||||
// RecoveryNoBundle — the hub holds no sealed package for this host at all.
|
||||
RecoveryNoBundle
|
||||
// RecoveryBundleTooOld — the bundle opened but predates the repository-password field.
|
||||
RecoveryBundleTooOld
|
||||
// RecoveryAgentUnreachable — the machine's own in-house service never answered, so there is no
|
||||
// agent verdict at all. **The code was not used.** Distinct from RecoveryHubUnreachable because
|
||||
// it is a different fault, with different words and a different remedy.
|
||||
RecoveryAgentUnreachable
|
||||
// RecoveryCodeOpensRetained — the code was used, it WORKED, and it opened a RETAINED earlier
|
||||
// package rather than the one currently held (R-311, agent >= v0.129.0).
|
||||
//
|
||||
// **The customer is not at fault here and must not be told they might be.** This class exists
|
||||
// because until 2026-08-12 this situation and a mistype were indistinguishable: both fail closed
|
||||
// against the current package, and nothing ever tried the retained ones. The screen said as much
|
||||
// out loud — a true sentence about our own incuriosity that a customer reads as a statement about
|
||||
// their code.
|
||||
RecoveryCodeOpensRetained
|
||||
)
|
||||
|
||||
// ClassifyRecoveryFailure maps an unlock error to its class, from the VALUE and never the text.
|
||||
//
|
||||
// ⚠ `trustRefusal` is the agent-version gate and it is not optional. An agent older than v0.126.0
|
||||
// answers **400 for BOTH** a fetch failure and a wrong code, so a 400 from one cannot be read as
|
||||
// "the code was refused" — it means "one of two things, and we cannot tell which". Pass false there
|
||||
// and the 400 degrades to RecoveryUnknown, which is neutral. That degradation is the point: it is
|
||||
// safe, it is silent, and it heals itself when the agent updates.
|
||||
// ⚠ `trustRetained` is the R-311 twin of `trustRefusal` and is separate on purpose: the two gates
|
||||
// name different agent versions (v0.126.0 and v0.129.0) and a box can sit between them. Passing
|
||||
// `trustRefusal` for both would let a v0.126–128 agent's unexpected 422 be read as a verdict it
|
||||
// cannot produce.
|
||||
func ClassifyRecoveryFailure(err error, trustRefusal, trustRetained bool) RecoveryFailure {
|
||||
if err == nil {
|
||||
return RecoveryUnknown
|
||||
}
|
||||
var ref *RecoveryRefusal
|
||||
if !errors.As(err, &ref) {
|
||||
// Not a refusal at all — the request never produced an agent verdict (dial failure, TLS,
|
||||
// timeout, or the channel could not be built). The machine could not even ASK its own service,
|
||||
// which is a different sentence from "the hub was unreachable" and a different thing to fix.
|
||||
return RecoveryAgentUnreachable
|
||||
}
|
||||
switch ref.Status {
|
||||
case http.StatusBadGateway, http.StatusServiceUnavailable, http.StatusGatewayTimeout:
|
||||
// 502 is agent >= v0.126.0's "the sealed bundle could not be fetched". 503 is its
|
||||
// "recovery is not configured on this agent (no hub client)". Neither used the code.
|
||||
return RecoveryHubUnreachable
|
||||
case http.StatusNotFound:
|
||||
return RecoveryNoBundle
|
||||
case http.StatusConflict:
|
||||
return RecoveryBundleTooOld
|
||||
case http.StatusUnprocessableEntity:
|
||||
// R-311. Gated on the SAME trust flag as 400, and for the mirror-image reason: an agent that
|
||||
// predates the retained lookup cannot emit 422 at all, so a 422 from anywhere else is a shape
|
||||
// we did not design and must not be read as a statement about the customer's code.
|
||||
if trustRetained {
|
||||
return RecoveryCodeOpensRetained
|
||||
}
|
||||
return RecoveryUnknown
|
||||
case http.StatusBadRequest:
|
||||
if trustRefusal {
|
||||
return RecoveryAskedAndRefused
|
||||
}
|
||||
return RecoveryUnknown
|
||||
default:
|
||||
return RecoveryUnknown
|
||||
}
|
||||
}
|
||||
|
||||
// String names the class for the operator log. The customer never sees these words.
|
||||
func (f RecoveryFailure) String() string {
|
||||
switch f {
|
||||
case RecoveryHubUnreachable:
|
||||
return "hub-unreachable"
|
||||
case RecoveryAgentUnreachable:
|
||||
return "agent-unreachable"
|
||||
case RecoveryAskedAndRefused:
|
||||
return "asked-and-refused"
|
||||
case RecoveryNoBundle:
|
||||
return "no-bundle"
|
||||
case RecoveryBundleTooOld:
|
||||
return "bundle-too-old"
|
||||
case RecoveryCodeOpensRetained:
|
||||
return "code-opens-retained"
|
||||
default:
|
||||
return "unknown"
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,9 @@ import (
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/logx"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
)
|
||||
|
||||
// Agent-capability probing (the publish-train backstop). A controller release that depends on
|
||||
@@ -22,6 +25,64 @@ type Feature string
|
||||
// capability signal.
|
||||
const FeatureNetstorageVerify Feature = "netstorage_verify"
|
||||
|
||||
// FeatureGuestMemoryResize is the guest RAM resize (agent v0.90.0, R-24): the resize endpoints
|
||||
// (GET/POST /guest/memory) shipped together, so GET /guest/memory IS the capability signal.
|
||||
const FeatureGuestMemoryResize Feature = "guest_memory_resize"
|
||||
|
||||
// FeatureBackupAgeState is R-88 Part 2 (agent v0.105.0): GET /backup/due carries `age_state`,
|
||||
// distinguishing "never backed up" (absent) from "could not tell" (unknown). There is no route
|
||||
// probe for it — the signal is a FIELD on an existing route, so the version floor is the gate and
|
||||
// an empty field means legacy.
|
||||
const FeatureBackupAgeState Feature = "backup_age_state"
|
||||
|
||||
// FeatureOffsiteKeyRecovery is the customer-facing off-site key recovery (agent v0.125.0, R-199
|
||||
// links 7–8): POST /escrow/recover-offsite-password fetches this host's sealed bundle, unseals it
|
||||
// with R and returns the single repository-password field.
|
||||
//
|
||||
// ⚠ THIS GATE FAILS CLOSED, and it is the ONLY feature in this table that does. Read §7.1 of the
|
||||
// R-216 fix before "correcting" it back to the package default.
|
||||
//
|
||||
// The package default is fail-OPEN: SupportUnknown proceeds, because for every other coupled feature
|
||||
// a wrong "unsupported" would block something harmless while a down agent already speaks through the
|
||||
// normal error paths. **That default is what produced R-216.** Measured live on 2026-08-05
|
||||
// (CAMPAIGN-11 Phase 1): an agent 0.120.0 answered the recovery route with 404, the unlock attempt
|
||||
// went ahead anyway, and the customer was told — in Hungarian, on the one screen whose whole purpose
|
||||
// is to be believed about backups — that their perfectly correct recovery code was not accepted and
|
||||
// they should check their typing. A correct code, refused in 0.134 s, blamed on the customer.
|
||||
//
|
||||
// So here: anything other than SupportYes means the screen says THE MACHINE cannot ask yet. The
|
||||
// unlock is never attempted when it cannot complete, because the failure of an attempt that could
|
||||
// never have worked is attributed to the code.
|
||||
const FeatureOffsiteKeyRecovery Feature = "offsite_key_recovery"
|
||||
|
||||
// FeatureRecoveryFailureClass is agent v0.126.0's SPLIT of a failed unlock into distinguishable
|
||||
// statuses (R-224): 502 the sealed bundle could not be FETCHED · 400 it was fetched and the code was
|
||||
// refused · 404 no bundle · 409 the bundle predates the repository-password field.
|
||||
//
|
||||
// ⚠ WHAT THIS GATE ACTUALLY GUARDS is the meaning of **400**, and nothing else. An agent older than
|
||||
// v0.126.0 answers 400 for BOTH a fetch failure and a wrong code — one status, one sentence, two
|
||||
// situations — so on such an agent a 400 cannot be read as "the code was refused". It means "one of
|
||||
// two things and we cannot tell which", which is `RecoveryUnknown`, which is neutral.
|
||||
//
|
||||
// So this gate does not block anything and has no fail-closed behaviour to get wrong: the unlock is
|
||||
// attempted either way (FeatureOffsiteKeyRecovery already decides THAT). It only decides whether the
|
||||
// customer may be told to check their typing. Unknown → they may not. **That is the safe direction,
|
||||
// and it heals itself the moment the agent updates.**
|
||||
const FeatureRecoveryFailureClass Feature = "recovery_failure_class"
|
||||
|
||||
// FeatureRetainedRecoveryClass is agent v0.129.0's FIFTH status on a failed unlock (R-311): 422, the
|
||||
// code is correct and opens a RETAINED earlier package rather than the current one.
|
||||
//
|
||||
// ⚠ WHAT THIS GATE GUARDS is whether the screen may say WHICH of the two causes it is. Before
|
||||
// v0.129.0 nothing ever tried the retained packages, so a correct-but-earlier code and a mistype were
|
||||
// genuinely indistinguishable and the screen said so. That sentence was HONEST then and becomes a
|
||||
// falsehood the moment the agent can tell them apart — so the gate decides which of two true
|
||||
// sentences to print, never whether to attempt the unlock.
|
||||
//
|
||||
// Unknown → the older, hedged sentence. That is the safe direction: it claims less, it was correct
|
||||
// for two months, and it heals itself when the agent updates.
|
||||
const FeatureRetainedRecoveryClass Feature = "retained_recovery_class"
|
||||
|
||||
// SupportState is a probe verdict. The zero value is SupportUnknown (fail-open: unknown never
|
||||
// refuses — the existing agent-error paths speak honestly when the agent is down).
|
||||
type SupportState int
|
||||
@@ -35,24 +96,113 @@ const (
|
||||
SupportNo
|
||||
)
|
||||
|
||||
// String returns the state's wire/template vocabulary: "yes" | "no" | "unknown".
|
||||
func (s SupportState) String() string {
|
||||
switch s {
|
||||
case SupportYes:
|
||||
return "yes"
|
||||
case SupportNo:
|
||||
return "no"
|
||||
default:
|
||||
return "unknown"
|
||||
}
|
||||
}
|
||||
|
||||
// SupportProber is the minimal agent surface a probe needs. *Client satisfies it, and so does the
|
||||
// web layer's netAgent seam — tests inject fakes there.
|
||||
type SupportProber interface {
|
||||
NetVerifyStatus(ctx context.Context) (NetVerifyStatus, error)
|
||||
}
|
||||
|
||||
// featureProbes maps each coupled feature to its route probe.
|
||||
// featureProbes maps each coupled feature to its route probe (the FALLBACK path for agents that
|
||||
// predate the v0.82.0 version header).
|
||||
//
|
||||
// CONVENTION (publish-train rules doc, felhom.eu/documentation/runbooks/publish-train-rules.md):
|
||||
// every future coupled feature adds a row here plus a Supports gate call at its entry point, and
|
||||
// declares MinAgent in its CHANGELOG header. When the agent someday reports an explicit version in
|
||||
// its envelope, Supports should prefer that version comparison over route probing — that
|
||||
// enhancement is roadmap, not built yet.
|
||||
// every future coupled feature adds a row here AND a featureMinAgent row, plus a Supports gate
|
||||
// call at its entry point, and declares MinAgent in its CHANGELOG header.
|
||||
var featureProbes = map[Feature]func(ctx context.Context, p SupportProber) error{
|
||||
FeatureNetstorageVerify: func(ctx context.Context, p SupportProber) error {
|
||||
_, err := p.NetVerifyStatus(ctx)
|
||||
return err
|
||||
},
|
||||
// The memory-resize prober needs GET /guest/memory, not NetVerifyStatus. Rather than couple the
|
||||
// shared SupportProber (and every unrelated prober/fake) to the memory surface, the probe
|
||||
// type-asserts the ONE method it needs — the memory feature is only ever probed with a
|
||||
// GuestMemory-capable prober (the web memAgent seam / *Client). A prober without it → a non-404
|
||||
// error → SupportUnknown (fail-open), never a false "supported".
|
||||
FeatureGuestMemoryResize: func(ctx context.Context, p SupportProber) error {
|
||||
gm, ok := p.(interface {
|
||||
GuestMemory(ctx context.Context) (GuestMemoryInfo, error)
|
||||
})
|
||||
if !ok {
|
||||
return errNoMemoryProbe
|
||||
}
|
||||
_, err := gm.GuestMemory(ctx)
|
||||
return err
|
||||
},
|
||||
// The recovery route is a POST that performs work and consumes a recovery code — it cannot be
|
||||
// probed. Like the memory prober's negative case this returns a sentinel that classifies to
|
||||
// SupportUnknown, so the decision falls to the VERSION path above.
|
||||
//
|
||||
// The row must exist even though it cannot probe: SupportsWithSource looks up featureProbes
|
||||
// FIRST and returns "unregistered"/SupportUnknown on a table gap, before the version path runs.
|
||||
// A featureMinAgent row without a featureProbes row is therefore never consulted at all.
|
||||
FeatureOffsiteKeyRecovery: func(ctx context.Context, p SupportProber) error {
|
||||
return errNoRecoveryProbe
|
||||
},
|
||||
// Same POST route, same reason it cannot be probed — the decision falls to the VERSION path.
|
||||
FeatureRecoveryFailureClass: func(ctx context.Context, p SupportProber) error {
|
||||
return errNoRecoveryProbe
|
||||
},
|
||||
// R-311, same route and same reason. The row must exist or SupportsWithSource returns
|
||||
// "unregistered"/SupportUnknown on the table gap and the version row is never consulted.
|
||||
FeatureRetainedRecoveryClass: func(ctx context.Context, p SupportProber) error {
|
||||
return errNoRecoveryProbe
|
||||
},
|
||||
}
|
||||
|
||||
// errNoMemoryProbe classifies to SupportUnknown (not a *StatusError 404), so a prober that cannot be
|
||||
// asked never reads as "unsupported".
|
||||
var errNoMemoryProbe = errors.New("agentapi: prober does not support the guest-memory probe")
|
||||
|
||||
// errNoRecoveryProbe classifies to SupportUnknown: the off-site key recovery route is a POST that
|
||||
// consumes a recovery code and so cannot be probed, leaving the VERSION path to decide. Its caller
|
||||
// fails CLOSED on Unknown — see FeatureOffsiteKeyRecovery.
|
||||
var errNoRecoveryProbe = errors.New("agentapi: the offsite key recovery route cannot be probed")
|
||||
|
||||
// featureMinAgent maps each coupled feature to the MINIMUM agent version that carries its coupled
|
||||
// semantics (the CHANGELOG `MinAgent:` header value). Used by Supports when the agent's version is
|
||||
// KNOWN (the v0.82.0 X-Felhom-Agent-Version channel) — a direct comparison, no probe traffic. A
|
||||
// feature missing here (or an unparseable table value) falls back to the probe.
|
||||
var featureMinAgent = map[Feature]string{
|
||||
FeatureNetstorageVerify: "0.81.0",
|
||||
FeatureGuestMemoryResize: "0.90.0",
|
||||
// R-88 Part 2: /backup/due carries age_state, distinguishing "never backed up" from "cannot tell".
|
||||
FeatureBackupAgeState: "0.105.0",
|
||||
// R-199 links 7–8: POST /escrow/recover-offsite-password. R-216 — this row is the whole reason a
|
||||
// correct recovery code can no longer be reported as wrong on an agent that cannot answer.
|
||||
FeatureOffsiteKeyRecovery: "0.125.0",
|
||||
|
||||
// R-224 — the four-way status split of a failed unlock.
|
||||
FeatureRecoveryFailureClass: "0.126.0",
|
||||
|
||||
// R-311 — the FIFTH status: 422, "your code is correct, it opens an EARLIER package". Before
|
||||
// v0.129.0 the agent never looked at retained packages, so this situation was indistinguishable
|
||||
// from a mistype and arrived as 400. An older agent therefore cannot produce a 422 at all, and the
|
||||
// screen must keep saying it cannot tell the two apart — which was true, and is what this gate
|
||||
// preserves for boxes that have not updated yet.
|
||||
FeatureRetainedRecoveryClass: "0.129.0",
|
||||
}
|
||||
|
||||
// MinAgentFor returns the declared minimum agent version for a feature ("" when the feature has no
|
||||
// row). Read-only accessor over featureMinAgent so a refusal can NAME the version it needs instead of
|
||||
// hard-coding the number a second time at the call site.
|
||||
func MinAgentFor(f Feature) string { return featureMinAgent[f] }
|
||||
|
||||
// AgentVersionReporter is optionally implemented by a SupportProber (*Client is one): it reports
|
||||
// the last strictly-validated agent version seen on its traffic ("" = unknown → probe fallback).
|
||||
type AgentVersionReporter interface {
|
||||
AgentVersion() string
|
||||
}
|
||||
|
||||
// supportTTL bounds how long a probe verdict (either polarity) is trusted. An agent updated
|
||||
@@ -73,13 +223,31 @@ type SupportCache struct {
|
||||
entries map[Feature]supportEntry
|
||||
}
|
||||
|
||||
// Supports reports whether the agent behind p provides feature f, answering from the cache inside
|
||||
// the TTL window and probing otherwise. The probe runs OUTSIDE the lock — concurrent misses may
|
||||
// double-probe (harmless: the probe is one cheap GET).
|
||||
// Supports reports whether the agent behind p provides feature f.
|
||||
//
|
||||
// Order (v0.115.0): (1) the agent's VERSION is known (the v0.82.0 header channel, strictly
|
||||
// validated at capture) AND the feature has a MinAgent row → pure semver comparison, NO probe
|
||||
// traffic, no cache involvement (the cache stays probe-only); (2) otherwise — pre-0.82 agent, no
|
||||
// traffic yet, or a table gap — the v0.114.0 probe path, byte-identical (cache inside the TTL
|
||||
// window, probe on miss). The probe runs OUTSIDE the lock — concurrent misses may double-probe
|
||||
// (harmless: the probe is one cheap GET).
|
||||
func (sc *SupportCache) Supports(ctx context.Context, p SupportProber, f Feature) SupportState {
|
||||
state, _ := sc.SupportsWithSource(ctx, p, f)
|
||||
return state
|
||||
}
|
||||
|
||||
// SupportsWithSource is Supports plus the DECISION SOURCE ("version" | "probe-cache" |
|
||||
// "probe" | "unregistered") — the v0.116.0 observability extension so the gate line can
|
||||
// say HOW the verdict was reached. Behavior is byte-identical to v0.115.0 Supports.
|
||||
func (sc *SupportCache) SupportsWithSource(ctx context.Context, p SupportProber, f Feature) (SupportState, string) {
|
||||
probe, ok := featureProbes[f]
|
||||
if !ok {
|
||||
return SupportUnknown // unregistered feature — never refuse on a table gap
|
||||
return SupportUnknown, "unregistered" // unregistered feature — never refuse on a table gap
|
||||
}
|
||||
if vr, hasVer := p.(AgentVersionReporter); hasVer {
|
||||
if state, decided := supportsByVersion(vr.AgentVersion(), f); decided {
|
||||
return state, "version"
|
||||
}
|
||||
}
|
||||
sc.mu.Lock()
|
||||
nowFn := sc.now
|
||||
@@ -88,7 +256,7 @@ func (sc *SupportCache) Supports(ctx context.Context, p SupportProber, f Feature
|
||||
}
|
||||
if e, hit := sc.entries[f]; hit && nowFn().Sub(e.at) < supportTTL {
|
||||
sc.mu.Unlock()
|
||||
return e.state
|
||||
return e.state, "probe-cache"
|
||||
}
|
||||
sc.mu.Unlock()
|
||||
|
||||
@@ -101,14 +269,42 @@ func (sc *SupportCache) Supports(ctx context.Context, p SupportProber, f Feature
|
||||
sc.entries[f] = supportEntry{state: state, at: nowFn()}
|
||||
sc.mu.Unlock()
|
||||
}
|
||||
return state
|
||||
return state, "probe"
|
||||
}
|
||||
|
||||
// Supports probes (cached, TTL 5m, both polarities) whether the connected agent provides the
|
||||
// feature. 2xx ⇒ Yes. 404 ⇒ No. Anything else ⇒ Unknown (never "too old"). The web layer drives
|
||||
// the same machinery through its netAgent seam (Server.netFeatures) so tests can fake the probe.
|
||||
func (c *Client) Supports(ctx context.Context, f Feature) SupportState {
|
||||
return c.features.Supports(ctx, c, f)
|
||||
state, source := c.features.SupportsWithSource(ctx, c, f)
|
||||
logx.Debugf(c.logger, "[agentapi] Supports(%s) = %s (source=%s agent_version=%q)",
|
||||
f, state, source, c.AgentVersion())
|
||||
return state
|
||||
}
|
||||
|
||||
// supportsByVersion decides a feature by version comparison alone. decided=false (unknown/garbage
|
||||
// version, missing MinAgent row, unparseable table value) sends the caller to the probe fallback —
|
||||
// a bad version string must never be trusted in EITHER direction.
|
||||
func supportsByVersion(agentVer string, f Feature) (state SupportState, decided bool) {
|
||||
if agentVer == "" {
|
||||
return SupportUnknown, false
|
||||
}
|
||||
minStr, ok := featureMinAgent[f]
|
||||
if !ok {
|
||||
return SupportUnknown, false
|
||||
}
|
||||
av, err := util.ParseVersion(agentVer)
|
||||
if err != nil {
|
||||
return SupportUnknown, false // capture validates the shape, but stay defensive
|
||||
}
|
||||
mv, err := util.ParseVersion(minStr)
|
||||
if err != nil {
|
||||
return SupportUnknown, false // a broken table row falls back to probing, never refuses
|
||||
}
|
||||
if av.Compare(mv) >= 0 {
|
||||
return SupportYes, true
|
||||
}
|
||||
return SupportNo, true
|
||||
}
|
||||
|
||||
// classifySupportErr maps a probe outcome to a SupportState. ONLY a typed HTTP 404 means
|
||||
@@ -124,3 +320,9 @@ func classifySupportErr(err error) SupportState {
|
||||
}
|
||||
return SupportUnknown
|
||||
}
|
||||
|
||||
// AgentVersionReporter witness. Asserted at SupportsWithSource (`p.(AgentVersionReporter)`); a failed
|
||||
// assertion falls back from the version gate to the live probe. That degrade is benign — both paths
|
||||
// decide correctly — but *Client is the production prober and losing the version path would silently
|
||||
// turn every MinAgent floor into a probe round-trip, which is a behaviour change nobody would see.
|
||||
var _ AgentVersionReporter = (*Client)(nil)
|
||||
|
||||
@@ -0,0 +1,179 @@
|
||||
package agentapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
)
|
||||
|
||||
// Version-aware Supports (v0.115.0, pairs with agent v0.82.0's X-Felhom-Agent-Version): a KNOWN
|
||||
// version decides by comparison with ZERO probe traffic; unknown/garbage stays on the v0.114.0
|
||||
// probe path byte-identically.
|
||||
|
||||
// verProber implements SupportProber + AgentVersionReporter with a probe counter.
|
||||
type verProber struct {
|
||||
ver string
|
||||
probes int
|
||||
}
|
||||
|
||||
func (p *verProber) NetVerifyStatus(context.Context) (NetVerifyStatus, error) {
|
||||
p.probes++
|
||||
return NetVerifyStatus{Phase: "none"}, nil // a live route (probe verdict would be Yes)
|
||||
}
|
||||
func (p *verProber) AgentVersion() string { return p.ver }
|
||||
|
||||
// --- B3.1: version known → compare path, probe count 0 -----------------------------------------
|
||||
// Companion red-proof: drop the supportsByVersion short-circuit in Supports → probes becomes 1.
|
||||
func TestSupports_VersionKnown_ComparesWithoutProbe(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
ver string
|
||||
want SupportState
|
||||
}{
|
||||
{"0.82.0", SupportYes},
|
||||
{"0.81.0", SupportYes}, // boundary: MinAgent itself qualifies
|
||||
{"1.0.0", SupportYes}, // numeric compare across majors
|
||||
{"0.100.0", SupportYes}, // numeric, NOT lexicographic (0.100 > 0.81)
|
||||
{"0.79.0", SupportNo}, // below MinAgent → No, still without probing
|
||||
{"0.80.9", SupportNo},
|
||||
} {
|
||||
t.Run(tc.ver, func(t *testing.T) {
|
||||
p := &verProber{ver: tc.ver}
|
||||
var sc SupportCache
|
||||
if got := sc.Supports(context.Background(), p, FeatureNetstorageVerify); got != tc.want {
|
||||
t.Errorf("Supports(ver=%s) = %v, want %v", tc.ver, got, tc.want)
|
||||
}
|
||||
if p.probes != 0 {
|
||||
t.Errorf("version-known path must NOT probe (ver=%s, probes=%d)", tc.ver, p.probes)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// --- B3.2: garbage/absent version → the probe fallback, byte-identical --------------------------
|
||||
// Companion red-proof: trust the unvalidated header text (compare without ParseVersion error
|
||||
// handling) → the garbage rows would refuse or panic instead of probing.
|
||||
func TestSupports_GarbageOrNoVersion_ProbeFallback(t *testing.T) {
|
||||
for _, ver := range []string{"", "dev", "0.82", "0.82.0-rc1", "v0.82.0-beta", "evil;rm -rf", "9999999999999999999999.0.0"} {
|
||||
t.Run("ver="+ver, func(t *testing.T) {
|
||||
p := &verProber{ver: ver}
|
||||
var sc SupportCache
|
||||
got := sc.Supports(context.Background(), p, FeatureNetstorageVerify)
|
||||
if p.probes != 1 {
|
||||
t.Fatalf("unusable version %q must fall back to EXACTLY one probe, got %d", ver, p.probes)
|
||||
}
|
||||
if got != SupportYes { // the fake's route answers → the probe decides Yes
|
||||
t.Errorf("probe fallback verdict = %v, want SupportYes", got)
|
||||
}
|
||||
// Second call: the probe verdict is cached (the v0.114.0 behavior, unchanged).
|
||||
_ = sc.Supports(context.Background(), p, FeatureNetstorageVerify)
|
||||
if p.probes != 1 {
|
||||
t.Errorf("cached probe verdict must not re-probe (probes=%d)", p.probes)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A prober that does NOT implement AgentVersionReporter (the web fakes' shape) keeps the pure
|
||||
// v0.114.0 behavior — the interface assertion must not change anything for it.
|
||||
func TestSupports_NonReporterProber_Unchanged(t *testing.T) {
|
||||
p := &fakeProber{} // the existing v0.114.0 test fake (features_test.go)
|
||||
var sc SupportCache
|
||||
if got := sc.Supports(context.Background(), p, FeatureNetstorageVerify); got != SupportYes {
|
||||
t.Fatalf("non-reporter prober = %v, want SupportYes via probe", got)
|
||||
}
|
||||
if p.calls != 1 {
|
||||
t.Errorf("non-reporter prober must probe exactly once, got %d", p.calls)
|
||||
}
|
||||
}
|
||||
|
||||
// --- B3.4: the ONE comparator, table-driven (incl. pre-release suffixes) ------------------------
|
||||
func TestVersionComparator_Table(t *testing.T) {
|
||||
lt := func(a, b string) {
|
||||
t.Helper()
|
||||
av, err1 := util.ParseVersion(a)
|
||||
bv, err2 := util.ParseVersion(b)
|
||||
if err1 != nil || err2 != nil {
|
||||
t.Fatalf("parse %q/%q: %v %v", a, b, err1, err2)
|
||||
}
|
||||
if av.Compare(bv) != -1 || bv.Compare(av) != 1 {
|
||||
t.Errorf("want %s < %s", a, b)
|
||||
}
|
||||
}
|
||||
lt("0.81.0", "0.82.0")
|
||||
lt("0.81.0", "0.100.0") // numeric minor, not lexicographic
|
||||
lt("0.99.9", "1.0.0")
|
||||
lt("1.2.3", "1.2.10")
|
||||
if v, err := util.ParseVersion("v0.82.0"); err != nil || v.Raw != "0.82.0" {
|
||||
t.Errorf("v-prefix must parse: %v %v", v, err)
|
||||
}
|
||||
if eq, _ := util.ParseVersion("0.81.0"); eq.Compare(eq) != 0 {
|
||||
t.Error("equal versions must compare 0")
|
||||
}
|
||||
// Pre-release suffixes are REJECTED by the house comparator — in Supports they mean "fall back
|
||||
// to the probe", never a trusted comparison.
|
||||
for _, bad := range []string{"1.2.3-rc1", "dev", "latest", "", "1.2", "a.b.c"} {
|
||||
if _, err := util.ParseVersion(bad); err == nil {
|
||||
t.Errorf("ParseVersion(%q) must error", bad)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- B3.5: wire-level — the header decides over the probe through the REAL pinned client --------
|
||||
// A ROUTELESS agent (probe would say No) that sends X-Felhom-Agent-Version 9.9.9 must be judged by
|
||||
// the VERSION (Yes): proof the channel takes precedence end-to-end, and that capture happens on an
|
||||
// ordinary (even failing) call.
|
||||
func TestClient_VersionHeaderWins_WireLevel(t *testing.T) {
|
||||
mux := http.NewServeMux() // NO /netstorage/verify-status (the pre-0.81 route shape)
|
||||
mux.HandleFunc("/storage", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("X-Felhom-Agent-Version", "9.9.9")
|
||||
_, _ = w.Write([]byte(`{"ok":true,"data":{"vmid":1,"mounts":[]}}`))
|
||||
})
|
||||
wrapped := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("X-Felhom-Agent-Version", "9.9.9") // every response, like the agent's wrap
|
||||
mux.ServeHTTP(w, r)
|
||||
})
|
||||
srv := httptest.NewTLSServer(wrapped)
|
||||
defer srv.Close()
|
||||
fp := sha256.Sum256(srv.Certificate().Raw)
|
||||
c, err := New(strings.TrimPrefix(srv.URL, "https://"), "test-token", hex.EncodeToString(fp[:]))
|
||||
if err != nil {
|
||||
t.Fatalf("New: %v", err)
|
||||
}
|
||||
if _, err := c.Storage(context.Background()); err != nil {
|
||||
t.Fatalf("storage: %v", err)
|
||||
}
|
||||
if got := c.AgentVersion(); got != "9.9.9" {
|
||||
t.Fatalf("AgentVersion = %q, want 9.9.9 (passive capture)", got)
|
||||
}
|
||||
if got := c.Supports(context.Background(), FeatureNetstorageVerify); got != SupportYes {
|
||||
t.Errorf("Supports = %v, want SupportYes via version despite the missing probe route", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A GARBAGE header must never be captured: the routeless agent stays on the probe → SupportNo.
|
||||
func TestClient_GarbageHeaderIgnored_WireLevel(t *testing.T) {
|
||||
wrapped := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("X-Felhom-Agent-Version", "not-a-version;x")
|
||||
http.NotFound(w, r)
|
||||
})
|
||||
srv := httptest.NewTLSServer(wrapped)
|
||||
defer srv.Close()
|
||||
fp := sha256.Sum256(srv.Certificate().Raw)
|
||||
c, err := New(strings.TrimPrefix(srv.URL, "https://"), "test-token", hex.EncodeToString(fp[:]))
|
||||
if err != nil {
|
||||
t.Fatalf("New: %v", err)
|
||||
}
|
||||
_, _ = c.Storage(context.Background()) // 404s; the garbage header must be dropped at capture
|
||||
if got := c.AgentVersion(); got != "" {
|
||||
t.Fatalf("AgentVersion = %q, want empty (strict validation at capture)", got)
|
||||
}
|
||||
if got := c.Supports(context.Background(), FeatureNetstorageVerify); got != SupportNo {
|
||||
t.Errorf("Supports = %v, want SupportNo via the probe fallback", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
package agentapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// memProber satisfies the SupportProber contract (NetVerifyStatus) PLUS the memory probe's
|
||||
// type-asserted GuestMemory + the AgentVersion fast-path — the shape the web memAgent seam has.
|
||||
type memProber struct {
|
||||
memErr error
|
||||
ver string
|
||||
}
|
||||
|
||||
func (p *memProber) NetVerifyStatus(context.Context) (NetVerifyStatus, error) {
|
||||
return NetVerifyStatus{Phase: "none"}, nil
|
||||
}
|
||||
func (p *memProber) GuestMemory(context.Context) (GuestMemoryInfo, error) {
|
||||
return GuestMemoryInfo{AllocatedMB: 8192}, p.memErr
|
||||
}
|
||||
func (p *memProber) AgentVersion() string { return p.ver }
|
||||
|
||||
// The capability table: v0.90.0 → Yes, v0.89.0 → No via the version fast-path; the probe (no
|
||||
// version) classifies 404 → No, nil → Yes.
|
||||
func TestGuestMemory_Capability(t *testing.T) {
|
||||
t.Run("version >= 0.90 → Yes", func(t *testing.T) {
|
||||
var sc SupportCache
|
||||
if got := sc.Supports(context.Background(), &memProber{ver: "0.90.0"}, FeatureGuestMemoryResize); got != SupportYes {
|
||||
t.Errorf("v0.90.0 = %v, want SupportYes", got)
|
||||
}
|
||||
})
|
||||
t.Run("version < 0.90 → No", func(t *testing.T) {
|
||||
var sc SupportCache
|
||||
if got := sc.Supports(context.Background(), &memProber{ver: "0.89.0"}, FeatureGuestMemoryResize); got != SupportNo {
|
||||
t.Errorf("v0.89.0 = %v, want SupportNo", got)
|
||||
}
|
||||
})
|
||||
t.Run("no version, probe 404 → No", func(t *testing.T) {
|
||||
var sc SupportCache
|
||||
p := &memProber{memErr: &StatusError{Path: "/guest/memory", Code: http.StatusNotFound}}
|
||||
if got := sc.Supports(context.Background(), p, FeatureGuestMemoryResize); got != SupportNo {
|
||||
t.Errorf("probe 404 = %v, want SupportNo", got)
|
||||
}
|
||||
})
|
||||
t.Run("no version, probe ok → Yes", func(t *testing.T) {
|
||||
var sc SupportCache
|
||||
if got := sc.Supports(context.Background(), &memProber{}, FeatureGuestMemoryResize); got != SupportYes {
|
||||
t.Errorf("probe ok = %v, want SupportYes", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// GuestMemory decodes the agent's payload; a 404 (pre-0.90) is the typed *StatusError.
|
||||
func TestClient_GuestMemory(t *testing.T) {
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/guest/memory", func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodGet {
|
||||
w.WriteHeader(405)
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte(`{"ok":true,"data":{"vmid":9201,"allocated_mb":8192,"usage_mb":3000,"host_total_mb":16384,"min_mb":2048,"max_mb":14336,"floor_mb":3512,"running":true}}`))
|
||||
})
|
||||
c := newMemTestClient(t, mux)
|
||||
info, err := c.GuestMemory(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("GuestMemory: %v", err)
|
||||
}
|
||||
if info.AllocatedMB != 8192 || info.UsageMB != 3000 || info.MaxMB != 14336 || info.FloorMB != 3512 || !info.Running {
|
||||
t.Errorf("decoded wrong: %+v", info)
|
||||
}
|
||||
}
|
||||
|
||||
func TestClient_GuestMemory_404(t *testing.T) {
|
||||
c := newMemTestClient(t, http.NewServeMux()) // no route → 404
|
||||
_, err := c.GuestMemory(context.Background())
|
||||
var se *StatusError
|
||||
if !errors.As(err, &se) || se.Code != http.StatusNotFound {
|
||||
t.Fatalf("pre-0.90 agent must 404 as *StatusError, got %T: %v", err, err)
|
||||
}
|
||||
}
|
||||
|
||||
// ResizeMemory: success returns old→new; a 412 refusal surfaces *MemoryRefusedError with the code.
|
||||
func TestClient_ResizeMemory(t *testing.T) {
|
||||
t.Run("success", func(t *testing.T) {
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/guest/memory", func(w http.ResponseWriter, r *http.Request) {
|
||||
_, _ = w.Write([]byte(`{"ok":true,"data":{"vmid":9201,"old_mb":8192,"new_mb":12288,"unchanged":false}}`))
|
||||
})
|
||||
c := newMemTestClient(t, mux)
|
||||
res, err := c.ResizeMemory(context.Background(), 12288)
|
||||
if err != nil {
|
||||
t.Fatalf("ResizeMemory: %v", err)
|
||||
}
|
||||
if res.OldMB != 8192 || res.NewMB != 12288 {
|
||||
t.Errorf("result = %+v", res)
|
||||
}
|
||||
})
|
||||
t.Run("refusal carries the code", func(t *testing.T) {
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/guest/memory", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusPreconditionFailed)
|
||||
_, _ = w.Write([]byte(`{"ok":false,"error":"requested 3300 MB is too close to current usage 3000 MB (floor 3512 MB)","data":{"code":"below_usage_floor","usage_mb":3000,"floor_mb":3512,"min_mb":2048,"max_mb":14336}}`))
|
||||
})
|
||||
c := newMemTestClient(t, mux)
|
||||
_, err := c.ResizeMemory(context.Background(), 3300)
|
||||
var refused *MemoryRefusedError
|
||||
if !errors.As(err, &refused) {
|
||||
t.Fatalf("want *MemoryRefusedError, got %T: %v", err, err)
|
||||
}
|
||||
if refused.Code != "below_usage_floor" || refused.Bounds.UsageMB != 3000 || refused.Bounds.FloorMB != 3512 {
|
||||
t.Errorf("refusal = %+v", refused)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func newMemTestClient(t *testing.T, mux *http.ServeMux) *Client {
|
||||
t.Helper()
|
||||
srv := httptest.NewTLSServer(mux)
|
||||
t.Cleanup(srv.Close)
|
||||
fp := sha256.Sum256(srv.Certificate().Raw)
|
||||
c, err := New(strings.TrimPrefix(srv.URL, "https://"), "test-token", hex.EncodeToString(fp[:]))
|
||||
if err != nil {
|
||||
t.Fatalf("New: %v", err)
|
||||
}
|
||||
return c
|
||||
}
|
||||
@@ -9,8 +9,8 @@ import (
|
||||
)
|
||||
|
||||
func netStub(t *testing.T) (*httptest.Server, string, *struct {
|
||||
addBody AddNetStorageRequest
|
||||
removed string
|
||||
addBody AddNetStorageRequest
|
||||
removed string
|
||||
}) {
|
||||
captured := &struct {
|
||||
addBody AddNetStorageRequest
|
||||
|
||||
@@ -27,6 +27,7 @@ func (p *snapshotsStubProvider) GetStackComposePath(name string) (string, bool)
|
||||
func (p *snapshotsStubProvider) ListDeployedStacks() []backup.StackSummary { return nil }
|
||||
func (p *snapshotsStubProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *snapshotsStubProvider) GetStackHDDPath(string) string { return p.hdd }
|
||||
func (p *snapshotsStubProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
func (p *snapshotsStubProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *snapshotsStubProvider) StopStack(string) error { return nil }
|
||||
func (p *snapshotsStubProvider) StartStack(string) error { return nil }
|
||||
@@ -34,10 +35,14 @@ func (p *snapshotsStubProvider) RefreshAndIsRunning(string) bool { ret
|
||||
func (p *snapshotsStubProvider) GetStackRecoveryInfo(string) (backup.RecoveryInfo, bool) {
|
||||
return backup.RecoveryInfo{}, false
|
||||
}
|
||||
func (p *snapshotsStubProvider) GetStackClassifiedBinds(string) ([]backup.ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
func (p *snapshotsStubProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *snapshotsStubProvider) RecreateStackFromUnit(string, string, map[string]string) error {
|
||||
func (p *snapshotsStubProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *snapshotsStubProvider) StartStackServices(string, []string) error { return nil }
|
||||
|
||||
// newSnapshotsRouter wires a Router with a real backup.Manager over a tempdir drive.
|
||||
func newSnapshotsRouter(t *testing.T) (*Router, string) {
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||||
)
|
||||
|
||||
func newDeployGateRouter(t *testing.T, class string) *Router {
|
||||
t.Helper()
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
sett, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), lg)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.AddStoragePath(settings.StoragePath{
|
||||
Path: "/mnt/felhom-drives/nas-media", Label: "NAS", Schedulable: true, Kind: settings.StorageKindNetwork,
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.AddStoragePath(settings.StoragePath{
|
||||
Path: "/mnt/felhom-drives/felhom-usb", Label: "USB", Schedulable: true,
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
r := &Router{sett: sett, logger: lg}
|
||||
r.classifyFSPath = func(string) string { return class }
|
||||
return r
|
||||
}
|
||||
|
||||
// The deploy-time stub gate (RCA fix 2): a registered network HDD_PATH that classifies as a STUB
|
||||
// in this namespace refuses with the §2.3 Hungarian message; the healthy idle autofs trigger, a
|
||||
// live mount, an unknown verdict (fail open), local paths and empty paths all proceed.
|
||||
// Companion red-proofs:
|
||||
// - remove the gate call from deployStack → the refusal test fails (deploy proceeds onto a stub);
|
||||
// - an impl requiring MOUNTED-only (refusing autofs) → the idle-autofs row fails (it would
|
||||
// wrongly block deploying onto a healthy idle share).
|
||||
func TestRefuseNetworkStubDeploy_Table(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
class string
|
||||
hdd string
|
||||
refuse bool
|
||||
}{
|
||||
{"stub on network path → REFUSE", system.FSClassStub, "/mnt/felhom-drives/nas-media", true},
|
||||
{"idle autofs on network path → proceed (healthy)", system.FSClassAutofs, "/mnt/felhom-drives/nas-media", false},
|
||||
{"live network fs → proceed", system.FSClassNetwork, "/mnt/felhom-drives/nas-media", false},
|
||||
{"unknown (timeout) → proceed (fail open)", system.FSClassUnknown, "/mnt/felhom-drives/nas-media", false},
|
||||
{"local path → proceed even when classifier says stub", system.FSClassStub, "/mnt/felhom-drives/felhom-usb", false},
|
||||
{"unregistered path → proceed", system.FSClassStub, "/mnt/elsewhere", false},
|
||||
{"empty HDD_PATH (SSD app) → proceed", system.FSClassStub, "", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
r := newDeployGateRouter(t, c.class)
|
||||
msg := r.refuseNetworkStubDeploy(c.hdd)
|
||||
if c.refuse && !strings.Contains(msg, "a telepítés nem indítható") {
|
||||
t.Errorf("%s: want the refusal message, got %q", c.name, msg)
|
||||
}
|
||||
if !c.refuse && msg != "" {
|
||||
t.Errorf("%s: must proceed, got refusal %q", c.name, msg)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// R-166 Part 1.3 — THE CUSTOMER-INTENT POINT.
|
||||
//
|
||||
// `stackMgr` is a concrete *stacks.Manager, so actionStack cannot be driven with a fake without
|
||||
// Docker. The two properties that actually carry the correctness are therefore pinned the only way
|
||||
// they can be: the mapping is a pure function with its own table test, and the ORDER (§8.2) is
|
||||
// asserted structurally over actionStack's AST. Both fail if someone reverses the write and the act,
|
||||
// which is the mistake that would undo a customer's Stop at the next boot.
|
||||
|
||||
func TestDesiredStateForAction_MapsEveryAction(t *testing.T) {
|
||||
cases := []struct {
|
||||
action string
|
||||
want string
|
||||
ok bool
|
||||
}{
|
||||
{"start", stacks.DesiredStateRunning, true},
|
||||
// restart and update both END in `compose up -d`, so a customer who presses either is asking
|
||||
// for the app to be up afterwards.
|
||||
{"restart", stacks.DesiredStateRunning, true},
|
||||
{"update", stacks.DesiredStateRunning, true},
|
||||
{"stop", stacks.DesiredStateStopped, true},
|
||||
// Anything unrecognised records NOTHING rather than guessing — a future action must not
|
||||
// silently acquire an intent it was never meant to carry.
|
||||
{"", "", false},
|
||||
{"delete", "", false},
|
||||
{"pause", "", false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
got, ok := desiredStateForAction(tc.action)
|
||||
if got != tc.want || ok != tc.ok {
|
||||
t.Fatalf("desiredStateForAction(%q) = (%q, %v), want (%q, %v)", tc.action, got, ok, tc.want, tc.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDesiredStateForAction_NeverRecordsStoppedForANonStop(t *testing.T) {
|
||||
// The asymmetry that matters: writing "stopped" for anything other than a Stop would permanently
|
||||
// disable auto-recovery for an app nobody stopped.
|
||||
for _, a := range []string{"start", "restart", "update", "deploy", "delete", ""} {
|
||||
if got, _ := desiredStateForAction(a); got == stacks.DesiredStateStopped {
|
||||
t.Fatalf("action %q maps to desired_state=stopped", a)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestActionStack_RecordsIntentBeforeActing is §8.2, asserted structurally.
|
||||
//
|
||||
// If the SetDesiredState call moved BELOW the action switch, a stop could remove every container
|
||||
// while app.yaml still recorded `running` — and the boot reconciler would then start an app the
|
||||
// customer had just deliberately stopped. That is the single worst outcome available in Part 1, and
|
||||
// no behavioural test in this package can reach it without a Docker daemon.
|
||||
func TestActionStack_RecordsIntentBeforeActing(t *testing.T) {
|
||||
body := funcBody(t, "actionStack")
|
||||
|
||||
setPos, switchPos := -1, -1
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
switch node := n.(type) {
|
||||
case *ast.CallExpr:
|
||||
if sel, ok := node.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "SetDesiredState" && setPos < 0 {
|
||||
setPos = int(node.Pos())
|
||||
}
|
||||
case *ast.SwitchStmt:
|
||||
// The action switch is the one whose tag is the `action` identifier.
|
||||
if id, ok := node.Tag.(*ast.Ident); ok && id.Name == "action" && switchPos < 0 {
|
||||
switchPos = int(node.Pos())
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if setPos < 0 {
|
||||
t.Fatal("actionStack no longer calls SetDesiredState — the customer's start/stop decision is " +
|
||||
"recorded nowhere, which is the R-166 defect un-fixed")
|
||||
}
|
||||
if switchPos < 0 {
|
||||
t.Fatal("actionStack no longer has a `switch action` — this test needs updating")
|
||||
}
|
||||
if setPos >= switchPos {
|
||||
t.Fatal("actionStack records the desired state AFTER performing the action (§8.2 violated): a " +
|
||||
"stop whose intent write fails or lands late leaves zero containers with `running` " +
|
||||
"recorded, and the boot reconciler would restart an app the customer just stopped")
|
||||
}
|
||||
}
|
||||
|
||||
// TestActionStack_RefusesTheActionWhenIntentCannotBeRecorded pins the other half of §8.2: a failed
|
||||
// write REFUSES the act. Proceeding anyway would perform a stop that nothing records — exactly the
|
||||
// ambiguity this release removes.
|
||||
func TestActionStack_RefusesTheActionWhenIntentCannotBeRecorded(t *testing.T) {
|
||||
body := funcBody(t, "actionStack")
|
||||
|
||||
refuses := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
ifst, ok := n.(*ast.IfStmt)
|
||||
if !ok || ifst.Init == nil {
|
||||
return true
|
||||
}
|
||||
// Look for `if derr := ...SetDesiredState(...); derr != nil { ... return }`
|
||||
assign, ok := ifst.Init.(*ast.AssignStmt)
|
||||
if !ok || len(assign.Rhs) != 1 {
|
||||
return true
|
||||
}
|
||||
call, ok := assign.Rhs[0].(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || sel.Sel.Name != "SetDesiredState" {
|
||||
return true
|
||||
}
|
||||
for _, stmt := range ifst.Body.List {
|
||||
if _, isReturn := stmt.(*ast.ReturnStmt); isReturn {
|
||||
refuses = true
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if !refuses {
|
||||
t.Fatal("actionStack does not RETURN when SetDesiredState fails — it would go on to stop or " +
|
||||
"start an app whose intent could not be recorded (§8.2)")
|
||||
}
|
||||
}
|
||||
|
||||
// funcBody parses router.go and returns the named method's body.
|
||||
func funcBody(t *testing.T, name string) *ast.BlockStmt {
|
||||
t.Helper()
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse router.go: %v", err)
|
||||
}
|
||||
for _, decl := range f.Decls {
|
||||
if fn, ok := decl.(*ast.FuncDecl); ok && fn.Name.Name == name && fn.Body != nil {
|
||||
return fn.Body
|
||||
}
|
||||
}
|
||||
t.Fatalf("func %s not found in router.go", name)
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestDeployStackWiresTheLifecycleGate — the seam-discipline test (§9 rule 6).
|
||||
//
|
||||
// The lifecycle predicate is unit-tested in internal/stacks, and a test there passes whether or not
|
||||
// deployStack ever calls it. Three inert-seam defects shipped fully-green in three days (controller
|
||||
// v0.154.0, agent v0.91.0, agent v0.92.0's missing sudoers grant), all this exact shape: correct
|
||||
// component, absent caller. So the CALLER is asserted here, from source.
|
||||
//
|
||||
// It walks the AST rather than doing strings.Contains on the file, because a commented-out call
|
||||
// still contains the string — the lesson recorded in PROMPT-TEMPLATE §10.
|
||||
//
|
||||
// It also asserts ORDER: the gate must precede the DeployStack call, or it is not fail-closed.
|
||||
func TestDeployStackWiresTheLifecycleGate(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0) // comments dropped: a commented call is not a call
|
||||
if err != nil {
|
||||
t.Fatalf("parse router.go: %v", err)
|
||||
}
|
||||
|
||||
var fn *ast.FuncDecl
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
if d, ok := n.(*ast.FuncDecl); ok && d.Name.Name == "deployStack" {
|
||||
fn = d
|
||||
return false
|
||||
}
|
||||
return true
|
||||
})
|
||||
if fn == nil {
|
||||
t.Fatal("deployStack not found in router.go — did it move? the gate's wiring is now unasserted")
|
||||
}
|
||||
|
||||
canInstallPos, deployPos := -1, -1
|
||||
ast.Inspect(fn, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
off := fset.Position(call.Pos()).Offset
|
||||
switch sel.Sel.Name {
|
||||
case "CanInstall":
|
||||
if canInstallPos == -1 {
|
||||
canInstallPos = off
|
||||
}
|
||||
case "DeployStack":
|
||||
if deployPos == -1 {
|
||||
deployPos = off
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if canInstallPos == -1 {
|
||||
t.Fatal("deployStack never calls Meta.CanInstall() — the lifecycle gate is INERT: " +
|
||||
"a withdrawn app is hidden from the catalog page but still installable by direct POST")
|
||||
}
|
||||
if deployPos == -1 {
|
||||
t.Fatal("deployStack no longer calls DeployStack — this test's ordering assertion is meaningless")
|
||||
}
|
||||
if canInstallPos > deployPos {
|
||||
t.Fatalf("the lifecycle gate (offset %d) runs AFTER DeployStack (offset %d) — a gate that "+
|
||||
"fires after the mutation is not fail-closed", canInstallPos, deployPos)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLifecycleRefusalMessageIsCustomerFacingHungarian: the refusal text reaches the customer via
|
||||
// showAlert(), so it must be the sentence the spec ruled, not a Go error string.
|
||||
func TestLifecycleRefusalMessageIsCustomerFacingHungarian(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
const want = "Ez az alkalmazás jelenleg nem telepíthető."
|
||||
found := false
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
if lit, ok := n.(*ast.BasicLit); ok && lit.Kind == token.STRING && strings.Contains(lit.Value, want) {
|
||||
found = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !found {
|
||||
t.Fatalf("the ruled refusal message %q is not present in router.go", want)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package api
|
||||
|
||||
import "testing"
|
||||
|
||||
// Group D (Scenario D, v0.139.0): with the seam unset (hub reporting disabled → main.go
|
||||
// never calls SetReportPushTrigger), reportPushNow is a strict nil-safe no-op — the
|
||||
// deploy/remove/geo handlers calling it must never panic.
|
||||
func TestReportPushNow_NilSeamIsNoOp(t *testing.T) {
|
||||
r := &Router{}
|
||||
r.reportPushNow() // must not panic with triggerReportPush == nil
|
||||
}
|
||||
@@ -47,6 +47,10 @@ type Router struct {
|
||||
// process back with fresh config; tests inject a recorder via SetRestarter.
|
||||
restart func()
|
||||
|
||||
// classifyFSPath classifies a path's filesystem in this process's namespace (the deploy-time
|
||||
// stub gate, RCA fix 2). Defaults to system.ClassifyPathFSTimeout; tests inject fake classes.
|
||||
classifyFSPath func(path string) string
|
||||
|
||||
// triggerReportPush fires an out-of-band, non-blocking hub report push (e.g. after a
|
||||
// geo settings change so the hub reflects the new state immediately). Nil = no-op.
|
||||
triggerReportPush func()
|
||||
@@ -95,6 +99,7 @@ func (r *Router) SetIntegrationManager(im *integrations.Manager) {
|
||||
func NewRouter(cfg *config.Config, configPath string, sett *settings.Settings, stackMgr *stacks.Manager, syncer *catalogsync.Syncer, cpuCollector *system.CPUCollector, backupMgr *backup.Manager, metricsStore *metrics.MetricsStore, updater *selfupdate.Updater, notif *notify.Notifier, logger *log.Logger) *Router {
|
||||
r := &Router{cfg: cfg, configPath: configPath, sett: sett, stackMgr: stackMgr, syncer: syncer, cpuCollector: cpuCollector, backupMgr: backupMgr, metricsStore: metricsStore, updater: updater, notifier: notif, logger: logger}
|
||||
r.restart = func() { gracefulSelfRestart(r.logger) }
|
||||
r.classifyFSPath = system.ClassifyPathFSTimeout
|
||||
return r
|
||||
}
|
||||
|
||||
@@ -102,6 +107,21 @@ func NewRouter(cfg *config.Config, configPath string, sett *settings.Settings, s
|
||||
// process is not actually killed.
|
||||
func (r *Router) SetRestarter(fn func()) { r.restart = fn }
|
||||
|
||||
// refuseNetworkStubDeploy is the deploy-time stub gate (RCA fix 2). Non-empty return = the
|
||||
// Hungarian refusal for a registered NETWORK HDD_PATH whose filesystem in THIS namespace is a
|
||||
// local stub. Everything else proceeds: idle autofs is HEALTHY (first app access mounts it);
|
||||
// classification timeout/unknown fails OPEN (a wedged share is the unreachable badge's business);
|
||||
// local (non-network) and empty paths keep today's behavior exactly.
|
||||
func (r *Router) refuseNetworkStubDeploy(hdd string) string {
|
||||
if hdd == "" || r.sett == nil || !r.sett.IsNetworkStoragePath(hdd) {
|
||||
return ""
|
||||
}
|
||||
if r.classifyFSPath(hdd) != system.FSClassStub {
|
||||
return ""
|
||||
}
|
||||
return "A kiválasztott hálózati tárhely jelenleg nem érhető el az alkalmazások környezetéből — a telepítés nem indítható. Próbálja újra pár perc múlva, vagy jelezze az üzemeltetőnek."
|
||||
}
|
||||
|
||||
// SetReportPushTrigger wires the out-of-band hub report push used after geo changes.
|
||||
// The provided func MUST be non-blocking (it is called from request handlers).
|
||||
func (r *Router) SetReportPushTrigger(fn func()) { r.triggerReportPush = fn }
|
||||
@@ -390,6 +410,21 @@ func (r *Router) deployStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
return
|
||||
}
|
||||
|
||||
// Lifecycle gate: an app withdrawn from the catalog (`lifecycle: hidden` / `abandoned`) is not
|
||||
// installable. FAIL-CLOSED and server-side on purpose — the catalog page already omits these, so
|
||||
// anything reaching here is a stale link, a bookmarked deploy form, or a direct POST, and a gate
|
||||
// that only hides the button is not a gate. Deliberately BEFORE every mutation.
|
||||
//
|
||||
// This does NOT touch an already-deployed instance: it is on the deploy path only, and the
|
||||
// manager refuses a redeploy of an existing stack through its own "already deployed" check.
|
||||
if st, ok := r.stackMgr.GetStack(name); ok && !st.Meta.CanInstall() {
|
||||
r.logger.Printf("[WARN] [api] Deploy refused for %s: lifecycle=%s (not offered for new installs)",
|
||||
name, st.Meta.EffectiveLifecycle())
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false,
|
||||
Error: "Ez az alkalmazás jelenleg nem telepíthető."})
|
||||
return
|
||||
}
|
||||
|
||||
// Prevention layer (storage-split): refuse a deploy when the Docker-data volume is at/under its
|
||||
// reserved buffer, so customer apps can't fill the volume the infra containers (controller,
|
||||
// traefik, cloudflared, filebrowser) depend on. Fail-OPEN on a measurement error — the buffer is
|
||||
@@ -403,6 +438,30 @@ func (r *Router) deployStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
return
|
||||
}
|
||||
|
||||
// RCA fix 2 (AUDIT-nas-cwa-rca-2026-07-11): a deploy targeting a registered NETWORK storage path
|
||||
// must see a network filesystem (or its healthy idle autofs trigger) in THIS namespace — the one
|
||||
// the app will consume the path in. A stub (plain local dir after a guest reboot) would silently
|
||||
// send the app's data to the system drive.
|
||||
if msg := r.refuseNetworkStubDeploy(body.Values["HDD_PATH"]); msg != "" {
|
||||
r.logger.Printf("[WARN] [api] Deploy refused for %s: network HDD_PATH %s is a stub in the controller namespace", name, body.Values["HDD_PATH"])
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false, Error: msg})
|
||||
return
|
||||
}
|
||||
|
||||
// R-108: an app's data namespace may NOT live on network storage — its backups would land at
|
||||
// `<share>/backups/primary/<stack>/`, inside the share-ROOT bind FileBrowser serves with
|
||||
// download:true (and that bind cannot be narrowed — see settings.RefuseAsAppNamespace).
|
||||
//
|
||||
// THIS is the boundary, not the deploy dropdown. The dropdown is a UI list; this endpoint accepts
|
||||
// whatever HDD_PATH a caller supplies and `DeployStack` validates only that it EXISTS on the
|
||||
// filesystem (os.Stat, internal/stacks/deploy.go). A filter on the list alone would have left the
|
||||
// surface wide open — the R-108 row's "no IsNetwork() filter on the dropdown" understates it.
|
||||
if refuse, why := r.sett.RefuseAsAppNamespace(body.Values["HDD_PATH"]); refuse {
|
||||
r.logger.Printf("[WARN] [api] Deploy refused for %s: HDD_PATH is not usable as an app namespace (R-108)", name)
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false, Error: why})
|
||||
return
|
||||
}
|
||||
|
||||
deployReq := stacks.DeployRequest{
|
||||
StackName: name,
|
||||
Values: body.Values,
|
||||
@@ -446,6 +505,9 @@ func (r *Router) deployStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
go r.OnGeoRelevantChange()
|
||||
}
|
||||
|
||||
// v0.139.0: the hub sees the deploy in seconds (debounced trigger, not per-request)
|
||||
r.reportPushNow()
|
||||
|
||||
// Re-apply integrations that target this newly deployed stack
|
||||
if r.integrationMgr != nil {
|
||||
go r.integrationMgr.OnStackStart(context.Background(), name)
|
||||
@@ -472,6 +534,24 @@ func (r *Router) startGatedByMissingDrive(name string) (bool, string) {
|
||||
return false, ""
|
||||
}
|
||||
|
||||
// desiredStateForAction maps a stack action to the customer intent it expresses, or (_, false) for
|
||||
// an action that expresses none. Pure, so the §8.1/§1.3 mapping is testable without a Manager.
|
||||
//
|
||||
// `restart` and `update` both mean running: a customer who updates or restarts an app is asking for
|
||||
// it to be up afterwards, and both end in `compose up -d`. Anything not listed here — an unknown
|
||||
// action string — records nothing rather than guessing, so a future action cannot silently acquire
|
||||
// an intent it was never meant to carry.
|
||||
func desiredStateForAction(action string) (string, bool) {
|
||||
switch action {
|
||||
case "start", "restart", "update":
|
||||
return stacks.DesiredStateRunning, true
|
||||
case "stop":
|
||||
return stacks.DesiredStateStopped, true
|
||||
default:
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
|
||||
func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
r.logger.Printf("[INFO] [api] %s requested for stack: %s", action, name)
|
||||
r.dbg("actionStack: action=%s name=%s", action, name)
|
||||
@@ -517,6 +597,26 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
}
|
||||
}
|
||||
|
||||
// R-166: THE CUSTOMER-INTENT POINT. This switch is where a human's decision about whether their
|
||||
// app should be running enters the system, and until v0.189.0 that decision was recorded nowhere
|
||||
// — so the box had to infer it from container counts, and inferred wrong for a power cut and for
|
||||
// an interrupted backup alike.
|
||||
//
|
||||
// Written BEFORE the act (§8.2) and a failed write REFUSES the act: performing a stop whose
|
||||
// intent could not be recorded would recreate exactly the ambiguity this closes. Both gates that
|
||||
// can legitimately refuse an action (protected-stack, drive-absent, memory) have already run
|
||||
// above, so nothing is recorded for an action that was never going to happen.
|
||||
if desired, ok := desiredStateForAction(action); ok {
|
||||
if derr := r.stackMgr.SetDesiredState(name, desired); derr != nil {
|
||||
r.logger.Printf("[ERROR] [api] %s for %s refused: could not record desired state: %v", action, name, derr)
|
||||
writeJSON(w, http.StatusInternalServerError, apiResponse{
|
||||
OK: false,
|
||||
Error: "A művelet nem hajtható végre: az alkalmazás beállításai nem menthetők.",
|
||||
})
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
var err error
|
||||
switch action {
|
||||
case "start":
|
||||
@@ -744,6 +844,9 @@ func (r *Router) removeStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
if r.OnGeoRelevantChange != nil {
|
||||
go r.OnGeoRelevantChange()
|
||||
}
|
||||
|
||||
// v0.139.0: the hub sees the removal in seconds (debounced trigger, not per-request)
|
||||
r.reportPushNow()
|
||||
}
|
||||
|
||||
func (r *Router) deleteStack(w http.ResponseWriter, req *http.Request, name string) {
|
||||
@@ -775,6 +878,9 @@ func (r *Router) deleteStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, apiResponse{OK: true, Data: resp, Message: "Stack " + name + " deleted"})
|
||||
|
||||
// v0.139.0: the hub sees the delete in seconds (debounced trigger, not per-request)
|
||||
r.reportPushNow()
|
||||
}
|
||||
|
||||
func (r *Router) triggerSync(w http.ResponseWriter, _ *http.Request) {
|
||||
@@ -904,6 +1010,7 @@ func (r *Router) triggerBackup(w http.ResponseWriter, _ *http.Request) {
|
||||
}
|
||||
|
||||
r.logger.Println("[INFO] [api] Manual app-data backup (DB dump) triggered")
|
||||
r.backupMgr.MarkManualRun() // R-182: operator-triggered — its digest must not be collapsed into the nightly one
|
||||
go r.backupMgr.RunDBDumps(context.Background())
|
||||
|
||||
writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Mentés elindítva"})
|
||||
|
||||
@@ -19,15 +19,19 @@ type StackDataProvider interface {
|
||||
GetStackComposePath(name string) (composePath string, ok bool)
|
||||
ListDeployedStacks() []StackSummary
|
||||
GetStackHDDMounts(name string) []string
|
||||
GetStackHDDPath(name string) string // raw HDD_PATH from app.yaml (empty if no HDD)
|
||||
GetStackHDDPath(name string) string // raw HDD_PATH from app.yaml (empty if no HDD)
|
||||
// GetImportRoot returns the CANONICAL drop-zone root (R-75): <system namespace root>/userdata/import.
|
||||
// It is app-INDEPENDENT and lives on the SYSTEM drive, so ${IMPORT_PATH} binds cannot be resolved
|
||||
// from GetStackHDDPath. Empty when unresolvable — structuralGuard refuses such binds loudly.
|
||||
GetImportRoot() string
|
||||
GetDockerVolumes(name string) []string // full Docker volume names (project-prefixed)
|
||||
StopStack(name string) error
|
||||
StartStack(name string) error
|
||||
RefreshAndIsRunning(name string) bool
|
||||
// GetStackRecoveryInfo returns the data needed to capture a SECRET-FREE recovery unit
|
||||
// (Phase 2): the stack dir, pinned image tags, the non-secret env, and the NAMES of the
|
||||
// secret/data-key env vars (values are NEVER returned — they are recovered at restore time
|
||||
// from the guest's own app.yaml, live or via the PBS whole-guest snapshot). ok=false if the
|
||||
// GetStackRecoveryInfo returns the data needed to capture a recovery unit: the stack dir,
|
||||
// pinned image tags, the non-secret env, the NAMES of the secret/data-key env vars, and (D5)
|
||||
// the decrypted VALUES of the portable class. A WITHHELD secret's value is never returned —
|
||||
// it is recovered at restore time from the guest's app.yaml, or regenerated. ok=false if the
|
||||
// stack is unknown.
|
||||
GetStackRecoveryInfo(name string) (RecoveryInfo, bool)
|
||||
|
||||
@@ -39,23 +43,46 @@ type StackDataProvider interface {
|
||||
// fail-closed gate decides what to do. The unit is never the source of secrets.
|
||||
RecoverStackSecrets(name string, names []string) map[string]string
|
||||
|
||||
// RecreateStackFromUnit restores an app's definition from the unit's compose dir into the stack
|
||||
// dir, writes app.yaml from fullEnv (encrypting secret fields), and (re-)deploys it via
|
||||
// `docker compose up -d`, which re-pulls the pinned image. Secrets are NEVER regenerated.
|
||||
RecreateStackFromUnit(name, composeSrcDir string, fullEnv map[string]string) error
|
||||
// RecreateStackDefinitionFromUnit restores an app's DEFINITION from the unit's compose dir into
|
||||
// the stack dir and writes app.yaml from fullEnv (encrypting secret fields). Secrets are NEVER
|
||||
// regenerated. It starts NOTHING: the caller owns the bring-up order, because a DB-bearing app
|
||||
// must have its database service started alone for the dump replay (R-47). It was
|
||||
// `RecreateStackFromUnit` until v0.153.0 and ended in a full `docker compose up -d` — that full
|
||||
// start before the replay IS the H4 race.
|
||||
RecreateStackDefinitionFromUnit(name, composeSrcDir string, fullEnv map[string]string) error
|
||||
|
||||
// StartStackServices brings up ONLY the named compose services, leaving the rest of the stack
|
||||
// down — the DB-only window in which a dump is replayed without the application racing it.
|
||||
// Implementations must REFUSE an empty list (an argument-less `up -d` is a full start).
|
||||
StartStackServices(name string, services []string) error
|
||||
|
||||
// GetStackClassifiedBinds returns the app's backup-classified compose binds + whether it carries a
|
||||
// (valid) backup block (Task 2, referential coupling). INERT — no tier consumes it yet; wired now
|
||||
// so Task 3 gets a tested seam. Implemented by delegating to stacks.Manager.ClassifiedBinds.
|
||||
GetStackClassifiedBinds(name string) ([]ClassifiedBind, bool)
|
||||
}
|
||||
|
||||
// RecoveryInfo carries everything needed to write a secret-free recovery unit for a stack.
|
||||
// It deliberately holds NO secret values — only the names of secret/data-key env vars, so the
|
||||
// manifest can record what must be recovered from elsewhere (guest app.yaml / PBS) without the
|
||||
// unit ever storing a secret or a data-encrypting key.
|
||||
// RecoveryInfo carries everything needed to write a recovery unit for a stack.
|
||||
//
|
||||
// D5: it now carries the VALUES of the PORTABLE secret class (stacks.PortableSecretEnvVars — every
|
||||
// `type: secret` field bar the nonPortableSecrets register), because a Tier-1/2 restore that depends
|
||||
// on the guest for a data-encrypting key or a DB password is not independent of the guest at all: the
|
||||
// data sits safely on the drive and cannot be read back. The EXCLUDED class (`type: password` admin
|
||||
// logins) is still name-only and never leaves the guest.
|
||||
type RecoveryInfo struct {
|
||||
StackDir string // dir holding docker-compose.yml + .felhom.yml + app.yaml
|
||||
DisplayName string // app display name
|
||||
ImagePins []string // pinned image tags from compose `image:` lines (re-pulled on restore)
|
||||
NonSecretEnv map[string]string // env with all secret/password/data-key values removed (plaintext only)
|
||||
SecretEnvVars []string // NAMES of stripped secret/password fields (recovered from guest/PBS)
|
||||
NonSecretEnv map[string]string // env with ALL secret/password values removed (plaintext only)
|
||||
SecretEnvVars []string // NAMES of every secret/password field
|
||||
DataKeyEnvVars []string // NAMES of data-encrypting-key fields (fail-closed gate on restore)
|
||||
// PortableSecretEnvVars are the NAMES of the secrets that travel in the unit (D5), and
|
||||
// PortableSecrets their DECRYPTED values. A name present here but absent from PortableSecrets was
|
||||
// unset/empty in the guest's app.yaml — the restore's fail-closed gate decides what that means.
|
||||
// Never logged, never in the manifest's value space: the values reach disk only inside the unit's
|
||||
// 0600 app.yaml.
|
||||
PortableSecretEnvVars []string
|
||||
PortableSecrets map[string]string
|
||||
}
|
||||
|
||||
// ParseComposeImages extracts the pinned image references (`image: repo:tag`) from a
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// fp joins the elements under an HDD path with OS separators — mounts in the ParseComposeHDDMounts
|
||||
// shape are already filepath.Clean'd, so tests build them the same way.
|
||||
func fp(elems ...string) string { return filepath.Join(elems...) }
|
||||
|
||||
// TestAppDataDirNames is the pure derivation table (Group A). Every case asserts the RESOLVED name
|
||||
// list, never mere absence of error. Companion RP-1: a resolver that ignores mounts and returns
|
||||
// []string{stackName} fails the paperless, two-name, and dedupe cases.
|
||||
func TestAppDataDirNames(t *testing.T) {
|
||||
const hdd = "/mnt/felhom-usb"
|
||||
cases := []struct {
|
||||
name string
|
||||
stack string
|
||||
mounts []string
|
||||
want []string
|
||||
}{
|
||||
{
|
||||
// paperless shape: stack "paperless-ngx", dir "paperless" (F-S2/F-S3 core).
|
||||
name: "paperless mismatch",
|
||||
stack: "paperless-ngx",
|
||||
mounts: []string{
|
||||
fp(hdd, "appdata", "paperless", "media"),
|
||||
fp(hdd, "appdata", "paperless", "export"),
|
||||
},
|
||||
want: []string{"paperless"}, // media+export dedupe to one name
|
||||
},
|
||||
{
|
||||
// match shape: dir name == stack name (immich/nextcloud/romm).
|
||||
name: "matching name",
|
||||
stack: "nextcloud",
|
||||
mounts: []string{fp(hdd, "appdata", "nextcloud")},
|
||||
want: []string{"nextcloud"},
|
||||
},
|
||||
{
|
||||
// two DISTINCT names → both, sorted (no catalog app does this today).
|
||||
name: "two distinct names sorted",
|
||||
stack: "weird",
|
||||
mounts: []string{
|
||||
fp(hdd, "appdata", "zebra", "x"),
|
||||
fp(hdd, "appdata", "alpha", "y"),
|
||||
},
|
||||
want: []string{"alpha", "zebra"},
|
||||
},
|
||||
{
|
||||
// non-appdata HDD binds + a foreign-drive mount are filtered → fallback.
|
||||
name: "non-appdata and foreign filtered",
|
||||
stack: "romm",
|
||||
mounts: []string{
|
||||
fp(hdd, "roms"), // under HDD but not appdata/
|
||||
fp("/mnt/other-drive", "appdata", "ghost"), // foreign drive — wrong prefix
|
||||
},
|
||||
want: []string{"romm"},
|
||||
},
|
||||
{
|
||||
// whole-appdata-root bind (no name derivable) → ignored → fallback.
|
||||
name: "whole appdata root bind",
|
||||
stack: "root-binder",
|
||||
mounts: []string{fp(hdd, "appdata")},
|
||||
want: []string{"root-binder"},
|
||||
},
|
||||
{
|
||||
name: "empty mounts fallback",
|
||||
stack: "vaultwarden",
|
||||
mounts: nil,
|
||||
want: []string{"vaultwarden"},
|
||||
},
|
||||
{
|
||||
// unclean paths still resolve (Clean applied both sides).
|
||||
name: "unclean path",
|
||||
stack: "paperless-ngx",
|
||||
mounts: []string{hdd + "/appdata/paperless/../paperless/media"},
|
||||
want: []string{"paperless"},
|
||||
},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := AppDataDirNames(hdd, tc.stack, tc.mounts)
|
||||
if !reflect.DeepEqual(got, tc.want) {
|
||||
t.Errorf("AppDataDirNames(%q, %q, %v) = %v, want %v", hdd, tc.stack, tc.mounts, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestAppDataBindsPresent pins the WARN predicate: true only when a mount sits under appdata/.
|
||||
func TestAppDataBindsPresent(t *testing.T) {
|
||||
const hdd = "/mnt/felhom-usb"
|
||||
if !AppDataBindsPresent(hdd, []string{fp(hdd, "appdata", "paperless", "media")}) {
|
||||
t.Error("declared appdata bind should report present")
|
||||
}
|
||||
if AppDataBindsPresent(hdd, []string{fp(hdd, "roms")}) {
|
||||
t.Error("non-appdata bind should NOT report present")
|
||||
}
|
||||
if AppDataBindsPresent(hdd, nil) {
|
||||
t.Error("no mounts should NOT report present")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,356 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"path"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Capture-set computation — Task 3-core of the backup-classification-redesign arc
|
||||
// (felhom.eu/documentation/architecture/07-backup-architecture.md §3; tier×class matrix §2; SQ2/SQ5
|
||||
// in SPIKE-backup-classification-2026-07-14.md). This is a PURE path-algebra layer: given an app's
|
||||
// classified binds, a tier, and the app's live hddPath, it returns the tier-filtered,
|
||||
// structurally-guarded, containment-deduped absolute capture set that the 3a (offsite) and 3b
|
||||
// (tier-2) engines will capture. Deliberately INERT — no engine consumes it yet.
|
||||
//
|
||||
// Purity contract: no os, no exec, no logging, no filepath. Reasons for refused captures are DATA
|
||||
// (SkippedPath.Reason); the engines log them (a skipped MANDATORY is a capture GAP the engines must
|
||||
// surface loudly). All resolution and prefix algebra uses path.Join/path.Clean and "/" string ops —
|
||||
// never filepath.* — because RelPath is defined forward-slash (classify.go) and every resolved Abs
|
||||
// is an in-container Linux path; filepath on the Windows `go test` host would flip separators and
|
||||
// break both expectations and the containment prefix checks.
|
||||
|
||||
// CaptureTier selects the tier column of §2 that the filter applies.
|
||||
type CaptureTier string
|
||||
|
||||
const (
|
||||
TierOffsite CaptureTier = "offsite" // §2: mandatory only (optional is the customer's local tier)
|
||||
TierSecondary CaptureTier = "secondary" // §2: mandatory + optional
|
||||
)
|
||||
|
||||
// CapturePath is one resolved path in the capture set. Abs is the in-container Linux absolute path;
|
||||
// Root/RelPath preserve the bind's identity (the tier-2 layout and restore relpath-mirroring need it).
|
||||
type CapturePath struct {
|
||||
Abs string
|
||||
Root BindRoot
|
||||
RelPath string
|
||||
Class BindClass
|
||||
}
|
||||
|
||||
// SkippedPath is a would-be capture the tier filter selected but a structural guard refused. Reason
|
||||
// is operator-English; the engines log it (a skipped mandatory path = a silent capture gap otherwise).
|
||||
type SkippedPath struct {
|
||||
Root BindRoot
|
||||
RelPath string
|
||||
Class BindClass
|
||||
Reason string
|
||||
}
|
||||
|
||||
// CaptureSet is the result of ComputeCaptureSet. HasClassification mirrors the classifier's bool;
|
||||
// engines derive unit-only as (!HasClassification || len(Paths)==0). Paths is sorted by Abs.
|
||||
type CaptureSet struct {
|
||||
HasClassification bool
|
||||
Paths []CapturePath
|
||||
Skipped []SkippedPath
|
||||
}
|
||||
|
||||
// Structural-guard reasons (distinct strings; each names the rule it enforces).
|
||||
const (
|
||||
reasonEscape = "path escapes the drive root"
|
||||
reasonBareRoot = "bare drive-root bind would capture the backups tree"
|
||||
reasonReserved = "path inside the reserved backups zone"
|
||||
// reasonNoImportRoot: a ${IMPORT_PATH} bind with no resolvable system namespace root (R-75).
|
||||
reasonNoImportRoot = "canonical import root unresolvable (system_data_path unconfigured)"
|
||||
)
|
||||
|
||||
// ComputeCaptureSet resolves an app's classified binds into the tier-filtered absolute capture set.
|
||||
// Pipeline (fixed order, §8): legacy short-circuit → tier filter (§2 columns) → structural guards →
|
||||
// equal-Abs collapse (mandatory > optional) → containment dedup (keep ancestor) → sort by Abs.
|
||||
//
|
||||
// Tier columns (§2): TierOffsite carries mandatory only; TierSecondary carries mandatory + optional;
|
||||
// excluded is silently dropped at every tier (never in Paths, never in Skipped). A legacy app
|
||||
// (hasClassification=false) resolves NOTHING — {HasClassification:false} with nil Paths/Skipped —
|
||||
// so the engines' no-block branch stays byte-identical to today (the SQ5 cost-regression guard).
|
||||
//
|
||||
// Resolution: RootHDD → path.Join(hddPath, relPath); RootUserdata → path.Join(hddPath, "userdata",
|
||||
// relPath); RootImport → path.Join(importRoot, relPath) — the SYSTEM drive, never hddPath (R-75).
|
||||
// Guards run AFTER the tier filter, so Skipped means exactly "would have been captured by this tier,
|
||||
// refused for structural safety".
|
||||
func ComputeCaptureSet(binds []ClassifiedBind, hasClassification bool, tier CaptureTier, hddPath, importRoot string) CaptureSet {
|
||||
if !hasClassification {
|
||||
return CaptureSet{HasClassification: false}
|
||||
}
|
||||
cs := CaptureSet{HasClassification: true}
|
||||
|
||||
// Stages 1–3: tier filter → structural guards → equal-Abs collapse (shared with ComputeFabBuckets).
|
||||
uniq, skipped := resolveGuardCollapse(binds, hddPath, importRoot, func(c BindClass) bool { return tierKeeps(tier, c) })
|
||||
cs.Skipped = skipped
|
||||
|
||||
// Stage 4: containment dedup — drop any path whose ancestor is already present (keep the ancestor).
|
||||
for _, cp := range uniq {
|
||||
if hasStrictAncestor(cp.Abs, uniq) {
|
||||
continue
|
||||
}
|
||||
cs.Paths = append(cs.Paths, cp)
|
||||
}
|
||||
|
||||
// Stage 5: deterministic order.
|
||||
sort.Slice(cs.Paths, func(i, j int) bool { return cs.Paths[i].Abs < cs.Paths[j].Abs })
|
||||
sort.Slice(cs.Skipped, func(i, j int) bool {
|
||||
if cs.Skipped[i].Root != cs.Skipped[j].Root {
|
||||
return cs.Skipped[i].Root < cs.Skipped[j].Root
|
||||
}
|
||||
return cs.Skipped[i].RelPath < cs.Skipped[j].RelPath
|
||||
})
|
||||
return cs
|
||||
}
|
||||
|
||||
// resolveGuardCollapse is the pipeline shared by ComputeCaptureSet and ComputeFabBuckets: keep-filter
|
||||
// (the caller's predicate over class) → structural guards (Skipped) → resolve to Abs → equal-Abs
|
||||
// collapse (mandatory > optional > excluded; ties by smaller Root/RelPath). It does NOT apply
|
||||
// containment dedup — the caller decides (ComputeCaptureSet does; ComputeFabBuckets must not, so a
|
||||
// mandatory child inside an excluded parent stays independently addressable).
|
||||
func resolveGuardCollapse(binds []ClassifiedBind, hddPath, importRoot string, keep func(BindClass) bool) (uniq []CapturePath, skipped []SkippedPath) {
|
||||
var resolved []CapturePath
|
||||
for _, b := range binds {
|
||||
if !keep(b.Class) {
|
||||
continue
|
||||
}
|
||||
if reason, bad := structuralGuard(b.Root, b.RelPath, importRoot); bad {
|
||||
skipped = append(skipped, SkippedPath{Root: b.Root, RelPath: b.RelPath, Class: b.Class, Reason: reason})
|
||||
continue
|
||||
}
|
||||
resolved = append(resolved, CapturePath{
|
||||
Abs: resolveAbs(hddPath, importRoot, b.Root, b.RelPath), Root: b.Root, RelPath: b.RelPath, Class: b.Class,
|
||||
})
|
||||
}
|
||||
byAbs := make(map[string]CapturePath, len(resolved))
|
||||
for _, cp := range resolved {
|
||||
if cur, ok := byAbs[cp.Abs]; ok {
|
||||
byAbs[cp.Abs] = strongerCapture(cur, cp)
|
||||
continue
|
||||
}
|
||||
byAbs[cp.Abs] = cp
|
||||
}
|
||||
uniq = make([]CapturePath, 0, len(byAbs))
|
||||
for _, cp := range byAbs {
|
||||
uniq = append(uniq, cp)
|
||||
}
|
||||
return uniq, skipped
|
||||
}
|
||||
|
||||
// FabBuckets is the class-bucketed capture set for the manual `.fab` export (Task 4). Unlike
|
||||
// ComputeCaptureSet it keeps ALL classes (structural guards run over every class — a traversal path is
|
||||
// never plannable, opt-in or not) and does NOT collapse across containment (mandatory `media/books`
|
||||
// inside excluded `media` both survive, in their own buckets). HasClassification=false ⇒ empty (the
|
||||
// legacy full-root capture, unchanged).
|
||||
type FabBuckets struct {
|
||||
HasClassification bool
|
||||
Mandatory []CapturePath
|
||||
Optional []CapturePath
|
||||
Excluded []CapturePath
|
||||
Skipped []SkippedPath
|
||||
}
|
||||
|
||||
// ComputeFabBuckets resolves an app's classified binds into per-class buckets for the `.fab` export
|
||||
// selection UI + plan. Same resolution + structural guards + equal-Abs collapse as ComputeCaptureSet
|
||||
// (via resolveGuardCollapse), bucketed by class, no cross-bucket containment dedup. Each bucket is
|
||||
// Abs-sorted (deterministic).
|
||||
func ComputeFabBuckets(binds []ClassifiedBind, hasClassification bool, hddPath, importRoot string) FabBuckets {
|
||||
if !hasClassification {
|
||||
return FabBuckets{HasClassification: false}
|
||||
}
|
||||
fb := FabBuckets{HasClassification: true}
|
||||
uniq, skipped := resolveGuardCollapse(binds, hddPath, importRoot, func(BindClass) bool { return true })
|
||||
fb.Skipped = skipped
|
||||
for _, cp := range uniq {
|
||||
switch cp.Class {
|
||||
case ClassMandatory:
|
||||
fb.Mandatory = append(fb.Mandatory, cp)
|
||||
case ClassOptional:
|
||||
fb.Optional = append(fb.Optional, cp)
|
||||
default:
|
||||
fb.Excluded = append(fb.Excluded, cp)
|
||||
}
|
||||
}
|
||||
for _, b := range []*[]CapturePath{&fb.Mandatory, &fb.Optional, &fb.Excluded} {
|
||||
bk := *b
|
||||
sort.Slice(bk, func(i, j int) bool { return bk[i].Abs < bk[j].Abs })
|
||||
}
|
||||
sort.Slice(fb.Skipped, func(i, j int) bool {
|
||||
if fb.Skipped[i].Root != fb.Skipped[j].Root {
|
||||
return fb.Skipped[i].Root < fb.Skipped[j].Root
|
||||
}
|
||||
return fb.Skipped[i].RelPath < fb.Skipped[j].RelPath
|
||||
})
|
||||
return fb
|
||||
}
|
||||
|
||||
// tierKeeps applies the §2 tier column: mandatory everywhere, optional only for secondary, excluded
|
||||
// never.
|
||||
func tierKeeps(tier CaptureTier, class BindClass) bool {
|
||||
switch class {
|
||||
case ClassMandatory:
|
||||
return true
|
||||
case ClassOptional:
|
||||
return tier == TierSecondary
|
||||
default: // ClassExcluded (and any unexpected value) — never captured automatically
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// structuralGuard refuses a (root, relPath) that would capture an unsafe location. Evaluated after
|
||||
// the tier filter. RelPath arrives path.Clean'd from the compose parser but is NOT traversal-checked
|
||||
// there (ParseComposeClassifiableBinds path.Cleans; ValidateBackupSpec vets only SPEC entries), so an
|
||||
// unlisted writable "${HDD_PATH}/../x" bind reaches here classed mandatory — this guard is
|
||||
// load-bearing security, not defence-in-depth.
|
||||
func structuralGuard(root BindRoot, relPath, importRoot string) (reason string, bad bool) {
|
||||
if relPathEscapes(relPath) {
|
||||
return reasonEscape, true
|
||||
}
|
||||
// RootImport (R-75) resolves against the SYSTEM drive, not hddPath. If that root is unresolvable
|
||||
// (system_data_path unconfigured) the bind cannot be placed at all — refuse it LOUDLY into Skipped
|
||||
// rather than let resolveAbs join onto "" and produce a relative, wrong-drive path. The other two
|
||||
// roots cannot hit this: hddPath is checked by their own callers.
|
||||
if root == RootImport && importRoot == "" {
|
||||
return reasonNoImportRoot, true
|
||||
}
|
||||
// A bare ${IMPORT_PATH} bind is allowed: it resolves to <sysNS>/userdata/import, which nests no
|
||||
// backups/ tree (backups live at <sysNS>/backups, a sibling of userdata).
|
||||
if root == RootHDD {
|
||||
if relPath == "" {
|
||||
return reasonBareRoot, true // bare ${HDD_PATH} would nest <hddPath>/backups into the capture
|
||||
}
|
||||
if relPath == "backups" || strings.HasPrefix(relPath, "backups/") {
|
||||
return reasonReserved, true
|
||||
}
|
||||
}
|
||||
// RootUserdata + "" is allowed: resolves to <hddPath>/userdata, which does not nest backups/.
|
||||
return "", false
|
||||
}
|
||||
|
||||
// relPathEscapes reports whether relPath is absolute or contains a ".." path segment. Detection is
|
||||
// SEGMENT-WISE on the slash-split path (".." as a whole component), so a legit dir literally named
|
||||
// "a..b" passes.
|
||||
func relPathEscapes(relPath string) bool {
|
||||
if path.IsAbs(relPath) {
|
||||
return true
|
||||
}
|
||||
for _, seg := range strings.Split(relPath, "/") {
|
||||
if seg == ".." {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// resolveAbs maps a guarded (root, relPath) to its in-container absolute path via slash algebra.
|
||||
//
|
||||
// RootImport is the one root that does NOT resolve against hddPath: the canonical drop-zone lives on
|
||||
// the SYSTEM drive (R-75), so importRoot is supplied separately by the caller. Resolving it against
|
||||
// hddPath would silently name a directory on the WRONG DRIVE — a .fab opt-in would then capture (or
|
||||
// on restore, write) somewhere that merely looks plausible. An empty importRoot is the unresolvable
|
||||
// case and is refused upstream by structuralGuard, never silently joined.
|
||||
func resolveAbs(hddPath, importRoot string, root BindRoot, relPath string) string {
|
||||
switch root {
|
||||
case RootUserdata:
|
||||
return path.Join(hddPath, "userdata", relPath)
|
||||
case RootImport:
|
||||
return path.Join(importRoot, relPath)
|
||||
default:
|
||||
return path.Join(hddPath, relPath)
|
||||
}
|
||||
}
|
||||
|
||||
// strongerCapture picks the winner of an equal-Abs collision: mandatory beats optional; on equal
|
||||
// class strength, the lexicographically-smaller (Root, RelPath) wins (determinism).
|
||||
func strongerCapture(a, b CapturePath) CapturePath {
|
||||
sa, sb := classStrength(a.Class), classStrength(b.Class)
|
||||
if sa != sb {
|
||||
if sa > sb {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
if a.Root != b.Root {
|
||||
if a.Root < b.Root {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
if a.RelPath <= b.RelPath {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// classStrength ranks capture classes for the equal-Abs collapse (mandatory must never degrade).
|
||||
func classStrength(c BindClass) int {
|
||||
switch c {
|
||||
case ClassMandatory:
|
||||
return 2
|
||||
case ClassOptional:
|
||||
return 1
|
||||
default:
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
// hasStrictAncestor reports whether some OTHER path in set is a strict directory ancestor of abs
|
||||
// (abs == ancestor+"/"+…). Slash-aware prefix so "/x/paper" does not "contain" "/x/paperless".
|
||||
func hasStrictAncestor(abs string, set []CapturePath) bool {
|
||||
for _, o := range set {
|
||||
if o.Abs == abs {
|
||||
continue
|
||||
}
|
||||
if strings.HasPrefix(abs, o.Abs+"/") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Overlap is one absolute path claimed non-excluded by more than one app's capture set (§4.2). Apps
|
||||
// is sorted.
|
||||
type Overlap struct {
|
||||
Abs string
|
||||
Apps []string
|
||||
}
|
||||
|
||||
// CrossAppOverlaps reports absolute paths that appear in ≥2 apps' Paths — the catalog-convention
|
||||
// tripwire (§4: at most one app may class a host path non-excluded). Pure and advisory here; the
|
||||
// WARN wiring lands in 3a/3b, not in 3-core. Match is EXACT-Abs only: cross-app CONTAINMENT
|
||||
// (calibre-web's mandatory media/books sitting inside plex's excluded reader bind) is legitimate per
|
||||
// §4 and must NOT report. Deterministic despite the map input: app names are scanned in sorted order
|
||||
// and the output is sorted by Abs. Empty input / no overlap → empty (non-nil) slice.
|
||||
func CrossAppOverlaps(sets map[string]CaptureSet) []Overlap {
|
||||
apps := make([]string, 0, len(sets))
|
||||
for app := range sets {
|
||||
apps = append(apps, app)
|
||||
}
|
||||
sort.Strings(apps)
|
||||
|
||||
byAbs := make(map[string][]string)
|
||||
for _, app := range apps {
|
||||
seen := make(map[string]bool) // guard against an app listing the same Abs twice
|
||||
for _, cp := range sets[app].Paths {
|
||||
if seen[cp.Abs] {
|
||||
continue
|
||||
}
|
||||
seen[cp.Abs] = true
|
||||
byAbs[cp.Abs] = append(byAbs[cp.Abs], app)
|
||||
}
|
||||
}
|
||||
|
||||
out := make([]Overlap, 0)
|
||||
for abs, owners := range byAbs {
|
||||
if len(owners) < 2 {
|
||||
continue
|
||||
}
|
||||
sorted := append([]string(nil), owners...)
|
||||
sort.Strings(sorted)
|
||||
out = append(out, Overlap{Abs: abs, Apps: sorted})
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].Abs < out[j].Abs })
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,247 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"path"
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// absList extracts the sorted Abs slice from a CaptureSet's Paths (Paths is already Abs-sorted).
|
||||
func absList(cs CaptureSet) []string {
|
||||
out := make([]string, 0, len(cs.Paths))
|
||||
for _, p := range cs.Paths {
|
||||
out = append(out, p.Abs)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// classOfAbs finds the resolved class for an Abs in a CaptureSet (empty if absent).
|
||||
func classOfAbs(cs CaptureSet, abs string) BindClass {
|
||||
for _, p := range cs.Paths {
|
||||
if p.Abs == abs {
|
||||
return p.Class
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
const drv = "/mnt/drv"
|
||||
|
||||
func hdd(p string) string { return path.Join(drv, p) }
|
||||
func udat(p string) string { return path.Join(drv, "userdata", p) }
|
||||
|
||||
// --- Group A (Scenario A): classified per-tier split, immich shape ---
|
||||
|
||||
func TestComputeCaptureSet_PerTierSplit(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/immich"}, Class: ClassMandatory, Origin: OriginExplicit},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/photos", ReadOnly: true}, Class: ClassOptional, Origin: OriginExplicit},
|
||||
}
|
||||
|
||||
off := ComputeCaptureSet(binds, true, TierOffsite, drv, "")
|
||||
if got, want := absList(off), []string{hdd("appdata/immich")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("offsite Paths = %v, want %v (mandatory only — the :ro optional must NOT ship offsite)", got, want)
|
||||
}
|
||||
|
||||
sec := ComputeCaptureSet(binds, true, TierSecondary, drv, "")
|
||||
want := []string{hdd("appdata/immich"), udat("media/photos")}
|
||||
if got := absList(sec); !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("secondary Paths = %v, want %v (sorted)", got, want)
|
||||
}
|
||||
// each CapturePath carries its originating identity
|
||||
for _, p := range sec.Paths {
|
||||
switch p.Abs {
|
||||
case hdd("appdata/immich"):
|
||||
if p.Root != RootHDD || p.RelPath != "appdata/immich" || p.Class != ClassMandatory {
|
||||
t.Errorf("immich CapturePath identity = %+v", p)
|
||||
}
|
||||
case udat("media/photos"):
|
||||
if p.Root != RootUserdata || p.RelPath != "media/photos" || p.Class != ClassOptional {
|
||||
t.Errorf("photos CapturePath identity = %+v", p)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group B (Scenario B): legacy inertness — THE single most important test (SQ5 guard) ---
|
||||
|
||||
func TestComputeCaptureSet_LegacyInert(t *testing.T) {
|
||||
// A legacy app still has binds (the parser returns them), but with no class and origin=legacy.
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/tv"}, Origin: OriginLegacy},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/sonarr"}, Origin: OriginLegacy},
|
||||
}
|
||||
for _, tier := range []CaptureTier{TierOffsite, TierSecondary} {
|
||||
cs := ComputeCaptureSet(binds, false, tier, drv, "")
|
||||
if cs.HasClassification {
|
||||
t.Errorf("%s: HasClassification=true for a legacy app", tier)
|
||||
}
|
||||
if cs.Paths != nil {
|
||||
t.Errorf("%s: legacy app resolved Paths=%v — MUST be nil (unmigrated-sonarr-ships-its-TV-library regression)", tier, cs.Paths)
|
||||
}
|
||||
if cs.Skipped != nil {
|
||||
t.Errorf("%s: legacy app Skipped=%v — MUST be nil", tier, cs.Skipped)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group C (Scenario C): excluded is invisible — not in Paths, not in Skipped ---
|
||||
|
||||
func TestComputeCaptureSet_ExcludedInvisible(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/paperless/media"}, Class: ClassMandatory, Origin: OriginExplicit},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/paperless/export"}, Class: ClassExcluded, Origin: OriginExplicit},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "import/paperless"}, Class: ClassExcluded, Origin: OriginExplicit},
|
||||
}
|
||||
for _, tier := range []CaptureTier{TierOffsite, TierSecondary} {
|
||||
cs := ComputeCaptureSet(binds, true, tier, drv, "")
|
||||
if got, want := absList(cs), []string{hdd("appdata/paperless/media")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("%s Paths = %v, want %v (excluded filtered)", tier, got, want)
|
||||
}
|
||||
if len(cs.Skipped) != 0 {
|
||||
t.Errorf("%s: excluded binds must NOT appear in Skipped, got %v", tier, cs.Skipped)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group D (Scenario D): structural guards + allowed bare-userdata + legit a..b name ---
|
||||
|
||||
func TestComputeCaptureSet_StructuralGuards(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "../evil"}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d1 traversal
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: ""}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d2 bare hdd root
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "backups/primary/x"}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d3 reserved zone
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: ""}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d4 bare userdata — ALLOWED
|
||||
}
|
||||
cs := ComputeCaptureSet(binds, true, TierOffsite, drv, "")
|
||||
|
||||
// Paths: ONLY d4's userdata root — no escaped root, no backups/ anywhere.
|
||||
if got, want := absList(cs), []string{udat("")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Paths = %v, want %v (only the allowed bare-userdata)", got, want)
|
||||
}
|
||||
for _, p := range cs.Paths {
|
||||
if p.Abs == "/mnt/evil" {
|
||||
t.Fatal("escaped-root Abs present in Paths — traversal guard failed")
|
||||
}
|
||||
if containsSeg(p.Abs, "backups") {
|
||||
t.Fatalf("reserved backups/ path present in Paths: %s", p.Abs)
|
||||
}
|
||||
}
|
||||
|
||||
// Skipped: d1,d2,d3 each with a DISTINCT reason naming its rule.
|
||||
reasons := map[string]string{} // "<root>/<rel>" -> reason
|
||||
for _, s := range cs.Skipped {
|
||||
reasons[string(s.Root)+"/"+s.RelPath] = s.Reason
|
||||
}
|
||||
if len(cs.Skipped) != 3 {
|
||||
t.Fatalf("want 3 skipped, got %d: %+v", len(cs.Skipped), cs.Skipped)
|
||||
}
|
||||
if reasons["hdd/../evil"] != reasonEscape {
|
||||
t.Errorf("../evil reason = %q, want %q", reasons["hdd/../evil"], reasonEscape)
|
||||
}
|
||||
if reasons["hdd/"] != reasonBareRoot {
|
||||
t.Errorf("bare-hdd reason = %q, want %q", reasons["hdd/"], reasonBareRoot)
|
||||
}
|
||||
if reasons["hdd/backups/primary/x"] != reasonReserved {
|
||||
t.Errorf("backups reason = %q, want %q", reasons["hdd/backups/primary/x"], reasonReserved)
|
||||
}
|
||||
// distinctness
|
||||
if reasonEscape == reasonBareRoot || reasonBareRoot == reasonReserved || reasonEscape == reasonReserved {
|
||||
t.Error("guard reasons are not distinct")
|
||||
}
|
||||
}
|
||||
|
||||
// TestComputeCaptureSet_LegitDotDotName: a component literally named "a..b" is NOT traversal.
|
||||
func TestComputeCaptureSet_LegitDotDotName(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/a..b"}, Class: ClassMandatory, Origin: OriginExplicit},
|
||||
}
|
||||
cs := ComputeCaptureSet(binds, true, TierOffsite, drv, "")
|
||||
if got, want := absList(cs), []string{hdd("appdata/a..b")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Paths = %v, want %v (a..b is a legit name, not traversal)", got, want)
|
||||
}
|
||||
if len(cs.Skipped) != 0 {
|
||||
t.Errorf("a..b must not be skipped, got %v", cs.Skipped)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group E (Scenario E): containment dedup + equal-Abs mandatory-wins + determinism ---
|
||||
|
||||
func TestComputeCaptureSet_ContainmentAndCollision(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/paperless"}, Class: ClassMandatory, Origin: OriginExplicit},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/paperless/media"}, Class: ClassMandatory, Origin: OriginExplicit}, // descendant
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "userdata/media"}, Class: ClassOptional, Origin: OriginExplicit}, // Abs collides with next
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media"}, Class: ClassMandatory, Origin: OriginExplicit}, // same Abs, mandatory
|
||||
}
|
||||
cs := ComputeCaptureSet(binds, true, TierSecondary, drv, "")
|
||||
|
||||
want := []string{hdd("appdata/paperless"), udat("media")}
|
||||
if got := absList(cs); !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Paths = %v, want %v (descendant dropped; two spellings collapsed)", got, want)
|
||||
}
|
||||
// mandatory beats optional on the equal-Abs collision
|
||||
if c := classOfAbs(cs, udat("media")); c != ClassMandatory {
|
||||
t.Errorf("collapsed /userdata/media class = %q, want mandatory (mandatory must never degrade)", c)
|
||||
}
|
||||
|
||||
// determinism: recompute and compare full struct
|
||||
cs2 := ComputeCaptureSet(binds, true, TierSecondary, drv, "")
|
||||
if !reflect.DeepEqual(cs, cs2) {
|
||||
t.Error("ComputeCaptureSet is non-deterministic across runs")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group F (Scenario F): cross-app overlap advisory (pure) ---
|
||||
|
||||
func TestCrossAppOverlaps(t *testing.T) {
|
||||
X, Y, Z := "/mnt/drv/x", "/mnt/drv/y", "/mnt/drv/z"
|
||||
sets := map[string]CaptureSet{
|
||||
"appA": {HasClassification: true, Paths: []CapturePath{{Abs: X}, {Abs: Y}}},
|
||||
"appB": {HasClassification: true, Paths: []CapturePath{{Abs: Y}}},
|
||||
"appC": {HasClassification: true, Paths: []CapturePath{{Abs: Z}}},
|
||||
}
|
||||
got := CrossAppOverlaps(sets)
|
||||
want := []Overlap{{Abs: Y, Apps: []string{"appA", "appB"}}}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("CrossAppOverlaps = %+v, want %+v", got, want)
|
||||
}
|
||||
|
||||
// exact-match only: cross-app CONTAINMENT is legitimate, must NOT report.
|
||||
cont := map[string]CaptureSet{
|
||||
"plex": {Paths: []CapturePath{{Abs: "/mnt/drv/userdata/media"}}},
|
||||
"calibre-web": {Paths: []CapturePath{{Abs: "/mnt/drv/userdata/media/books"}}},
|
||||
}
|
||||
if got := CrossAppOverlaps(cont); len(got) != 0 {
|
||||
t.Errorf("containment across apps must NOT report an overlap, got %+v", got)
|
||||
}
|
||||
|
||||
// empty input → empty (non-nil) slice, not a flaky nil
|
||||
if got := CrossAppOverlaps(map[string]CaptureSet{}); got == nil || len(got) != 0 {
|
||||
t.Errorf("empty input → empty non-nil slice, got %#v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// containsSeg reports whether abs has seg as a path component (test helper).
|
||||
func containsSeg(abs, seg string) bool {
|
||||
for _, s := range splitSlash(abs) {
|
||||
if s == seg {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func splitSlash(s string) []string {
|
||||
var out []string
|
||||
cur := ""
|
||||
for _, r := range s {
|
||||
if r == '/' {
|
||||
out = append(out, cur)
|
||||
cur = ""
|
||||
continue
|
||||
}
|
||||
cur += string(r)
|
||||
}
|
||||
return append(out, cur)
|
||||
}
|
||||
@@ -0,0 +1,235 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"path"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Backup classification (referential coupling) — Task 2 of the backup-classification-redesign arc
|
||||
// (felhom.eu/documentation/audits/SPIKE-backup-classification-2026-07-14.md). This file is the SCHEMA
|
||||
// + PURE CLASSIFIER only; it is deliberately INERT — no backup tier consumes it yet. Task 3 (tier
|
||||
// policy engine) and Task 4 (manual .fab UI) are the consumers. Classes describe how a bind couples
|
||||
// to the app's referential state:
|
||||
//
|
||||
// - mandatory: COUPLED — restoring the app WITHOUT this bind yields a broken (not merely empty)
|
||||
// app, because the DB/state references the content (SQ3: immich DB-only restore = broken).
|
||||
// - optional: DECOUPLED-precious — absent ⇒ empty-not-broken, but the content is user-precious
|
||||
// (not re-downloadable): an external photo library, a curated comic/ROM set.
|
||||
// - excluded: DECOUPLED-bulk/transient — re-downloadable media, scraper caches, ingest inboxes,
|
||||
// transient export/download dirs; never shipped offsite, opt-in only for a manual .fab.
|
||||
|
||||
// BindClass is the referential-coupling class of a single host bind.
|
||||
type BindClass string
|
||||
|
||||
const (
|
||||
ClassMandatory BindClass = "mandatory" // COUPLED: restore-without is broken, not empty (SQ3)
|
||||
ClassOptional BindClass = "optional" // DECOUPLED-precious: empty-not-broken, not re-downloadable
|
||||
ClassExcluded BindClass = "excluded" // DECOUPLED-bulk/transient: never offsite, .fab opt-in
|
||||
)
|
||||
|
||||
// BindRoot names the deploy-time variable a bind's host path is relative to.
|
||||
type BindRoot string
|
||||
|
||||
const (
|
||||
RootUserdata BindRoot = "userdata" // relative to ${USERDATA_PATH}
|
||||
RootHDD BindRoot = "hdd" // relative to ${HDD_PATH}
|
||||
// RootImport is relative to ${IMPORT_PATH} — the CANONICAL drop-zone root (R-75). Unlike the
|
||||
// other two it does NOT resolve against the app's own drive: it lives on the system drive's
|
||||
// namespace, so every app's ingest folder is in one place. Resolvers therefore need the import
|
||||
// root passed in separately; they cannot derive it from hddPath.
|
||||
RootImport BindRoot = "import"
|
||||
)
|
||||
|
||||
// BackupSpec is the .felhom.yml `backup:` block. Paths are forward-slash, relative, path.Clean'd.
|
||||
type BackupSpec struct {
|
||||
Userdata []BindSpec `yaml:"userdata,omitempty" json:"userdata,omitempty"`
|
||||
HDD []BindSpec `yaml:"hdd,omitempty" json:"hdd,omitempty"`
|
||||
// Import classifies ${IMPORT_PATH}-relative binds (R-75). An app whose ingest bind moved from
|
||||
// ${USERDATA_PATH}/import/<app> to ${IMPORT_PATH}/<app> MUST move its backup entry here in the
|
||||
// same change: ValidateBackupSpec rejects an entry matching no compose bind, and the rejection is
|
||||
// WHOLE-BLOCK, so a stale `userdata: import/<app>` would discard the app's OTHER classifications
|
||||
// (e.g. an hdd appdata path classed mandatory) and silently degrade it to legacy.
|
||||
Import []BindSpec `yaml:"import,omitempty" json:"import,omitempty"`
|
||||
}
|
||||
|
||||
// BindSpec is one classified entry in a BackupSpec.
|
||||
type BindSpec struct {
|
||||
Path string `yaml:"path" json:"path"`
|
||||
Class BindClass `yaml:"class" json:"class"`
|
||||
}
|
||||
|
||||
// ComposeBind is a ${VAR}-relative host bind extracted from docker-compose.yml (Part 2 parser). It
|
||||
// lives in relative ${VAR} space (NOT resolved to an absolute path) and carries the :ro flag, both of
|
||||
// which the classifier needs — this is why the classifier does NOT reuse ParseComposeHDDMounts (which
|
||||
// resolves absolutes and drops the mode).
|
||||
type ComposeBind struct {
|
||||
Root BindRoot
|
||||
RelPath string // path.Clean'd, forward-slash, relative; "" for a bare-root bind (${VAR} itself)
|
||||
ReadOnly bool
|
||||
}
|
||||
|
||||
// ClassOrigin records HOW a bind's class was decided — for logs/UI and to prove the precedence rule.
|
||||
type ClassOrigin string
|
||||
|
||||
const (
|
||||
OriginExplicit ClassOrigin = "explicit" // matched an entry in the backup block
|
||||
OriginDefaultWritable ClassOrigin = "default_writable" // unlisted + writable → mandatory (capture)
|
||||
OriginDefaultRO ClassOrigin = "default_ro" // unlisted + :ro → excluded (reader rule)
|
||||
OriginLegacy ClassOrigin = "legacy" // no backup block at all → no class semantics
|
||||
)
|
||||
|
||||
// ClassifiedBind pairs a compose bind with its resolved class + origin.
|
||||
type ClassifiedBind struct {
|
||||
ComposeBind
|
||||
Class BindClass
|
||||
Origin ClassOrigin
|
||||
}
|
||||
|
||||
// validClass reports whether c is one of the three known classes (empty is INVALID — a typoed
|
||||
// `clas:` key makes yaml.v3 silently leave Class "", which must be rejected, not defaulted).
|
||||
func validClass(c BindClass) bool {
|
||||
switch c {
|
||||
case ClassMandatory, ClassOptional, ClassExcluded:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// ValidateRelPath is THE path-safety refusal set for every ${VAR}-relative catalog path — the
|
||||
// `backup:` block and `data_paths:` both run through it, so there is exactly ONE definition of what
|
||||
// a safe relative path is. Refuses: empty, backslash, absolute, non-path.Clean'd, and any leading
|
||||
// ".." escape. It deliberately does NOT check "matches a compose bind" — that rule needs the bind
|
||||
// list and differs per caller (whole-block reject for backup:, per-entry for data_paths:).
|
||||
func ValidateRelPath(root BindRoot, p string) error {
|
||||
where := fmt.Sprintf("%s[%q]", root, p)
|
||||
if p == "" {
|
||||
return fmt.Errorf("%s: empty path", where)
|
||||
}
|
||||
if strings.ContainsRune(p, '\\') {
|
||||
return fmt.Errorf("%s: backslash in path (paths are forward-slash relative)", where)
|
||||
}
|
||||
if path.IsAbs(p) {
|
||||
return fmt.Errorf("%s: absolute path (must be relative to the %s root)", where, root)
|
||||
}
|
||||
if p != path.Clean(p) {
|
||||
return fmt.Errorf("%s: non-clean path (want %q)", where, path.Clean(p))
|
||||
}
|
||||
// path.Clean has run — ".." can only survive as a leading "../" segment.
|
||||
if p == ".." || strings.HasPrefix(p, "../") {
|
||||
return fmt.Errorf("%s: path escapes the root (..)", where)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ValidRoot reports whether r is one of the three known bind roots.
|
||||
func ValidRoot(r BindRoot) bool {
|
||||
switch r {
|
||||
case RootUserdata, RootHDD, RootImport:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// ValidateBackupSpec checks a parsed backup block against the app's actual compose binds and returns
|
||||
// the FIRST defect (whole-block semantics — the caller rejects the ENTIRE block on any error, so the
|
||||
// app degrades to legacy rather than partially classifying). A nil spec is vacuously valid (legacy).
|
||||
//
|
||||
// Rejects: unknown/empty class; empty path; a path that is not already path.Clean'd, or is absolute,
|
||||
// or contains "..", or contains a backslash; a duplicate (root, path); an entry whose (root, path)
|
||||
// matches NO compose bind (a typo/stale entry must not silently shift the real bind onto the
|
||||
// mandatory default). Match is exact (Root, RelPath) equality.
|
||||
func ValidateBackupSpec(spec *BackupSpec, binds []ComposeBind) error {
|
||||
if spec == nil {
|
||||
return nil
|
||||
}
|
||||
present := make(map[BindRoot]map[string]bool)
|
||||
for _, b := range binds {
|
||||
if present[b.Root] == nil {
|
||||
present[b.Root] = make(map[string]bool)
|
||||
}
|
||||
present[b.Root][b.RelPath] = true
|
||||
}
|
||||
|
||||
seen := make(map[string]bool) // "<root>\x00<path>"
|
||||
check := func(root BindRoot, list []BindSpec) error {
|
||||
for _, e := range list {
|
||||
where := fmt.Sprintf("%s[%q]", root, e.Path)
|
||||
if !validClass(e.Class) {
|
||||
return fmt.Errorf("%s: invalid class %q (want mandatory|optional|excluded)", where, e.Class)
|
||||
}
|
||||
if err := ValidateRelPath(root, e.Path); err != nil {
|
||||
return err
|
||||
}
|
||||
key := string(root) + "\x00" + e.Path
|
||||
if seen[key] {
|
||||
return fmt.Errorf("%s: duplicate path in the backup block", where)
|
||||
}
|
||||
seen[key] = true
|
||||
if !present[root][e.Path] {
|
||||
return fmt.Errorf("%s: matches no compose bind (stale or typoed path)", where)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := check(RootUserdata, spec.Userdata); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := check(RootHDD, spec.HDD); err != nil {
|
||||
return err
|
||||
}
|
||||
return check(RootImport, spec.Import)
|
||||
}
|
||||
|
||||
// ClassifyBinds resolves every compose bind to a class + origin, applying the two-level default. The
|
||||
// second return reports whether the app carries a backup block at all.
|
||||
//
|
||||
// - spec == nil → every bind is emitted with Origin=legacy and an EMPTY Class (no class semantics),
|
||||
// and hasClassification=false. This is the block-ABSENT branch: nothing downstream may change
|
||||
// behavior for it (SQ5 two-level default — no block means today's per-tier legacy behavior).
|
||||
// - spec present → an explicit block entry ALWAYS wins, regardless of the bind's :ro flag (an
|
||||
// explicit `optional` on immich's :ro external library beats the reader default). An UNLISTED
|
||||
// bind defaults by mode: writable → mandatory (default_writable — the C6B-F1 direction: capture
|
||||
// rather than silently drop), read-only → excluded (default_ro — reader rule, SQ2).
|
||||
//
|
||||
// Pure. Assumes a validated spec (see ValidateBackupSpec) but never panics on an unvalidated one:
|
||||
// unmatched/invalid spec entries simply don't match any bind here.
|
||||
//
|
||||
// A bare-root bind (RelPath "") can never be matched by an explicit entry — an empty path is invalid
|
||||
// in the spec — so it always falls to the ro/writable default.
|
||||
func ClassifyBinds(spec *BackupSpec, binds []ComposeBind) (classified []ClassifiedBind, hasClassification bool) {
|
||||
out := make([]ClassifiedBind, 0, len(binds))
|
||||
if spec == nil {
|
||||
for _, b := range binds {
|
||||
out = append(out, ClassifiedBind{ComposeBind: b, Origin: OriginLegacy})
|
||||
}
|
||||
return out, false
|
||||
}
|
||||
explicit := make(map[BindRoot]map[string]BindClass)
|
||||
add := func(root BindRoot, list []BindSpec) {
|
||||
for _, e := range list {
|
||||
if explicit[root] == nil {
|
||||
explicit[root] = make(map[string]BindClass)
|
||||
}
|
||||
explicit[root][e.Path] = e.Class
|
||||
}
|
||||
}
|
||||
add(RootUserdata, spec.Userdata)
|
||||
add(RootHDD, spec.HDD)
|
||||
add(RootImport, spec.Import)
|
||||
|
||||
for _, b := range binds {
|
||||
cb := ClassifiedBind{ComposeBind: b}
|
||||
if cls, ok := explicit[b.Root][b.RelPath]; ok {
|
||||
cb.Class, cb.Origin = cls, OriginExplicit
|
||||
} else if b.ReadOnly {
|
||||
cb.Class, cb.Origin = ClassExcluded, OriginDefaultRO
|
||||
} else {
|
||||
cb.Class, cb.Origin = ClassMandatory, OriginDefaultWritable
|
||||
}
|
||||
out = append(out, cb)
|
||||
}
|
||||
return out, true
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// classOf finds the resolved class+origin for a (root, relpath) in a ClassifiedBind slice.
|
||||
func classOf(cbs []ClassifiedBind, root BindRoot, rel string) (BindClass, ClassOrigin, bool) {
|
||||
for _, c := range cbs {
|
||||
if c.Root == root && c.RelPath == rel {
|
||||
return c.Class, c.Origin, true
|
||||
}
|
||||
}
|
||||
return "", "", false
|
||||
}
|
||||
|
||||
// --- Group A: classifier ---
|
||||
|
||||
// TestClassify_ImmichShape is Scenario A: explicit classes resolve, and an EXPLICIT entry beats the
|
||||
// :ro reader-default (media/photos is :ro but ruled optional). Companion RP-2: making the ro-default
|
||||
// override explicit entries forces media/photos to excluded and fails the optional assertion.
|
||||
func TestClassify_ImmichShape(t *testing.T) {
|
||||
binds := []ComposeBind{
|
||||
{Root: RootHDD, RelPath: "appdata/immich", ReadOnly: false},
|
||||
{Root: RootUserdata, RelPath: "media/photos", ReadOnly: true}, // :ro external library
|
||||
}
|
||||
spec := &BackupSpec{
|
||||
HDD: []BindSpec{{Path: "appdata/immich", Class: ClassMandatory}},
|
||||
Userdata: []BindSpec{{Path: "media/photos", Class: ClassOptional}},
|
||||
}
|
||||
cbs, has := ClassifyBinds(spec, binds)
|
||||
if !has {
|
||||
t.Fatal("hasClassification should be true with a spec present")
|
||||
}
|
||||
if cls, org, ok := classOf(cbs, RootHDD, "appdata/immich"); !ok || cls != ClassMandatory || org != OriginExplicit {
|
||||
t.Errorf("appdata/immich = %v/%v, want mandatory/explicit", cls, org)
|
||||
}
|
||||
// The crux: an explicit optional beats the :ro default_ro that would otherwise force excluded.
|
||||
if cls, org, ok := classOf(cbs, RootUserdata, "media/photos"); !ok || cls != ClassOptional || org != OriginExplicit {
|
||||
t.Errorf("media/photos (:ro, explicit optional) = %v/%v, want optional/explicit (explicit beats ro-default)", cls, org)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClassify_TwoLevelDefault is Scenario B: with a block PRESENT, an unlisted writable bind
|
||||
// defaults mandatory (capture, the C6B-F1 direction) and an unlisted :ro bind defaults excluded
|
||||
// (reader rule). Companion RP-3: flipping the unlisted-writable default to excluded fails the
|
||||
// mandatory assertion.
|
||||
func TestClassify_TwoLevelDefault(t *testing.T) {
|
||||
binds := []ComposeBind{
|
||||
{Root: RootHDD, RelPath: "appdata/app", ReadOnly: false}, // listed
|
||||
{Root: RootUserdata, RelPath: "data/extra", ReadOnly: false}, // UNLISTED writable
|
||||
{Root: RootUserdata, RelPath: "media/ro", ReadOnly: true}, // UNLISTED :ro
|
||||
}
|
||||
spec := &BackupSpec{HDD: []BindSpec{{Path: "appdata/app", Class: ClassMandatory}}}
|
||||
cbs, has := ClassifyBinds(spec, binds)
|
||||
if !has {
|
||||
t.Fatal("hasClassification should be true")
|
||||
}
|
||||
if cls, org, _ := classOf(cbs, RootUserdata, "data/extra"); cls != ClassMandatory || org != OriginDefaultWritable {
|
||||
t.Errorf("unlisted writable = %v/%v, want mandatory/default_writable (capture direction)", cls, org)
|
||||
}
|
||||
if cls, org, _ := classOf(cbs, RootUserdata, "media/ro"); cls != ClassExcluded || org != OriginDefaultRO {
|
||||
t.Errorf("unlisted :ro = %v/%v, want excluded/default_ro (reader rule)", cls, org)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClassify_NilSpecLegacy is Scenario C: a nil spec → every bind is legacy with no class, and
|
||||
// hasClassification=false. This is the inertness gate at the classifier level.
|
||||
func TestClassify_NilSpecLegacy(t *testing.T) {
|
||||
binds := []ComposeBind{{Root: RootUserdata, RelPath: "media", ReadOnly: true}}
|
||||
cbs, has := ClassifyBinds(nil, binds)
|
||||
if has {
|
||||
t.Error("nil spec must report hasClassification=false")
|
||||
}
|
||||
if len(cbs) != 1 || cbs[0].Origin != OriginLegacy || cbs[0].Class != "" {
|
||||
t.Errorf("nil-spec bind = %+v, want origin=legacy, empty class", cbs[0])
|
||||
}
|
||||
}
|
||||
|
||||
// TestClassify_BareRootFallsToDefault: a bare-root bind (RelPath "") can't be matched by any explicit
|
||||
// entry (empty paths are invalid), so it falls to the ro/writable default.
|
||||
func TestClassify_BareRootFallsToDefault(t *testing.T) {
|
||||
spec := &BackupSpec{Userdata: []BindSpec{{Path: "media/x", Class: ClassOptional}}}
|
||||
cbs, _ := ClassifyBinds(spec, []ComposeBind{
|
||||
{Root: RootUserdata, RelPath: "", ReadOnly: false}, // bare ${USERDATA_PATH}
|
||||
{Root: RootHDD, RelPath: "", ReadOnly: true}, // bare ${HDD_PATH} :ro
|
||||
})
|
||||
if cls, org, _ := classOf(cbs, RootUserdata, ""); cls != ClassMandatory || org != OriginDefaultWritable {
|
||||
t.Errorf("bare writable root = %v/%v, want mandatory/default_writable", cls, org)
|
||||
}
|
||||
if cls, org, _ := classOf(cbs, RootHDD, ""); cls != ClassExcluded || org != OriginDefaultRO {
|
||||
t.Errorf("bare :ro root = %v/%v, want excluded/default_ro", cls, org)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClassify_SameRelPathBothRoots: userdata/x and hdd/x are DISTINCT binds — Root is part of
|
||||
// identity, so an explicit hdd entry must not classify the userdata bind.
|
||||
func TestClassify_SameRelPathBothRoots(t *testing.T) {
|
||||
binds := []ComposeBind{
|
||||
{Root: RootUserdata, RelPath: "shared", ReadOnly: false},
|
||||
{Root: RootHDD, RelPath: "shared", ReadOnly: false},
|
||||
}
|
||||
spec := &BackupSpec{HDD: []BindSpec{{Path: "shared", Class: ClassExcluded}}}
|
||||
cbs, _ := ClassifyBinds(spec, binds)
|
||||
if cls, org, _ := classOf(cbs, RootHDD, "shared"); cls != ClassExcluded || org != OriginExplicit {
|
||||
t.Errorf("hdd/shared = %v/%v, want excluded/explicit", cls, org)
|
||||
}
|
||||
if cls, org, _ := classOf(cbs, RootUserdata, "shared"); cls != ClassMandatory || org != OriginDefaultWritable {
|
||||
t.Errorf("userdata/shared = %v/%v, want mandatory/default_writable (hdd entry must NOT match it)", cls, org)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group B: validation (Scenario D) — every defect rejects the WHOLE block; error names the entry ---
|
||||
|
||||
func TestValidateBackupSpec_Defects(t *testing.T) {
|
||||
// The compose binds the valid entries reference (so only the seeded defect is the failure).
|
||||
binds := []ComposeBind{
|
||||
{Root: RootUserdata, RelPath: "media/tv"},
|
||||
{Root: RootHDD, RelPath: "appdata/x"},
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
spec *BackupSpec
|
||||
wantFrag string // substring the error must contain (the offending entry / rule)
|
||||
}{
|
||||
{"unknown class", &BackupSpec{Userdata: []BindSpec{{Path: "media/tv", Class: "keepit"}}}, "invalid class"},
|
||||
{"empty class (typoed key)", &BackupSpec{Userdata: []BindSpec{{Path: "media/tv", Class: ""}}}, "invalid class"},
|
||||
{"empty path", &BackupSpec{HDD: []BindSpec{{Path: "", Class: ClassMandatory}}}, "empty path"},
|
||||
{"absolute path", &BackupSpec{HDD: []BindSpec{{Path: "/etc/x", Class: ClassMandatory}}}, "absolute"},
|
||||
{"dotdot path", &BackupSpec{HDD: []BindSpec{{Path: "../escape", Class: ClassMandatory}}}, "escapes"},
|
||||
{"backslash path", &BackupSpec{HDD: []BindSpec{{Path: "appdata\\x", Class: ClassMandatory}}}, "backslash"},
|
||||
{"non-clean path", &BackupSpec{HDD: []BindSpec{{Path: "appdata/./x", Class: ClassMandatory}}}, "non-clean"},
|
||||
{"duplicate path", &BackupSpec{HDD: []BindSpec{
|
||||
{Path: "appdata/x", Class: ClassMandatory}, {Path: "appdata/x", Class: ClassExcluded},
|
||||
}}, "duplicate"},
|
||||
{"no matching bind (typo)", &BackupSpec{Userdata: []BindSpec{{Path: "media/tvv", Class: ClassExcluded}}}, "matches no compose bind"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
err := ValidateBackupSpec(tc.spec, binds)
|
||||
if err == nil {
|
||||
t.Fatalf("expected rejection, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), tc.wantFrag) {
|
||||
t.Errorf("error %q must contain %q", err.Error(), tc.wantFrag)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestValidateBackupSpec_ValidAndNil: a clean block validates, and a nil spec is vacuously valid.
|
||||
func TestValidateBackupSpec_ValidAndNil(t *testing.T) {
|
||||
binds := []ComposeBind{{Root: RootHDD, RelPath: "appdata/x"}, {Root: RootUserdata, RelPath: "media/tv"}}
|
||||
spec := &BackupSpec{
|
||||
HDD: []BindSpec{{Path: "appdata/x", Class: ClassMandatory}},
|
||||
Userdata: []BindSpec{{Path: "media/tv", Class: ClassExcluded}},
|
||||
}
|
||||
if err := ValidateBackupSpec(spec, binds); err != nil {
|
||||
t.Errorf("clean block should validate: %v", err)
|
||||
}
|
||||
if err := ValidateBackupSpec(nil, binds); err != nil {
|
||||
t.Errorf("nil spec must be vacuously valid: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -51,6 +51,20 @@ type DumpValidation struct {
|
||||
Error string
|
||||
FileSize int64
|
||||
ModTime time.Time
|
||||
// R-44 (v0.148.0) content sniff — a WARN-LEVEL signal, never a gate.
|
||||
//
|
||||
// Structural validity says nothing about whether a dump holds the customer's data. The immich
|
||||
// dump of 2026-07-19 was 52MB, had a valid header and 60+ CREATE TABLEs, and contained zero
|
||||
// users and zero assets: its whole bulk was the geodata reference tables immich ships. Size and
|
||||
// table count are therefore both useless as emptiness heuristics — but an accounts table with
|
||||
// no rows is a strong, cheap, app-agnostic hint that a dump predates the customer entirely.
|
||||
//
|
||||
// Deliberately NOT a refusal: plenty of legitimate apps have no users table (UserTableFound
|
||||
// false → inconclusive → silent), and a false positive that blocked a restore would be far
|
||||
// worse than the skew it guards against. The restore confirm shows it as one extra line.
|
||||
UserTableFound bool
|
||||
UserRows int
|
||||
LooksEmpty bool // UserTableFound && UserRows == 0
|
||||
}
|
||||
|
||||
// DumpFileInfo holds info about a dump file on disk.
|
||||
@@ -101,12 +115,10 @@ func DiscoverDatabases(ctx context.Context, logger *log.Logger, debug bool, know
|
||||
|
||||
id, name, image := parts[0], parts[1], strings.ToLower(parts[2])
|
||||
|
||||
var dbType DBType
|
||||
if strings.Contains(image, "postgres") {
|
||||
dbType = DBTypePostgres
|
||||
} else if strings.Contains(image, "mariadb") || strings.Contains(image, "mysql") {
|
||||
dbType = DBTypeMariaDB
|
||||
} else {
|
||||
// R-47: the same predicate that DBServiceNames applies to compose `image:` values, so a dump
|
||||
// that exists is always attributable to a startable service (see dbservices.go).
|
||||
dbType, isDB := dbTypeForImage(image)
|
||||
if !isDB {
|
||||
if debug {
|
||||
logger.Printf("[DEBUG] DiscoverDatabases: skipping container %s (image=%s, not a database)", name, image)
|
||||
}
|
||||
@@ -363,6 +375,9 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
lineNum := 0
|
||||
headerFound := false
|
||||
tableCount := 0
|
||||
// R-44 sniff state. inUserCopy tracks a postgres `COPY … FROM stdin;` block for an accounts
|
||||
// table; rows are counted until the `\.` terminator.
|
||||
inUserCopy := false
|
||||
for {
|
||||
lineBytes, isPrefix, err := reader.ReadLine()
|
||||
if err != nil {
|
||||
@@ -374,7 +389,12 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
break // EOF
|
||||
}
|
||||
if isPrefix {
|
||||
// Line exceeds buffer — skip remainder (COPY data, large INSERTs)
|
||||
// Line exceeds buffer — skip remainder (COPY data, large INSERTs).
|
||||
// A long line inside a user COPY block is still a ROW: count it before discarding it,
|
||||
// or a table whose rows happen to be wide would sniff as empty and raise a false alarm.
|
||||
if inUserCopy {
|
||||
v.UserRows++
|
||||
}
|
||||
for isPrefix && err == nil {
|
||||
_, isPrefix, err = reader.ReadLine()
|
||||
}
|
||||
@@ -384,6 +404,23 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
line := string(lineBytes)
|
||||
lineNum++
|
||||
|
||||
// R-44 content sniff (warn-level; see DumpValidation).
|
||||
if inUserCopy {
|
||||
if line == `\.` {
|
||||
inUserCopy = false
|
||||
} else {
|
||||
v.UserRows++
|
||||
}
|
||||
} else if isUserCopyStart(line, dbType) {
|
||||
inUserCopy = true
|
||||
v.UserTableFound = true
|
||||
} else if dbType == DBTypeMariaDB && isUserInsert(line) {
|
||||
// mysqldump writes multi-row `INSERT INTO \`users\` VALUES (…),(…);` — the row count is
|
||||
// not worth parsing out of it, and presence alone answers the only question asked here.
|
||||
v.UserTableFound = true
|
||||
v.UserRows++
|
||||
}
|
||||
|
||||
// Header check — scan first 10 lines for expected dump header
|
||||
// MariaDB 11.4+ prepends a sandbox comment before the header line
|
||||
if lineNum <= 10 && !headerFound {
|
||||
@@ -427,10 +464,65 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
return v
|
||||
}
|
||||
|
||||
v.LooksEmpty = v.UserTableFound && v.UserRows == 0
|
||||
if v.LooksEmpty {
|
||||
log.Printf("[WARN] [backup] ValidateDump: %s is structurally valid (%d tables) but its accounts table has NO rows — the dump may predate the customer's data", filePath, tableCount)
|
||||
}
|
||||
|
||||
v.Valid = true
|
||||
return v
|
||||
}
|
||||
|
||||
// userTableNames are the table names treated as "the accounts table" by the R-44 sniff. Kept
|
||||
// deliberately short: a wider net (anything containing "user") would match join/audit tables like
|
||||
// `user_metadata` or `album_user`, which are legitimately empty on a healthy single-user install
|
||||
// and would produce exactly the false alarm this signal must not raise.
|
||||
var userTableNames = []string{"user", "users", "account", "accounts"}
|
||||
|
||||
// isUserCopyStart reports whether a line opens a postgres `COPY <accounts-table> … FROM stdin;`
|
||||
// block. pg_dump writes the table qualified and optionally quoted — `COPY public."user" (…)`,
|
||||
// `COPY public.users (…)` — so both forms are matched.
|
||||
func isUserCopyStart(line string, dbType DBType) bool {
|
||||
if dbType != DBTypePostgres || !strings.HasPrefix(line, "COPY ") {
|
||||
return false
|
||||
}
|
||||
rest := strings.TrimPrefix(line, "COPY ")
|
||||
sp := strings.IndexByte(rest, ' ')
|
||||
if sp < 0 {
|
||||
return false
|
||||
}
|
||||
return matchesUserTable(rest[:sp])
|
||||
}
|
||||
|
||||
// isUserInsert reports whether a line is a mysqldump INSERT into an accounts table.
|
||||
func isUserInsert(line string) bool {
|
||||
const pfx = "INSERT INTO "
|
||||
if !strings.HasPrefix(line, pfx) {
|
||||
return false
|
||||
}
|
||||
rest := strings.TrimPrefix(line, pfx)
|
||||
sp := strings.IndexByte(rest, ' ')
|
||||
if sp < 0 {
|
||||
return false
|
||||
}
|
||||
return matchesUserTable(rest[:sp])
|
||||
}
|
||||
|
||||
// matchesUserTable strips schema qualification and quoting from a dumped table reference and
|
||||
// reports whether the bare name is an accounts table.
|
||||
func matchesUserTable(ref string) bool {
|
||||
if dot := strings.LastIndexByte(ref, '.'); dot >= 0 {
|
||||
ref = ref[dot+1:]
|
||||
}
|
||||
ref = strings.Trim(ref, "\"`")
|
||||
for _, n := range userTableNames {
|
||||
if strings.EqualFold(ref, n) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ListDumpFiles returns info about SQL dump files on disk.
|
||||
//
|
||||
// M18: ValidateDump scans the dump line-by-line; on a customer with hundreds-of-MB dumps that is wasted
|
||||
@@ -672,7 +764,8 @@ func getMariaDBPassword(ctx context.Context, containerID string) string {
|
||||
// - else known[containerName] → containerName (the container name IS the stack — don't strip, e.g. my-cache).
|
||||
// - else longest known prefix → handles <stack>_postgres / <stack>-1 / compose-suffixed names.
|
||||
// - else → candidate (fall back to today's suffix-strip; preserves behaviour when
|
||||
// the stack list is empty/unavailable, so nothing regresses).
|
||||
// the stack list is empty/unavailable, so nothing regresses).
|
||||
//
|
||||
// A nil/empty `known` map = the legacy fast path (pure suffix-strip).
|
||||
func deriveStackName(containerName string, known map[string]bool) string {
|
||||
candidate := suffixStripStackName(containerName)
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-44 content-sniff tests.
|
||||
//
|
||||
// The dump that triggered this work (DIAG-immich-restore-2026-07-19) was 52MB, had a valid
|
||||
// PostgreSQL header and 60+ CREATE TABLE statements, and contained zero users and zero assets —
|
||||
// its entire bulk was immich's shipped geodata reference tables. Both of the signals the product
|
||||
// already had (file size, table count) called it healthy. These tests pin the one signal that
|
||||
// would have caught it, and the boundaries that keep it from crying wolf.
|
||||
|
||||
func writeDump(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "d.sql")
|
||||
if err := os.WriteFile(p, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
const pgHead = `-- PostgreSQL database dump
|
||||
-- Dumped from database version 16.10
|
||||
SET statement_timeout = 0;
|
||||
SET client_encoding = 'UTF8';
|
||||
CREATE TABLE public.asset (id uuid NOT NULL);
|
||||
CREATE TABLE public."user" (id uuid NOT NULL, email text);
|
||||
`
|
||||
|
||||
// TestSniffFlagsEmptyAccountsTable is the 2026-07-19 shape: structurally perfect, no customer.
|
||||
func TestSniffFlagsEmptyAccountsTable(t *testing.T) {
|
||||
body := pgHead + "COPY public.\"user\" (id, email) FROM stdin;\n\\.\n" +
|
||||
"COPY public.asset (id) FROM stdin;\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if !v.Valid {
|
||||
t.Fatalf("the dump is structurally valid; sniff must not change that: %s", v.Error)
|
||||
}
|
||||
if !v.UserTableFound {
|
||||
t.Fatal("the accounts table COPY block was not recognised")
|
||||
}
|
||||
if v.UserRows != 0 {
|
||||
t.Fatalf("UserRows = %d, want 0", v.UserRows)
|
||||
}
|
||||
if !v.LooksEmpty {
|
||||
t.Fatal("a valid dump with zero account rows MUST raise the warn signal — this is the whole point of R-44")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffQuietOnPopulatedDump — the common case must stay silent, or the warning becomes noise
|
||||
// and gets ignored precisely when it matters.
|
||||
func TestSniffQuietOnPopulatedDump(t *testing.T) {
|
||||
body := pgHead + "COPY public.\"user\" (id, email) FROM stdin;\n" +
|
||||
"a\tone@example.invalid\nb\ttwo@example.invalid\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserRows != 2 {
|
||||
t.Fatalf("UserRows = %d, want 2", v.UserRows)
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("a dump with account rows must not be flagged")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffInconclusiveWithoutAccountsTable — plenty of legitimate apps have no users table. No
|
||||
// table, no claim: a false positive here would warn on every restore of such an app forever.
|
||||
func TestSniffInconclusiveWithoutAccountsTable(t *testing.T) {
|
||||
body := "-- PostgreSQL database dump\nCREATE TABLE public.thing (id int);\n" +
|
||||
"COPY public.thing (id) FROM stdin;\n\\.\n" + strings.Repeat("-- pad\n", 20)
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserTableFound {
|
||||
t.Fatal("no accounts table exists — none must be reported")
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("an app without an accounts table must be INCONCLUSIVE, never flagged empty")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffIgnoresJoinAndAuditTables is the false-alarm guard that shaped the name list, and it is
|
||||
// written as the case that DISCRIMINATES: an app with NO accounts table but with `user_metadata` /
|
||||
// `album_user` / `user_audit` — all legitimately empty on a healthy box. Exact-matching leaves this
|
||||
// inconclusive (silent, correct). A substring match on "user" would treat a join table as the
|
||||
// accounts table, find zero rows, and shout "your backup looks empty" on every single restore of a
|
||||
// perfectly healthy app — which is how a warning signal becomes noise and then gets ignored.
|
||||
func TestSniffIgnoresJoinAndAuditTables(t *testing.T) {
|
||||
body := "-- PostgreSQL database dump\nCREATE TABLE public.album (id int);\n" +
|
||||
"COPY public.user_metadata (id) FROM stdin;\n\\.\n" +
|
||||
"COPY public.album_user (id) FROM stdin;\n\\.\n" +
|
||||
"COPY public.user_audit (id) FROM stdin;\n\\.\n" +
|
||||
"COPY public.album (id) FROM stdin;\n1\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserTableFound {
|
||||
t.Fatal("a join/audit table must never be mistaken for the accounts table")
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("empty join/audit tables must not trigger the warning — this app has no accounts table at all")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffCountsOnlyTheAccountsTable pins the counting boundary separately: with a real accounts
|
||||
// table present, rows from neighbouring user-ish tables must not inflate it.
|
||||
func TestSniffCountsOnlyTheAccountsTable(t *testing.T) {
|
||||
body := pgHead +
|
||||
"COPY public.user_metadata (id) FROM stdin;\nm1\nm2\nm3\n\\.\n" +
|
||||
"COPY public.\"user\" (id, email) FROM stdin;\na\tone@example.invalid\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserRows != 1 {
|
||||
t.Fatalf("only the real accounts table may be counted; UserRows = %d, want 1", v.UserRows)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffCountsWideRows — a row wider than the read buffer is skipped by the structural scan, but
|
||||
// it is still a row. Counting it wrong would flag a populated table as empty (immich asset rows are
|
||||
// genuinely long, which is what makes this reachable).
|
||||
func TestSniffCountsWideRows(t *testing.T) {
|
||||
wide := strings.Repeat("x", 300*1024)
|
||||
body := pgHead + "COPY public.\"user\" (id, email) FROM stdin;\n" + wide + "\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserRows != 1 {
|
||||
t.Fatalf("a buffer-exceeding row must still count; UserRows = %d, want 1", v.UserRows)
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("a table whose single row is very wide must not sniff as empty")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffMariaDBInsertForm — mysqldump writes multi-row INSERTs, not COPY blocks.
|
||||
func TestSniffMariaDBInsertForm(t *testing.T) {
|
||||
head := "-- MariaDB dump 10.19\nCREATE TABLE `users` (id int);\n" + strings.Repeat("-- pad\n", 20)
|
||||
empty := ValidateDump(writeDump(t, head), DBTypeMariaDB)
|
||||
if empty.UserTableFound {
|
||||
t.Fatal("a CREATE TABLE alone is not an accounts-table row source")
|
||||
}
|
||||
full := ValidateDump(writeDump(t, head+"INSERT INTO `users` VALUES (1),(2);\n"), DBTypeMariaDB)
|
||||
if !full.UserTableFound || full.LooksEmpty {
|
||||
t.Fatalf("a populated mariadb dump must not be flagged: %+v", full)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// R-47 — naming the database SERVICE, not just the running container.
|
||||
//
|
||||
// A dump replay must never race the application's own schema management. Proven live on 2026-07-19
|
||||
// (DIAG-immich-restore-round2-2026-07-19, H4): the reconstitution started the whole stack before
|
||||
// replaying, immich-server rebuilt `clip_index` two seconds before the dump's own CREATE INDEX, and
|
||||
// the replay aborted `already exists` under ON_ERROR_STOP=1 — leaving a half-applied schema that the
|
||||
// app itself then reported as drift. The fix is to bring up ONLY the database service(s) for the
|
||||
// replay, which requires knowing their compose SERVICE names (docker `up -d <svc>` takes service
|
||||
// names, not container names).
|
||||
//
|
||||
// The symmetry that makes this safe: a `.sql` dump can only exist because DiscoverDatabases matched
|
||||
// the running container's image string, and the compose `image:` value IS that image string. So the
|
||||
// same predicate — dbTypeForImage — decides both "is there a dump" and "which service holds it".
|
||||
|
||||
// dbTypeForImage maps a container/compose image reference to the database engine the backup code
|
||||
// supports, or ok=false for anything else (redis/valkey/app images — never started in the DB-only
|
||||
// phase). Extracted from DiscoverDatabases so the discovery heuristic and the compose heuristic can
|
||||
// never drift apart; behaviour is byte-equivalent to the inline form it replaced.
|
||||
func dbTypeForImage(image string) (DBType, bool) {
|
||||
img := strings.ToLower(image)
|
||||
switch {
|
||||
case strings.Contains(img, "postgres"):
|
||||
return DBTypePostgres, true
|
||||
case strings.Contains(img, "mariadb"), strings.Contains(img, "mysql"):
|
||||
return DBTypeMariaDB, true
|
||||
}
|
||||
return DBType(""), false
|
||||
}
|
||||
|
||||
// composeServicesDoc is the minimal view of a compose file needed here: the `services:` MAP and each
|
||||
// service's `image:`. Deliberately a real YAML parse and not a line scan — a top-level `volumes:`
|
||||
// block (immich's `immich_ml_cache:`) has exactly the shape a naive scan misreads as a service, and
|
||||
// starting a phantom service, or missing the real one, both land in the wrong branch.
|
||||
type composeServicesDoc struct {
|
||||
Services map[string]struct {
|
||||
Image string `yaml:"image"`
|
||||
} `yaml:"services"`
|
||||
}
|
||||
|
||||
// DBServiceNames returns the sorted compose SERVICE names in composePath whose `image:` identifies a
|
||||
// supported database engine — the exact argument list for `docker compose up -d <svc>...`.
|
||||
//
|
||||
// A file with no (or an empty) `services:` key returns (nil, nil): an app with no identifiable DB
|
||||
// service is a legitimate, common case and the caller decides what it means. An unreadable or
|
||||
// unparseable file returns an error, because "cannot tell" must never silently read as "no database"
|
||||
// — the callers turn that into a refusal when a dump exists.
|
||||
//
|
||||
// Image values are matched literally. Catalog templates pin their images literally (enforced since
|
||||
// Campaign 7), so an interpolated `${...}` image simply does not match and lands in the caller's
|
||||
// fail-closed branch by design, rather than being guessed at.
|
||||
func DBServiceNames(composePath string) ([]string, error) {
|
||||
data, err := os.ReadFile(composePath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("reading compose file: %w", err)
|
||||
}
|
||||
var doc composeServicesDoc
|
||||
if err := yaml.Unmarshal(data, &doc); err != nil {
|
||||
return nil, fmt.Errorf("parsing compose file %s: %w", composePath, err)
|
||||
}
|
||||
var names []string
|
||||
for name, svc := range doc.Services {
|
||||
if _, ok := dbTypeForImage(svc.Image); ok {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names, nil
|
||||
}
|
||||
@@ -0,0 +1,195 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-47 (v0.153.0) — the DB-service resolver.
|
||||
//
|
||||
// These exist because a dump replay that starts the WHOLE stack races the application's own schema
|
||||
// management: proven live on 2026-07-19 (DIAG-immich-restore-round2-2026-07-19, H4) when
|
||||
// immich-server rebuilt `clip_index` two seconds before the dump's CREATE INDEX and the replay
|
||||
// aborted `already exists`. Closing that window means bringing up ONLY the database service, which
|
||||
// means naming it correctly — every case below is a way of naming it wrongly.
|
||||
|
||||
// writeCompose drops a compose file in a temp dir and returns its path.
|
||||
func writeCompose(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "docker-compose.yml")
|
||||
if err := os.WriteFile(p, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// TestDBTypeForImage pins the shared heuristic. It is the SAME predicate DiscoverDatabases applies to
|
||||
// a running container's image, which is what makes "a dump exists ⇒ a service can be named" hold:
|
||||
// the compose `image:` value IS the container's image string. The table reproduces the inline form
|
||||
// this function replaced, byte for byte, including the redis/valkey negatives that must never be
|
||||
// started in the DB-only window.
|
||||
func TestDBTypeForImage(t *testing.T) {
|
||||
cases := []struct {
|
||||
image string
|
||||
want DBType
|
||||
ok bool
|
||||
}{
|
||||
{"docker.io/library/postgres:16-alpine", DBTypePostgres, true},
|
||||
// immich's real pin — a vector-extended postgres whose REPO segment carries the substring.
|
||||
{"ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0", DBTypePostgres, true},
|
||||
{"postgres", DBTypePostgres, true},
|
||||
{"POSTGRES:16", DBTypePostgres, true}, // the discovery path lowercases; so does this
|
||||
{"mariadb:11", DBTypeMariaDB, true},
|
||||
{"mysql:8.4", DBTypeMariaDB, true},
|
||||
{"docker.io/library/MySQL:8", DBTypeMariaDB, true},
|
||||
{"redis:7-alpine", "", false},
|
||||
{"valkey/valkey:8", "", false},
|
||||
{"ghcr.io/immich-app/immich-server:v1.119.0", "", false},
|
||||
{"", "", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, ok := dbTypeForImage(c.image)
|
||||
if ok != c.ok || (ok && got != c.want) {
|
||||
t.Errorf("dbTypeForImage(%q) = (%q, %v), want (%q, %v)", c.image, got, ok, c.want, c.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDBServiceNames(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
body string
|
||||
want []string
|
||||
}{
|
||||
{
|
||||
name: "postgres service is named",
|
||||
body: "services:\n app:\n image: ghcr.io/x/app:1\n database:\n image: postgres:16\n",
|
||||
want: []string{"database"},
|
||||
},
|
||||
{
|
||||
name: "mariadb service is named",
|
||||
body: "services:\n db:\n image: mariadb:11\n web:\n image: nextcloud:30\n",
|
||||
want: []string{"db"},
|
||||
},
|
||||
{
|
||||
name: "mysql service is named",
|
||||
body: "services:\n mysql:\n image: mysql:8.4\n",
|
||||
want: []string{"mysql"},
|
||||
},
|
||||
{
|
||||
name: "redis-only app has no database service",
|
||||
body: "services:\n app:\n image: ghcr.io/x/app:1\n redis:\n image: redis:7-alpine\n",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "multiple databases are returned SORTED (one up -d carries them all)",
|
||||
body: "services:\n zdb:\n image: postgres:16\n adb:\n image: mariadb:11\n app:\n image: x:1\n",
|
||||
want: []string{"adb", "zdb"},
|
||||
},
|
||||
{
|
||||
name: "no services key at all",
|
||||
body: "volumes:\n data:\n",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "empty services map",
|
||||
body: "services:\n",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "an interpolated image is not guessed at",
|
||||
body: "services:\n db:\n image: ${DB_IMAGE}\n",
|
||||
want: nil,
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got, err := DBServiceNames(writeCompose(t, c.body))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(got, c.want) {
|
||||
t.Errorf("DBServiceNames = %v, want %v", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBServiceNames_TopLevelKeysAreNotServices is the decoy test, and the reason this is a YAML
|
||||
// parse rather than a line scan. immich's real compose carries a top-level `volumes:` block whose
|
||||
// entry (`immich_ml_cache:`) sits at exactly the indentation a service name does, and a top-level
|
||||
// `networks:` block does the same. A scanner that collected "indented keys followed by image-ish
|
||||
// lines" would either invent a service that `docker compose up -d` cannot start, or — worse — match
|
||||
// the wrong one and leave the real database down while the app came up around the replay.
|
||||
func TestDBServiceNames_TopLevelKeysAreNotServices(t *testing.T) {
|
||||
// The service/volume/network names and the image pins are the catalog's real immich template.
|
||||
// `immich_postgres_data` is the trap made concrete: a top-level VOLUME key whose name contains
|
||||
// "postgres" and which no `up -d` could ever start.
|
||||
body := `services:
|
||||
immich-server:
|
||||
image: ghcr.io/immich-app/immich-server:v3.0.3
|
||||
immich-machine-learning:
|
||||
image: ghcr.io/immich-app/immich-machine-learning:v3.0.3
|
||||
immich-postgres:
|
||||
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0
|
||||
immich-redis:
|
||||
image: redis:7-alpine
|
||||
volumes:
|
||||
immich_ml_cache:
|
||||
immich_postgres_data:
|
||||
immich_redis_data:
|
||||
networks:
|
||||
traefik-public:
|
||||
external: true
|
||||
immich-internal:
|
||||
`
|
||||
got, err := DBServiceNames(writeCompose(t, body))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(got, []string{"immich-postgres"}) {
|
||||
t.Fatalf("DBServiceNames = %v, want [immich-postgres] — a top-level volume/network key was mistaken for a service", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBServiceNames_UnreadableAndUnparseableError proves the fail-closed direction: "cannot tell"
|
||||
// must surface as an ERROR, never as the empty (= "this app has no database") answer. The callers
|
||||
// turn an empty result into a refusal only when a dump exists; if a read failure silently produced
|
||||
// the same empty slice for an app with no dump, a genuinely broken compose would flow on unnoticed.
|
||||
func TestDBServiceNames_UnreadableAndUnparseableError(t *testing.T) {
|
||||
if _, err := DBServiceNames(filepath.Join(t.TempDir(), "nope.yml")); err == nil {
|
||||
t.Fatal("a missing compose file must be an error, not an empty service list")
|
||||
}
|
||||
// Valid YAML scalar where a map is required, plus outright broken YAML.
|
||||
if _, err := DBServiceNames(writeCompose(t, "services: [1, 2, 3\n broken")); err == nil {
|
||||
t.Fatal("an unparseable compose file must be an error, not an empty service list")
|
||||
}
|
||||
}
|
||||
|
||||
// TestDiscoverAndComposeAgreeOnTheSameImages is the SYMMETRY guard: whatever image string makes
|
||||
// DiscoverDatabases produce a dump must also make DBServiceNames name a service. They now share one
|
||||
// predicate; this asserts the property that sharing is FOR, so a future edit to either side that
|
||||
// breaks it fails here rather than in a customer's restore.
|
||||
func TestDiscoverAndComposeAgreeOnTheSameImages(t *testing.T) {
|
||||
images := []string{"postgres:16", "mariadb:11", "mysql:8.4", "redis:7", "ghcr.io/x/app:1"}
|
||||
var body strings.Builder
|
||||
body.WriteString("services:\n")
|
||||
var wantDB []string
|
||||
for i, img := range images {
|
||||
svc := string(rune('a' + i))
|
||||
body.WriteString(" " + svc + ":\n image: " + img + "\n")
|
||||
if _, ok := dbTypeForImage(img); ok {
|
||||
wantDB = append(wantDB, svc)
|
||||
}
|
||||
}
|
||||
got, err := DBServiceNames(writeCompose(t, body.String()))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(got, wantDB) {
|
||||
t.Fatalf("compose resolver named %v but the discovery predicate says %v — the two sides have drifted", got, wantDB)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func bucketAbs(b []CapturePath) []string {
|
||||
out := make([]string, 0, len(b))
|
||||
for _, p := range b {
|
||||
out = append(out, p.Abs)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// classified app → three buckets, resolved + Abs-sorted.
|
||||
func TestComputeFabBuckets_Classified(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/books"}, Class: ClassMandatory},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/comics"}, Class: ClassOptional},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/movies"}, Class: ClassExcluded},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/app"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
if !fb.HasClassification {
|
||||
t.Fatal("HasClassification must be true")
|
||||
}
|
||||
if got, want := bucketAbs(fb.Mandatory), []string{hdd("appdata/app"), udat("media/books")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Mandatory = %v, want %v", got, want)
|
||||
}
|
||||
if got, want := bucketAbs(fb.Optional), []string{udat("media/comics")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Optional = %v, want %v", got, want)
|
||||
}
|
||||
if got, want := bucketAbs(fb.Excluded), []string{udat("media/movies")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Excluded = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// legacy (no block) → empty buckets (the full-root capture stays out of the classified plan).
|
||||
func TestComputeFabBuckets_LegacyEmpty(t *testing.T) {
|
||||
binds := []ClassifiedBind{{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/tv"}, Origin: OriginLegacy}}
|
||||
fb := ComputeFabBuckets(binds, false, drv, "")
|
||||
if fb.HasClassification || fb.Mandatory != nil || fb.Optional != nil || fb.Excluded != nil {
|
||||
t.Errorf("legacy app must yield empty buckets, got %+v", fb)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario E: structural guards run over ALL classes — a traversal path in an EXCLUDED bind is Skipped,
|
||||
// never plannable (opt-in or not).
|
||||
func TestComputeFabBuckets_GuardsAllClasses(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "../evil"}, Class: ClassExcluded},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/ok"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
for _, b := range [][]CapturePath{fb.Mandatory, fb.Optional, fb.Excluded} {
|
||||
for _, p := range b {
|
||||
if p.RelPath == "../evil" {
|
||||
t.Fatal("a traversal path must never enter a bucket (guards run over all classes)")
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(fb.Skipped) != 1 || fb.Skipped[0].RelPath != "../evil" {
|
||||
t.Errorf("traversal excluded path must be Skipped, got %+v", fb.Skipped)
|
||||
}
|
||||
}
|
||||
|
||||
// no cross-bucket containment dedup: a mandatory CHILD inside an excluded PARENT both survive.
|
||||
func TestComputeFabBuckets_NoCrossBucketContainment(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media"}, Class: ClassExcluded},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/books"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
if got, want := bucketAbs(fb.Mandatory), []string{udat("media/books")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("mandatory child must survive independently: Mandatory = %v, want %v", got, want)
|
||||
}
|
||||
if got, want := bucketAbs(fb.Excluded), []string{udat("media")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("excluded parent must survive: Excluded = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// equal-Abs collapse: two spellings of one path collapse, mandatory wins (into the mandatory bucket).
|
||||
func TestComputeFabBuckets_EqualAbsMandatoryWins(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "userdata/media"}, Class: ClassOptional},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
if got, want := bucketAbs(fb.Mandatory), []string{udat("media")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("collapsed path must land in Mandatory, got Mandatory=%v", got)
|
||||
}
|
||||
if len(fb.Optional) != 0 {
|
||||
t.Errorf("optional spelling must collapse away, got %v", bucketAbs(fb.Optional))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
package appbackup
|
||||
|
||||
import "testing"
|
||||
|
||||
// R-203 — the ONE drive-kind rule. Table-driven over BOTH drive kinds on purpose: this defect
|
||||
// survived because it is invisible on the kind that already worked, so a test that only covers the
|
||||
// enrolled drive proves nothing about the fix.
|
||||
|
||||
func TestNamespaceRootFor_BothDriveKinds(t *testing.T) {
|
||||
const sys = "/mnt/sys_drive"
|
||||
cases := []struct {
|
||||
name, drive, want string
|
||||
}{
|
||||
// Scenario B — the enrolled drive must be BYTE-IDENTICAL to pre-R-203 behaviour. The
|
||||
// in-guest mount already IS the namespace root; appending felhom-data here would recreate
|
||||
// the .../felhom-data/felhom-data/... double-nest NamespaceRoot's comment exists to prevent.
|
||||
{"enrolled usb", "/mnt/felhom-usb", "/mnt/felhom-usb"},
|
||||
{"enrolled hdd", "/mnt/felhom-drives/hdd_1", "/mnt/felhom-drives/hdd_1"},
|
||||
{"enrolled nvme", "/mnt/felhom-drives/nvme-1tb", "/mnt/felhom-drives/nvme-1tb"},
|
||||
// Scenario A — the system-data fallback gains the segment. This is the case that was wrong.
|
||||
{"system drive", "/mnt/sys_drive", "/mnt/sys_drive/felhom-data"},
|
||||
// A trailing slash is the same drive. Before R-203 the backup package's copy of this rule
|
||||
// compared WITHOUT Clean while the stacks package's copy compared WITH it — so a config value
|
||||
// with a trailing slash would have flipped the mode in one package and not the other.
|
||||
{"system drive, trailing slash", "/mnt/sys_drive/", "/mnt/sys_drive/felhom-data"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := NamespaceRootFor(tc.drive, sys); got != tc.want {
|
||||
t.Fatalf("NamespaceRootFor(%q, %q) = %q, want %q", tc.drive, sys, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The rule must survive a trailing slash on the SYSTEM path too — it comes from config.
|
||||
func TestIsEnrolledDrive_CleansBothSides(t *testing.T) {
|
||||
if IsEnrolledDrive("/mnt/sys_drive", "/mnt/sys_drive/") {
|
||||
t.Error("a trailing slash on the system path must not make the system drive look enrolled")
|
||||
}
|
||||
if IsEnrolledDrive("/mnt/sys_drive/", "/mnt/sys_drive") {
|
||||
t.Error("a trailing slash on the drive path must not make the system drive look enrolled")
|
||||
}
|
||||
if !IsEnrolledDrive("/mnt/felhom-usb", "/mnt/sys_drive") {
|
||||
t.Error("an enrolled drive must report enrolled")
|
||||
}
|
||||
}
|
||||
|
||||
// The consequence the whole item is about: the directory an app binds and the directory the capture
|
||||
// set looks in must be the SAME on both drive kinds.
|
||||
//
|
||||
// RED-PROOF: replace `UserdataDir(NamespaceRootFor(drive, sys))` with `UserdataDir(drive)` — the
|
||||
// pre-R-203 call — and the system-drive row FAILS with the two paths differing by exactly
|
||||
// `/felhom-data`. That is production behaviour up to v0.196.0.
|
||||
func TestAppBindAndCaptureRootAgree(t *testing.T) {
|
||||
const sys = "/mnt/sys_drive"
|
||||
for _, drive := range []string{"/mnt/felhom-usb", "/mnt/felhom-drives/hdd_1", "/mnt/sys_drive"} {
|
||||
nsRoot := NamespaceRootFor(drive, sys)
|
||||
appBind := UserdataDir(nsRoot) // what the deploy sets as ${USERDATA_PATH}
|
||||
captureRoot := UserdataDir(nsRoot) // what the capture set resolves RootUserdata against
|
||||
if appBind != captureRoot {
|
||||
t.Fatalf("drive %q: the app binds %q while the backup captures %q", drive, appBind, captureRoot)
|
||||
}
|
||||
// And it must be the canonical location — the one EnsureUserdataSkeleton creates.
|
||||
if drive == sys && appBind != "/mnt/sys_drive/felhom-data/userdata" {
|
||||
t.Fatalf("system drive resolved to %q, want the canonical /mnt/sys_drive/felhom-data/userdata", appBind)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -5,7 +5,11 @@
|
||||
// cross-drive, or drive-mount code in the backup package.
|
||||
package appbackup
|
||||
|
||||
import "path/filepath"
|
||||
import (
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// FelhomDataDir is the namespace directory on storage drives for all felhom-managed data.
|
||||
const FelhomDataDir = "felhom-data"
|
||||
@@ -28,6 +32,35 @@ func NamespaceRoot(drivePath string, inGuestDrive bool) string {
|
||||
return filepath.Join(drivePath, FelhomDataDir)
|
||||
}
|
||||
|
||||
// IsEnrolledDrive reports whether a drive path is an ENROLLED user-data drive (Model A: its in-guest
|
||||
// mount already IS the namespace root) rather than the system-data fallback. It is the ONE comparison
|
||||
// that decides which NamespaceRoot mode applies, and it lives here so no package re-derives it.
|
||||
//
|
||||
// Both sides are Clean'd: `/mnt/sys_drive/` and `/mnt/sys_drive` are the same drive, and a trailing
|
||||
// slash arriving from config must not silently flip the mode.
|
||||
func IsEnrolledDrive(drivePath, systemDataPath string) bool {
|
||||
return filepath.Clean(drivePath) != filepath.Clean(systemDataPath)
|
||||
}
|
||||
|
||||
// NamespaceRootFor is the resolver every caller should use when it holds a bare DRIVE path and the
|
||||
// system-data path — i.e. everywhere outside the backup package, which already had this rule.
|
||||
//
|
||||
// R-203: FIVE call sites passed a bare drive path straight to UserdataDir (and its siblings), which
|
||||
// take a NAMESPACE ROOT. On an enrolled drive the two coincide, so nothing showed; on the system-data
|
||||
// fallback they differ by exactly the felhom-data segment, and the app then bound a directory the
|
||||
// backup never looked at. The run still reported ok. Measured live on demo-hp 2026-08-04:
|
||||
// the app wrote to /mnt/sys_drive/userdata/media/books while the off-site capture set looked for
|
||||
// /mnt/sys_drive/felhom-data/userdata/media/books.
|
||||
//
|
||||
// THE CONTRACT, restated because four callers got it wrong and a fifth will: UserdataDir,
|
||||
// PrimaryBackupPath, RecoveryUnitPath and AppDataDir all take a NAMESPACE ROOT. If you are holding
|
||||
// something that came out of HDD_PATH or a StoragePath, it is a DRIVE path — put it through here
|
||||
// first. `UserdataDir(bareDrivePath)` still compiles and is still wrong; TestNoBareDrivePathToUserdataDir
|
||||
// is the guard that keeps the count from growing.
|
||||
func NamespaceRootFor(drivePath, systemDataPath string) string {
|
||||
return NamespaceRoot(drivePath, IsEnrolledDrive(drivePath, systemDataPath))
|
||||
}
|
||||
|
||||
// PrimaryBackupPath returns the root primary backup directory under a felhom-data namespace root.
|
||||
func PrimaryBackupPath(nsRoot string) string {
|
||||
return filepath.Join(nsRoot, "backups", "primary")
|
||||
@@ -36,9 +69,10 @@ func PrimaryBackupPath(nsRoot string) string {
|
||||
// RecoveryUnitPath returns the per-app self-contained recovery-unit ROOT under a namespace root.
|
||||
// It is the existing per-app backup dir (`backups/primary/<stack>/`) — the legacy name is kept so the
|
||||
// db-dumps/ and volume-dumps/ already written there need no migration; the unit gains compose/ and
|
||||
// manifest.json as siblings, making the whole dir a complete, recreatable unit (Phase 2). The unit is
|
||||
// secret-free: secrets/data-keys are recovered from the guest's own app.yaml (live or via PBS), never
|
||||
// stored here. See backup.recoveryUnit / restore for the capture + restore flow.
|
||||
// manifest.json as siblings, making the whole dir a complete, recreatable unit (Phase 2). Since D5 the
|
||||
// unit's compose/app.yaml CARRIES the portable secret class (data keys, DB passwords, internal signing
|
||||
// secrets) at mode 0600, so a Tier-1/2 restore needs the drive and nothing else; internet-reachable
|
||||
// admin logins are still withheld. See backup.recoveryUnit / restore for the capture + restore flow.
|
||||
func RecoveryUnitPath(nsRoot, stackName string) string {
|
||||
return filepath.Join(nsRoot, "backups", "primary", stackName)
|
||||
}
|
||||
@@ -64,7 +98,61 @@ func AppVolumeDumpPath(nsRoot, stackName string) string {
|
||||
return filepath.Join(RecoveryUnitPath(nsRoot, stackName), "volume-dumps")
|
||||
}
|
||||
|
||||
// AppDataDir returns the app data directory under a felhom-data namespace root.
|
||||
// AppDataDir returns the app data directory under a felhom-data namespace root. The final segment
|
||||
// is the app's real appdata dir NAME — usually the stack name, but NOT always: paperless-ngx writes
|
||||
// appdata/paperless (F-S2/F-S3). Callers that key by stack name silently miss such apps; use
|
||||
// AppDataDirNames to resolve the real name(s) from the app's compose binds and pass them here.
|
||||
func AppDataDir(nsRoot, stackName string) string {
|
||||
return filepath.Join(nsRoot, "appdata", stackName)
|
||||
}
|
||||
|
||||
// AppDataDirNames returns the app's real directory name(s) under <hddPath>/appdata, derived from its
|
||||
// compose HDD bind mounts (F-S2/F-S3: the dir name is NOT always the stack name — paperless-ngx
|
||||
// writes appdata/paperless). hddMounts are resolved host paths in the ParseComposeHDDMounts shape
|
||||
// (each is <hddPath> itself or a subpath, filepath.Clean'd). The first path element under
|
||||
// <hddPath>/appdata/ is taken as the dir name; results are deduped and sorted. Falls back to
|
||||
// []string{stackName} when no appdata-prefixed mount is derivable (no HDD appdata binds, unreadable
|
||||
// compose, nil provider) — the exact legacy behavior.
|
||||
//
|
||||
// Today every catalog app resolves to exactly ONE name (immich→immich, nextcloud→nextcloud,
|
||||
// romm→romm, paperless-ngx→paperless). The N>1 return is defensive: tier-2 refuses it loudly,
|
||||
// migrate handles it naturally.
|
||||
func AppDataDirNames(hddPath, stackName string, hddMounts []string) []string {
|
||||
prefix := filepath.Clean(hddPath) + string(filepath.Separator) + "appdata" + string(filepath.Separator)
|
||||
seen := make(map[string]bool)
|
||||
var names []string
|
||||
for _, mnt := range hddMounts {
|
||||
cm := filepath.Clean(mnt)
|
||||
if !strings.HasPrefix(cm, prefix) {
|
||||
continue // not under appdata/ (a whole-root bind, a different subtree, a foreign drive)
|
||||
}
|
||||
rem := strings.TrimPrefix(cm, prefix)
|
||||
first := strings.Split(rem, string(filepath.Separator))[0]
|
||||
if first == "" {
|
||||
continue
|
||||
}
|
||||
if !seen[first] {
|
||||
seen[first] = true
|
||||
names = append(names, first)
|
||||
}
|
||||
}
|
||||
if len(names) == 0 {
|
||||
return []string{stackName}
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names
|
||||
}
|
||||
|
||||
// AppDataBindsPresent reports whether any of the app's resolved HDD mounts sits under
|
||||
// <hddPath>/appdata/ — i.e. the compose actually DECLARES an appdata bind. Callers use it to
|
||||
// distinguish "no appdata to back up" (silent skip is correct) from "declared appdata dir missing
|
||||
// on disk" (the silence that hid F-S2 — worth a WARN). Same prefix rule as AppDataDirNames.
|
||||
func AppDataBindsPresent(hddPath string, hddMounts []string) bool {
|
||||
prefix := filepath.Clean(hddPath) + string(filepath.Separator) + "appdata" + string(filepath.Separator)
|
||||
for _, mnt := range hddMounts {
|
||||
if strings.HasPrefix(filepath.Clean(mnt), prefix) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-75 Scenario C — DETERMINISM. This is the P6 gate.
|
||||
//
|
||||
// The spike measured the naive map-order derivation producing 20 DISTINCT outputs from 20 identical
|
||||
// runs. fbNeedsRecreate force-recreates FileBrowser on ANY byte difference in the generated config,
|
||||
// and SyncFileBrowserMounts has ~14 call sites — so a non-deterministic skeleton is a fleet-wide
|
||||
// FileBrowser restart loop, the v0.151-class bug. 20 identical generations or this fails.
|
||||
func TestScenarioC_SkeletonDeterminism(t *testing.T) {
|
||||
// Deliberately UNSORTED input, with duplicates and a deep path, so the function has real work to
|
||||
// normalise. A sort applied only to the input would not save a map-ordered implementation.
|
||||
derived := []string{
|
||||
"media/podcasts", "roms", "media/books", "downloads", "media",
|
||||
"media/photos", "media/books", "a/b/c/d",
|
||||
}
|
||||
const n = 20
|
||||
first := BuildUserdataSkeleton(derived)
|
||||
for i := 1; i < n; i++ {
|
||||
got := BuildUserdataSkeleton(derived)
|
||||
if !slices.Equal(got, first) {
|
||||
t.Fatalf("generation %d/%d differs — a non-deterministic skeleton force-recreates FileBrowser on every sync pass\n first: %v\n got: %v",
|
||||
i+1, n, first, got)
|
||||
}
|
||||
}
|
||||
if !slices.IsSorted(first) {
|
||||
t.Errorf("skeleton must be sorted, got %v", first)
|
||||
}
|
||||
// Ancestor expansion: a deep derived path implies its whole chain.
|
||||
for _, want := range []string{"a", "a/b", "a/b/c", "a/b/c/d"} {
|
||||
if !slices.Contains(first, want) {
|
||||
t.Errorf("ancestor chain incomplete: %q missing from %v", want, first)
|
||||
}
|
||||
}
|
||||
// Dedup: "media/books" appeared twice in the input and "media" both derived and as an ancestor.
|
||||
for _, d := range []string{"media", "media/books"} {
|
||||
if c := countOf(first, d); c != 1 {
|
||||
t.Errorf("%q appears %d times, want exactly 1", d, c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func countOf(xs []string, want string) int {
|
||||
n := 0
|
||||
for _, x := range xs {
|
||||
if x == want {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// R-75 Scenario D — ZERO REMOVALS, proven by construction.
|
||||
//
|
||||
// The derived set drops `documents` (implied by no catalog app) and, after the R-75 move, the two
|
||||
// import/* entries. The carry-list is what keeps them. This asserts the merged set is a strict
|
||||
// SUPERSET of the historical hardcoded skeleton for any derived input — including the empty one, the
|
||||
// fresh-box case where the catalog has not synced yet.
|
||||
func TestScenarioD_SkeletonNeverDropsACarriedDir(t *testing.T) {
|
||||
for _, derived := range [][]string{
|
||||
nil, // fresh box, catalog not yet synced
|
||||
{"media/podcasts"}, // the one genuinely new entry
|
||||
{"roms", "downloads", "media/photos"}, // a partial catalog
|
||||
} {
|
||||
got := BuildUserdataSkeleton(derived)
|
||||
for _, carried := range UserdataSkeletonCarry() {
|
||||
if !slices.Contains(got, carried) {
|
||||
t.Errorf("derived=%v: carried dir %q was DROPPED — zero-removals violated", derived, carried)
|
||||
}
|
||||
}
|
||||
}
|
||||
// And the new entry really is added when the catalog implies it.
|
||||
if !slices.Contains(BuildUserdataSkeleton([]string{"media/podcasts"}), "media/podcasts") {
|
||||
t.Error("media/podcasts must be added when the catalog implies it")
|
||||
}
|
||||
// `documents` is the specific entry the spike flagged: in the carry-list, in no catalog app.
|
||||
if !slices.Contains(BuildUserdataSkeleton([]string{"media/podcasts"}), "documents") {
|
||||
t.Error("`documents` must survive — it exists on both demo boxes and may hold customer files")
|
||||
}
|
||||
}
|
||||
|
||||
// A traversal or absolute entry reaching the skeleton would make EnsureUserdataSkeleton create a
|
||||
// directory outside the userdata root. The derived set comes from a compose parser, so this is a
|
||||
// guard on untrusted-ish catalog input, not defence in depth.
|
||||
func TestSkeletonRefusesEscapes(t *testing.T) {
|
||||
got := BuildUserdataSkeleton([]string{"../escape", "..", "", "/abs/path", "ok/dir"})
|
||||
for _, bad := range []string{"../escape", "..", "", "/abs/path"} {
|
||||
if slices.Contains(got, bad) {
|
||||
t.Errorf("escape entry %q must not reach the skeleton: %v", bad, got)
|
||||
}
|
||||
}
|
||||
for _, d := range got {
|
||||
if filepath.IsAbs(d) || d == ".." || len(d) > 3 && d[:3] == "../" {
|
||||
t.Errorf("unsafe skeleton entry %q", d)
|
||||
}
|
||||
}
|
||||
if !slices.Contains(got, "ok/dir") {
|
||||
t.Error("a legitimate entry alongside bad ones must still be kept")
|
||||
}
|
||||
// "/abs/path" is not dropped outright — it is normalised to a relative path and kept, which is
|
||||
// safe (it lands under the userdata root). Pin that so the behaviour is a decision, not a guess.
|
||||
if !slices.Contains(got, "abs/path") {
|
||||
t.Errorf("an absolute entry should be normalised to relative, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// EnsureUserdataSkeleton creates every dir it is given and NOTHING ELSE, and never removes.
|
||||
func TestEnsureUserdataSkeletonCreatesOnly(t *testing.T) {
|
||||
ns := t.TempDir()
|
||||
// A pre-existing customer dir that no catalog app implies and the carry-list does not contain.
|
||||
stray := filepath.Join(UserdataDir(ns), "sajat-mappa")
|
||||
if err := os.MkdirAll(stray, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
dirs := BuildUserdataSkeleton([]string{"media/podcasts"})
|
||||
if err := EnsureUserdataSkeleton(ns, dirs); err != nil {
|
||||
// chown to gid 1000 fails for a non-root test user; the dirs are still created.
|
||||
t.Logf("EnsureUserdataSkeleton returned %v (expected when not running as root)", err)
|
||||
}
|
||||
for _, d := range dirs {
|
||||
if fi, err := os.Stat(filepath.Join(UserdataDir(ns), d)); err != nil || !fi.IsDir() {
|
||||
t.Errorf("skeleton dir %q not created: %v", d, err)
|
||||
}
|
||||
}
|
||||
if _, err := os.Stat(stray); err != nil {
|
||||
t.Errorf("a pre-existing customer dir was removed — zero-removals violated: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// R-75: a DATA drive must never get a per-drive drop-zone from the skeleton. Carrying the old
|
||||
// `import/*` entries would have the skeleton re-create a dead lookalike on every drive forever —
|
||||
// one that is also never backed up, since import paths are class: excluded.
|
||||
//
|
||||
// This is NOT a zero-removals violation: nothing deletes the dirs a box already has (see
|
||||
// TestEnsureUserdataSkeletonCreatesOnly). They stop being maintained and stop appearing on fresh boxes.
|
||||
func TestSkeletonNeverCreatesAPerDriveDropZone(t *testing.T) {
|
||||
// The catalog no longer implies any ${USERDATA_PATH}/import path — the binds moved to
|
||||
// ${IMPORT_PATH} — so the only way one could appear is via the carry-list.
|
||||
for _, derived := range [][]string{nil, {"media/podcasts", "roms"}} {
|
||||
for _, d := range BuildUserdataSkeleton(derived) {
|
||||
if d == "import" || strings.HasPrefix(d, "import/") {
|
||||
t.Errorf("derived=%v: skeleton created a per-drive drop-zone %q — the canonical root is on the SYSTEM drive", derived, d)
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, c := range UserdataSkeletonCarry() {
|
||||
if c == "import" || strings.HasPrefix(c, "import/") {
|
||||
t.Errorf("the carry-list still holds %q", c)
|
||||
}
|
||||
}
|
||||
// A catalog app that genuinely declares a ${USERDATA_PATH}/import/... bind would still be
|
||||
// honoured — the rule is "don't carry them", not "filter them out".
|
||||
if !slices.Contains(BuildUserdataSkeleton([]string{"import/valami"}), "import/valami") {
|
||||
t.Error("a genuinely derived userdata import path must still be created")
|
||||
}
|
||||
}
|
||||
@@ -2,7 +2,10 @@ package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Customer-facing userdata layout + the shared-storage ownership convention (v0.66.0).
|
||||
@@ -28,19 +31,91 @@ func UserdataDir(nsRoot string) string {
|
||||
return filepath.Join(nsRoot, "userdata")
|
||||
}
|
||||
|
||||
// UserdataSkeleton is the standard subtree created on every storage path (relative to UserdataDir).
|
||||
// ImportDirName is the single import (drop-zone) subtree name under a userdata root.
|
||||
const ImportDirName = "import"
|
||||
|
||||
// ImportDir returns the CANONICAL drop-zone root under a namespace root (R-75).
|
||||
//
|
||||
// Unlike every other userdata dir, this one is drive-INDEPENDENT: the caller resolves it against the
|
||||
// SYSTEM drive's namespace root, never against the app's own HDD_PATH, so a multi-drive box has
|
||||
// exactly ONE import tree. That is the whole point. Each drop-zone app has exactly one ingest bind,
|
||||
// so a per-drive import/ would put a folder that LOOKS like a drop-zone on every drive while only
|
||||
// one of them does anything — and because import paths are `class: excluded`, files stranded in a
|
||||
// dead one are never backed up either.
|
||||
//
|
||||
// It deliberately stays INSIDE the userdata tree, so the 2775/setgid/GID-1000 convention, the
|
||||
// FileBrowser mount and the ownership rules all apply to it unchanged.
|
||||
func ImportDir(nsRoot string) string {
|
||||
return filepath.Join(UserdataDir(nsRoot), ImportDirName)
|
||||
}
|
||||
|
||||
// UserdataSkeletonCarry is the explicit NON-DERIVED carry-list: every entry the v0.171.0 hardcoded
|
||||
// skeleton created, retained verbatim and forever.
|
||||
//
|
||||
// It exists so the catalog-derived skeleton (R-75) can only ever ADD. That makes the zero-removals
|
||||
// invariant true BY CONSTRUCTION rather than by review, and it is not hypothetical:
|
||||
//
|
||||
// - `documents` is implied by NO catalog app (SPIKE P0(a)) yet exists on both demo boxes and is
|
||||
// customer-visible — it may hold customer files. Derivation alone would drop it.
|
||||
//
|
||||
// It doubles as the fresh-box floor: on a box whose catalog has not synced yet the derived set is
|
||||
// empty, and the customer still gets the full standard tree instead of a nearly-empty one.
|
||||
//
|
||||
// DELIBERATELY ABSENT: `import`, `import/paperless`, `import/calibre`. They were in the v0.171.0
|
||||
// hardcoded list, and carrying them would have the skeleton RE-CREATE a per-drive drop-zone on every
|
||||
// drive forever — the exact dead-lookalike R-75 exists to remove, and one that is never backed up
|
||||
// (`class: excluded`). Zero-removals is about not DELETING what a box already has, not about
|
||||
// re-creating it on boxes that never had it: nothing here removes the pre-existing dirs on
|
||||
// demo-felhom / demo-hp, they simply stop being maintained and stop appearing on fresh boxes.
|
||||
// Verified before the change: both boxes' old drop-zones held ZERO files (2026-07-26). A box with
|
||||
// pending files in an old drop-zone would need an operator-run move — see REPORT.md.
|
||||
//
|
||||
// ASCII, no spaces (flows through ${} interpolation, shell, and the rsync merge walk).
|
||||
func UserdataSkeleton() []string {
|
||||
func UserdataSkeletonCarry() []string {
|
||||
return []string{
|
||||
"media", "media/movies", "media/tv", "media/music", "media/audiobooks",
|
||||
"media/books", "media/comics", "media/photos",
|
||||
"downloads",
|
||||
"import", "import/paperless", "import/calibre",
|
||||
"roms",
|
||||
"documents",
|
||||
}
|
||||
}
|
||||
|
||||
// BuildUserdataSkeleton merges the catalog-derived dirs with the carry-list into the final, SORTED
|
||||
// set. Each entry is expanded to its ancestor chain ("media/podcasts" implies "media"), deduped, and
|
||||
// sorted.
|
||||
//
|
||||
// SORTING IS A HARD REQUIREMENT, not tidiness. The FileBrowser config is regenerated from this set
|
||||
// and fbNeedsRecreate force-recreates the container on ANY byte difference. Go randomises map
|
||||
// iteration, and the spike measured the naive map-order derivation producing 20 DISTINCT outputs from
|
||||
// 20 identical runs (SPIKE P6) — which across SyncFileBrowserMounts' ~14 call sites is a fleet-wide
|
||||
// FileBrowser restart loop. TestSkeletonDeterminism pins this.
|
||||
func BuildUserdataSkeleton(derived []string) []string {
|
||||
set := make(map[string]bool, len(derived)+16)
|
||||
addChain := func(rel string) {
|
||||
rel = path.Clean(strings.TrimPrefix(filepath.ToSlash(rel), "/"))
|
||||
if rel == "" || rel == "." || rel == ".." || strings.HasPrefix(rel, "../") {
|
||||
return // never let a traversal or an empty entry become a directory to create
|
||||
}
|
||||
parts := strings.Split(rel, "/")
|
||||
for i := range parts {
|
||||
set[strings.Join(parts[:i+1], "/")] = true
|
||||
}
|
||||
}
|
||||
for _, d := range UserdataSkeletonCarry() {
|
||||
addChain(d)
|
||||
}
|
||||
for _, d := range derived {
|
||||
addChain(d)
|
||||
}
|
||||
out := make([]string, 0, len(set))
|
||||
for d := range set { // map order is RANDOM — the sort below is what makes this deterministic
|
||||
out = append(out, d)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// EnsureDirOwned creates path (idempotent) and enforces the convention: mode 2775 via an explicit
|
||||
// Chmod incl. setgid (MkdirAll cannot) + group = gid. Setting an arbitrary group needs CAP_CHOWN —
|
||||
// the in-guest controller runs as root, so this succeeds in production. Returns the first hard error.
|
||||
@@ -60,7 +135,11 @@ func EnsureUserdataDir(path string) error { return EnsureDirOwned(path, SharedCo
|
||||
// EnsureUserdataSkeleton creates the full userdata tree under a namespace root with the convention.
|
||||
// It creates ALL dirs even if one errors (so a single chown/chmod hiccup doesn't truncate the tree),
|
||||
// returning the first error seen for the caller to log.
|
||||
func EnsureUserdataSkeleton(nsRoot string) error {
|
||||
//
|
||||
// dirs is the merged, sorted set from BuildUserdataSkeleton. This function only ever CREATES: there
|
||||
// is no removal path here or anywhere in R-75, so a directory the current catalog no longer implies
|
||||
// simply stays where it is (Scenario D).
|
||||
func EnsureUserdataSkeleton(nsRoot string, dirs []string) error {
|
||||
base := UserdataDir(nsRoot)
|
||||
var firstErr error
|
||||
rec := func(e error) {
|
||||
@@ -69,7 +148,7 @@ func EnsureUserdataSkeleton(nsRoot string) error {
|
||||
}
|
||||
}
|
||||
rec(EnsureUserdataDir(base))
|
||||
for _, sub := range UserdataSkeleton() {
|
||||
for _, sub := range dirs {
|
||||
rec(EnsureUserdataDir(filepath.Join(base, sub)))
|
||||
}
|
||||
return firstErr
|
||||
|
||||
@@ -13,21 +13,31 @@ func TestSharedContentGID(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestUserdataSkeleton_List asserts the locked skeleton subdir set.
|
||||
// TestUserdataSkeleton_List asserts the locked carry-list. R-75 renamed the hardcoded list to
|
||||
// UserdataSkeletonCarry (it is now the non-derived carry-list) and DELIBERATELY dropped the three
|
||||
// `import*` entries: carrying them would re-create a per-drive drop-zone on every drive forever, the
|
||||
// dead lookalike the canonical root exists to remove. That is not a removal — nothing deletes the
|
||||
// dirs an existing box has; they stop being maintained and stop appearing on fresh boxes. Every other
|
||||
// entry is unchanged, which is the zero-removals promise.
|
||||
func TestUserdataSkeleton_List(t *testing.T) {
|
||||
got := map[string]bool{}
|
||||
for _, s := range UserdataSkeleton() {
|
||||
for _, s := range UserdataSkeletonCarry() {
|
||||
got[s] = true
|
||||
}
|
||||
for _, want := range []string{
|
||||
"media/movies", "media/tv", "media/music", "media/audiobooks", "media/books",
|
||||
"media/comics", "media/photos", "downloads", "import/paperless", "import/calibre",
|
||||
"media/comics", "media/photos", "downloads",
|
||||
"roms", "documents",
|
||||
} {
|
||||
if !got[want] {
|
||||
t.Errorf("skeleton missing %q", want)
|
||||
}
|
||||
}
|
||||
for _, gone := range []string{"import", "import/paperless", "import/calibre"} {
|
||||
if got[gone] {
|
||||
t.Errorf("carry-list must NOT hold %q — the drop-zone is canonical on the system drive (R-75)", gone)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestUserdataDir confirms the userdata root is a sibling under the namespace.
|
||||
@@ -41,9 +51,10 @@ func TestUserdataDir(t *testing.T) {
|
||||
// is ignored — dirs + setgid still land). Runs cross-platform.
|
||||
func TestEnsureUserdataSkeleton_Structure(t *testing.T) {
|
||||
ns := t.TempDir()
|
||||
_ = EnsureUserdataSkeleton(ns) // ignore chown error on a non-root CI host
|
||||
dirs := BuildUserdataSkeleton(nil) // no catalog derived → the carry-list floor
|
||||
_ = EnsureUserdataSkeleton(ns, dirs) // ignore chown error on a non-root CI host
|
||||
base := UserdataDir(ns)
|
||||
for _, sub := range append([]string{""}, UserdataSkeleton()...) {
|
||||
for _, sub := range append([]string{""}, dirs...) {
|
||||
p := filepath.Join(base, sub)
|
||||
if fi, err := os.Stat(p); err != nil || !fi.IsDir() {
|
||||
t.Errorf("skeleton dir missing: %s (%v)", p, err)
|
||||
|
||||
@@ -15,8 +15,8 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
magicHeader = "FABE" // Felhom App Bundle Encrypted
|
||||
scryptN = 1 << 15 // 32768
|
||||
magicHeader = "FABE" // Felhom App Bundle Encrypted
|
||||
scryptN = 1 << 15 // 32768
|
||||
scryptR = 8
|
||||
scryptP = 1
|
||||
saltSize = 32
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
@@ -22,6 +23,31 @@ type ExportEstimate struct {
|
||||
DestFreeBytes int64 `json:"dest_free_bytes"`
|
||||
DestFreeHuman string `json:"dest_free_human"`
|
||||
FitsOnDest bool `json:"fits_on_dest"`
|
||||
// SizeUnknown is set (v0.129.0 F-A) when a volume's size could not be read (docker helper
|
||||
// failed). When true, DataSizeBytes is a partial/understated sum and FitsOnDest is FORCED false
|
||||
// — a failed read must NEVER render as "fits". The UI shows "ismeretlen méret".
|
||||
SizeUnknown bool `json:"size_unknown"`
|
||||
|
||||
// Task 4 class split (classified apps only; empty for legacy — existing fields above are
|
||||
// unchanged, so old JSON consumers keep working). BaseBytes = config + DB + volumes + mandatory
|
||||
// (always in the bundle). OptionalItems are pre-selected, ExcludedItems are opt-in — each carries
|
||||
// its own size so the UI recomputes the total client-side per checkbox toggle (no extra du calls).
|
||||
HasClassification bool `json:"has_classification"`
|
||||
BaseBytes int64 `json:"base_bytes"`
|
||||
BaseHuman string `json:"base_human"`
|
||||
MandatoryItems []FabItem `json:"mandatory_items,omitempty"`
|
||||
OptionalItems []FabItem `json:"optional_items,omitempty"`
|
||||
ExcludedItems []FabItem `json:"excluded_items,omitempty"`
|
||||
}
|
||||
|
||||
// FabItem is one class-scoped path in the `.fab` selection UI: Key is the DeselectOptional/OptInExcluded
|
||||
// value ("root/rel"), RelPath is the display path, Bytes/Human its du size.
|
||||
type FabItem struct {
|
||||
Key string `json:"key"`
|
||||
Root string `json:"root"`
|
||||
RelPath string `json:"rel_path"`
|
||||
Bytes int64 `json:"bytes"`
|
||||
Human string `json:"human"`
|
||||
}
|
||||
|
||||
// EstimateExport calculates size estimates for an app export.
|
||||
@@ -39,7 +65,8 @@ func (e *Exporter) EstimateExport(stackName, destDrive string) (*ExportEstimate,
|
||||
est.ConfigSizeHuman = humanizeBytes(est.ConfigSizeBytes)
|
||||
e.debugf("EstimateExport: configSize=%s (%d bytes)", est.ConfigSizeHuman, est.ConfigSizeBytes)
|
||||
|
||||
// Data size: HDD bind mounts or Docker volumes
|
||||
// Data size: HDD bind mounts PLUS Docker volumes. v0.130.0 (C6B-F1): additive, mirroring the
|
||||
// export itself — a needs_hdd app bundles BOTH, so the fits-on-dest gate must count both.
|
||||
if e.provider.GetStackNeedsHDD(stackName) {
|
||||
mounts := e.provider.GetStackHDDMounts(stackName)
|
||||
e.debugf("EstimateExport: HDD mounts: %v", mounts)
|
||||
@@ -48,20 +75,36 @@ func (e *Exporter) EstimateExport(stackName, destDrive string) (*ExportEstimate,
|
||||
e.debugf("EstimateExport: mount %s = %s", mount, humanizeBytes(mountSize))
|
||||
est.DataSizeBytes += mountSize
|
||||
}
|
||||
} else {
|
||||
volumes := e.provider.GetDockerVolumes(stackName)
|
||||
e.debugf("EstimateExport: Docker volumes: %v", volumes)
|
||||
for _, vol := range volumes {
|
||||
volSize := dockerVolumeSize(vol)
|
||||
e.debugf("EstimateExport: volume %s = %s", vol, humanizeBytes(volSize))
|
||||
est.DataSizeBytes += volSize
|
||||
}
|
||||
}
|
||||
est.DataSizeHuman = humanizeBytes(est.DataSizeBytes)
|
||||
volumes := e.provider.GetDockerVolumes(stackName)
|
||||
e.debugf("EstimateExport: Docker volumes: %v", volumes)
|
||||
var volumeBytes int64
|
||||
for _, vol := range volumes {
|
||||
volSize, err := volumeSizer(vol)
|
||||
if err != nil {
|
||||
// F-A: the controller runs containerized, so a failed helper read must not
|
||||
// silently become 0-that-reads-as-fits. Mark unknown and keep going.
|
||||
e.logger.Printf("[WARN] appexport: volume size unknown for %s: %v", vol, err)
|
||||
est.SizeUnknown = true
|
||||
continue
|
||||
}
|
||||
e.debugf("EstimateExport: volume %s = %s", vol, humanizeBytes(volSize))
|
||||
est.DataSizeBytes += volSize
|
||||
volumeBytes += volSize
|
||||
}
|
||||
if est.SizeUnknown {
|
||||
est.DataSizeHuman = "ismeretlen méret"
|
||||
} else {
|
||||
est.DataSizeHuman = humanizeBytes(est.DataSizeBytes)
|
||||
}
|
||||
|
||||
est.TotalSizeBytes = est.ConfigSizeBytes + est.DataSizeBytes
|
||||
est.TotalSizeHuman = humanizeBytes(est.TotalSizeBytes)
|
||||
|
||||
// Task 4: the class split (classified apps only). Independent du over each bucket path — additive,
|
||||
// never touches the fields/fits gate above.
|
||||
e.fabEstimateSplit(stackName, est, volumeBytes)
|
||||
|
||||
// Rough time estimate: ~500 MB/min for HDDs, minimum 1 minute
|
||||
minutes := int(est.TotalSizeBytes / (500 * 1024 * 1024))
|
||||
if minutes < 1 {
|
||||
@@ -72,12 +115,13 @@ func (e *Exporter) EstimateExport(stackName, destDrive string) (*ExportEstimate,
|
||||
// Destination free space
|
||||
exportDir := ExportDir(destDrive)
|
||||
os.MkdirAll(exportDir, 0755)
|
||||
est.DestFreeBytes = diskFree(exportDir)
|
||||
est.DestFreeBytes = DiskFree(exportDir)
|
||||
est.DestFreeHuman = humanizeBytes(est.DestFreeBytes)
|
||||
|
||||
// Need ~10% overhead for tar.gz metadata + compression margin
|
||||
// Need ~10% overhead for tar.gz metadata + compression margin. F-A: a size we could not read
|
||||
// must never render as "fits" — an unknown-size estimate is conservatively not-fits.
|
||||
needed := est.TotalSizeBytes + est.TotalSizeBytes/10
|
||||
est.FitsOnDest = est.DestFreeBytes >= needed
|
||||
est.FitsOnDest = !est.SizeUnknown && est.DestFreeBytes >= needed
|
||||
|
||||
e.debugf("EstimateExport: total=%s free=%s fits=%v needed=%s minutes=%d",
|
||||
est.TotalSizeHuman, est.DestFreeHuman, est.FitsOnDest, humanizeBytes(needed), est.EstimatedMinutes)
|
||||
@@ -118,25 +162,38 @@ func duBytes(path string) int64 {
|
||||
return size
|
||||
}
|
||||
|
||||
// dockerVolumeSize estimates the size of a Docker named volume.
|
||||
func dockerVolumeSize(volumeName string) int64 {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
// volumeSizer returns the byte size of a named Docker volume as seen from a CONTAINER view.
|
||||
// Package var so unit tests inject a fake (returning a known size or an error) without shelling out
|
||||
// to real docker. F-A (v0.129.0): the old dockerVolumeSize `du`d the host mountpoint from
|
||||
// `docker volume inspect`, which is NOT visible inside the containerized controller → always 0.
|
||||
var volumeSizer = realVolumeSize
|
||||
|
||||
// realVolumeSize `du -sb`s the volume mounted read-only into a throwaway helper container — the same
|
||||
// container-view pattern the export path uses (appexport/export.go withVolumeHelper). It mounts the
|
||||
// NAMED VOLUME by name (never a controller-host path — the v0.125.0 strand class). Returns an error
|
||||
// on any failure; callers treat that as "unknown size", never as 0.
|
||||
func realVolumeSize(volumeName string) (int64, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
// Use docker system df -v and parse, or inspect the volume mount path
|
||||
out, err := exec.CommandContext(ctx, "docker", "volume", "inspect",
|
||||
"--format", "{{.Mountpoint}}", volumeName).Output()
|
||||
var out bytes.Buffer
|
||||
stderr, err := dockerExec(ctx, nil, &out, "run", "--rm", "-v", volumeName+":/vol:ro", "alpine", "du", "-sb", "/vol")
|
||||
if err != nil {
|
||||
return 0
|
||||
return 0, fmt.Errorf("sizing volume %s: %s: %w", volumeName, stderr, err)
|
||||
}
|
||||
mountpoint := strings.TrimSpace(string(out))
|
||||
if mountpoint == "" {
|
||||
return 0
|
||||
fields := strings.Fields(out.String())
|
||||
if len(fields) == 0 {
|
||||
return 0, fmt.Errorf("sizing volume %s: empty du output", volumeName)
|
||||
}
|
||||
return duBytes(mountpoint)
|
||||
var size int64
|
||||
if _, err := fmt.Sscanf(fields[0], "%d", &size); err != nil {
|
||||
return 0, fmt.Errorf("sizing volume %s: parse %q: %w", volumeName, fields[0], err)
|
||||
}
|
||||
return size, nil
|
||||
}
|
||||
|
||||
// diskFree returns available bytes on the filesystem containing path.
|
||||
func diskFree(path string) int64 {
|
||||
// DiskFree returns available bytes on the filesystem containing path (0 on any error).
|
||||
// Exported since v0.128.0 — the browser-upload space gate reuses it via a web-package seam.
|
||||
func DiskFree(path string) int64 {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
out, err := exec.CommandContext(ctx, "df", "--output=avail", "-B1", path).Output()
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// hddProvider is an rtProvider that reports an HDD-backed stack (estimate scenario H + the
|
||||
// v0.130.0 additive-export tests in export_additive_test.go).
|
||||
type hddProvider struct {
|
||||
*rtProvider
|
||||
mounts []string
|
||||
hddPath string
|
||||
binds []appbackup.ClassifiedBind
|
||||
hasBinds bool
|
||||
}
|
||||
|
||||
func (p *hddProvider) GetStackNeedsHDD(string) bool { return true }
|
||||
func (p *hddProvider) GetStackHDDMounts(string) []string { return p.mounts }
|
||||
func (p *hddProvider) GetStackHDDPath(string) string { return p.hddPath }
|
||||
func (p *hddProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
|
||||
// R-203: these fixtures use ENROLLED drive paths, where the namespace root IS the drive path.
|
||||
// Delegating keeps that identity explicit rather than hardcoding it.
|
||||
func (p *hddProvider) GetStackNamespaceRoot(name string) string { return p.GetStackHDDPath(name) }
|
||||
func (p *hddProvider) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return p.binds, p.hasBinds
|
||||
}
|
||||
|
||||
func newEstimator(t *testing.T, provider ExportStackProvider) *Exporter {
|
||||
t.Helper()
|
||||
return NewExporter(provider, log.New(io.Discard, "", 0), "test")
|
||||
}
|
||||
|
||||
// Scenario F (the F-A fix): a volume-only app with a >1 GiB volume reports the REAL size via the
|
||||
// container-view sizer — not 0/"3.6 KB". This is the F-A red-proof anchor (revert EstimateExport to
|
||||
// dockerVolumeSize → reads 0).
|
||||
func TestEstimate_VolumeSize_RealNotZero(t *testing.T) {
|
||||
const twoGiB = int64(2) << 30
|
||||
orig := volumeSizer
|
||||
volumeSizer = func(vol string) (int64, error) { return twoGiB, nil }
|
||||
defer func() { volumeSizer = orig }()
|
||||
|
||||
e := newEstimator(t, &rtProvider{stackDir: t.TempDir(), volumes: []string{"app_data"}})
|
||||
est, err := e.EstimateExport("app", t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if est.SizeUnknown {
|
||||
t.Fatalf("size must be known when the sizer succeeds")
|
||||
}
|
||||
if est.DataSizeBytes != twoGiB {
|
||||
t.Fatalf("DataSizeBytes = %d, want %d (WRONG would be 0 — the F-A bug)", est.DataSizeBytes, twoGiB)
|
||||
}
|
||||
if !strings.Contains(est.DataSizeHuman, "GB") {
|
||||
t.Fatalf("DataSizeHuman = %q, want GB-scale (WRONG would be \"3.6 KB\")", est.DataSizeHuman)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario G: a failed volume read must never render as "fits". Size is marked unknown, the human
|
||||
// string says so, and FitsOnDest is forced false.
|
||||
func TestEstimate_VolumeSize_FailureNeverFits(t *testing.T) {
|
||||
orig := volumeSizer
|
||||
volumeSizer = func(vol string) (int64, error) { return 0, errors.New("docker: no such image") }
|
||||
defer func() { volumeSizer = orig }()
|
||||
|
||||
e := newEstimator(t, &rtProvider{stackDir: t.TempDir(), volumes: []string{"app_data"}})
|
||||
est, err := e.EstimateExport("app", t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !est.SizeUnknown {
|
||||
t.Fatalf("a failed volume read must set SizeUnknown")
|
||||
}
|
||||
if est.FitsOnDest {
|
||||
t.Fatalf("an unknown size must NEVER render as fits_on_dest:true")
|
||||
}
|
||||
if est.DataSizeHuman != "ismeretlen méret" {
|
||||
t.Fatalf("DataSizeHuman = %q, want \"ismeretlen méret\"", est.DataSizeHuman)
|
||||
}
|
||||
if est.DataSizeBytes != 0 {
|
||||
t.Fatalf("no successful read → DataSizeBytes should be 0, got %d", est.DataSizeBytes)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario H (regression): an HDD-backed stack must NOT touch the new volume sizer — the HDD branch
|
||||
// (duBytes on the mounted /mnt path) is unchanged. Platform-independent: assert the seam is not
|
||||
// invoked and SizeUnknown stays false.
|
||||
func TestEstimate_HDDPath_DoesNotUseVolumeSizer(t *testing.T) {
|
||||
called := false
|
||||
orig := volumeSizer
|
||||
volumeSizer = func(vol string) (int64, error) { called = true; return 0, nil }
|
||||
defer func() { volumeSizer = orig }()
|
||||
|
||||
p := &hddProvider{rtProvider: &rtProvider{stackDir: t.TempDir()}, mounts: []string{t.TempDir()}}
|
||||
e := newEstimator(t, p)
|
||||
est, err := e.EstimateExport("app", t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if called {
|
||||
t.Fatalf("HDD-backed stack must not call the docker volume sizer")
|
||||
}
|
||||
if est.SizeUnknown {
|
||||
t.Fatalf("HDD branch must not set SizeUnknown")
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package appexport
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"bytes"
|
||||
"compress/gzip"
|
||||
"context"
|
||||
"fmt"
|
||||
@@ -74,6 +75,14 @@ type ExportRequest struct {
|
||||
DestDrive string // drive mount path (e.g., "/mnt/hdd_1")
|
||||
Password string // empty = no encryption
|
||||
StopApp bool // stop app before export
|
||||
|
||||
// `.fab` class-scoped selection (Task 4; classified apps only — legacy apps ignore these). Both
|
||||
// empty = ruling #1 defaults (mandatory in, optional in, excluded out). Values are "root/rel" keys
|
||||
// (matching a CapturePath: "hdd/appdata/x" | "userdata/media/y"). The server enforces the floor:
|
||||
// a DeselectOptional entry naming a MANDATORY path is ignored with a WARN (the client cannot weaken
|
||||
// the mandatory floor). OptInExcluded pulls an excluded bind into the bundle.
|
||||
DeselectOptional []string
|
||||
OptInExcluded []string
|
||||
}
|
||||
|
||||
// Exporter manages app export/import operations.
|
||||
@@ -83,10 +92,30 @@ type Exporter struct {
|
||||
version string
|
||||
debug bool
|
||||
|
||||
// dirLister (Task 4) lists child DIR names of a path — the seam the `.fab` userdata-exclude
|
||||
// computation walks. Nil → the real os.ReadDir-based lister.
|
||||
dirLister func(dir string) []string
|
||||
|
||||
// stopGuard (R-166) marks the stop→export→start window so a controller killed inside it leaves a
|
||||
// durable record that the app is owed a restart. Declared consumer-side as a two-method interface
|
||||
// so this package does not import internal/backup; main.go passes the backup manager's guard, so
|
||||
// BOTH packages write ONE marker file — an exporter with its own file would be a second writer
|
||||
// racing the same recovery. Nil = not wired (tests): the export runs exactly as it did before.
|
||||
stopGuard appStopGuard
|
||||
|
||||
mu sync.Mutex
|
||||
activeJob *Job
|
||||
}
|
||||
|
||||
// appStopGuard is the app-stop crash-marker seam. The REASON is deliberately not a parameter: it is
|
||||
// always "app export" from here, and the adapter in main.go supplies it. Passing it as a string
|
||||
// would duplicate backup.ReasonAppExport's value in a second package with nothing keeping the two in
|
||||
// step — a drift this codebase has paid for before (the offbox key that was guessed, R-7b).
|
||||
type appStopGuard interface {
|
||||
Begin(opID string, stacks []string) error
|
||||
End()
|
||||
}
|
||||
|
||||
// NewExporter creates a new export/import engine.
|
||||
func NewExporter(provider ExportStackProvider, logger *log.Logger, version string) *Exporter {
|
||||
return &Exporter{
|
||||
@@ -96,6 +125,18 @@ func NewExporter(provider ExportStackProvider, logger *log.Logger, version strin
|
||||
}
|
||||
}
|
||||
|
||||
// SetStopGuard wires the app-stop crash marker. INIT-ONLY — call once at startup, before any export.
|
||||
func (e *Exporter) SetStopGuard(g appStopGuard) { e.stopGuard = g }
|
||||
|
||||
// stopGuardBegin records the app-stop marker before an export stops an app. An unwired guard is a
|
||||
// no-op (pre-v0.189.0 behaviour), never an error — a test exporter must not be forced to have one.
|
||||
func (e *Exporter) stopGuardBegin(stackName string) error {
|
||||
if e.stopGuard == nil {
|
||||
return nil
|
||||
}
|
||||
return e.stopGuard.Begin("app-export:"+stackName, []string{stackName})
|
||||
}
|
||||
|
||||
// SetDebug enables or disables verbose debug logging.
|
||||
func (e *Exporter) SetDebug(debug bool) {
|
||||
e.debug = debug
|
||||
@@ -196,9 +237,14 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
if err != nil {
|
||||
e.debugf("estimate error (non-fatal): %v", err)
|
||||
} else {
|
||||
e.debugf("estimate: config=%s data=%s total=%s destFree=%s fits=%v",
|
||||
est.ConfigSizeHuman, est.DataSizeHuman, est.TotalSizeHuman, est.DestFreeHuman, est.FitsOnDest)
|
||||
if !est.FitsOnDest {
|
||||
e.debugf("estimate: config=%s data=%s total=%s destFree=%s fits=%v unknown=%v",
|
||||
est.ConfigSizeHuman, est.DataSizeHuman, est.TotalSizeHuman, est.DestFreeHuman, est.FitsOnDest, est.SizeUnknown)
|
||||
// Hard-abort only on a KNOWN doesn't-fit. F-A: est.SizeUnknown forces FitsOnDest=false for
|
||||
// the UI honesty signal, but an unmeasured size must NOT block the export here — the tar
|
||||
// streaming and the destination filesystem surface a real ENOSPC if it genuinely won't fit.
|
||||
if est.SizeUnknown {
|
||||
e.logger.Printf("[WARN] appexport: export space pre-check skipped for %s — volume size unknown", req.StackName)
|
||||
} else if !est.FitsOnDest {
|
||||
e.failJob(job, step, fmt.Sprintf("Nincs elég hely: szükséges ~%s, szabad %s",
|
||||
est.TotalSizeHuman, est.DestFreeHuman))
|
||||
return
|
||||
@@ -208,6 +254,14 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
// Optionally stop the app
|
||||
wasRunning := false
|
||||
if req.StopApp && e.provider.IsStackRunning(req.StackName) {
|
||||
// R-166: mark BEFORE the stop. The defer below covers the graceful exits; it does NOT cover a
|
||||
// SIGKILL or a power cut, which run no deferred function (Campaign 8 fault 10, on live
|
||||
// hardware) — only this marker does, and a big export is a long window to be killed in.
|
||||
if err := e.stopGuardBegin(req.StackName); err != nil {
|
||||
e.failJob(job, step, "Az alkalmazás leállítása előtti jelölő nem menthető — az exportálás nem indult el.")
|
||||
e.logger.Printf("[ERROR] Export: could not record the app-stop marker for %s (refusing to stop it unprotected): %v", req.StackName, err)
|
||||
return
|
||||
}
|
||||
wasRunning = true
|
||||
e.logger.Printf("[INFO] Export: stopping %s", req.StackName)
|
||||
e.debugf("stopping stack %s before export", req.StackName)
|
||||
@@ -228,6 +282,11 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
e.logger.Printf("[WARN] Export: could not restart %s: %v", req.StackName, err)
|
||||
} else {
|
||||
e.debugf("stack %s restarted successfully", req.StackName)
|
||||
// Cleared only on a restart that succeeded — a failed one keeps the marker so the
|
||||
// next startup retries.
|
||||
if e.stopGuard != nil {
|
||||
e.stopGuard.End()
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
@@ -318,15 +377,24 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
dataDir := filepath.Join(tmpDir, "data")
|
||||
os.MkdirAll(dataDir, 0755)
|
||||
|
||||
// C6B-F1 cause 1 (v0.130.0): user-data capture is ADDITIVE, not either/or. A needs_hdd app
|
||||
// can hold state in BOTH its HDD binds and its named volumes (sonarr: ${USERDATA_PATH} media
|
||||
// binds + the sonarr_config volume with the entire app DB) — the old else-branch silently
|
||||
// dropped every named volume of every needs_hdd app.
|
||||
if e.provider.GetStackNeedsHDD(req.StackName) {
|
||||
e.debugf("exporting HDD data for %s", req.StackName)
|
||||
e.exportHDDData(req.StackName, dataDir, manifest)
|
||||
if err := e.exportHDDData(req, dataDir, manifest); err != nil {
|
||||
e.failJob(job, step, fmt.Sprintf("Felhasználói adatok mentése sikertelen: %v", err))
|
||||
return
|
||||
}
|
||||
e.debugf("HDD data exported: subdirs=%v hasData=%v", manifest.HDDSubdirs, manifest.HasHDDData)
|
||||
} else {
|
||||
e.debugf("exporting Docker volumes for %s", req.StackName)
|
||||
e.exportVolumeData(req.StackName, dataDir, manifest)
|
||||
e.debugf("volume data exported: volumes=%v hasData=%v", manifest.VolumeNames, manifest.HasVolumeData)
|
||||
}
|
||||
e.debugf("exporting Docker volumes for %s", req.StackName)
|
||||
if err := e.exportVolumeData(req.StackName, dataDir, manifest); err != nil {
|
||||
e.failJob(job, step, fmt.Sprintf("Kötet mentése sikertelen: %v", err))
|
||||
return
|
||||
}
|
||||
e.debugf("volume data exported: volumes=%v hasData=%v", manifest.VolumeNames, manifest.HasVolumeData)
|
||||
e.debugf("step 3 (user data) done in %v", time.Since(stepStart))
|
||||
|
||||
job.setStep(step, "done", "")
|
||||
@@ -336,6 +404,13 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
job.setStep(step, "running", "")
|
||||
stepStart = time.Now()
|
||||
|
||||
// v0.125.0 fail-loud guard (scenario B): never package a bundle whose manifest claims data
|
||||
// that is not actually in the staging tree.
|
||||
if err := assertBundleDataComplete(tmpDir, manifest); err != nil {
|
||||
e.failJob(job, step, fmt.Sprintf("A csomag hiányos lenne — az export leállt: %v", err))
|
||||
return
|
||||
}
|
||||
|
||||
// Calculate total size
|
||||
manifest.TotalSizeBytes = calcDirSize(tmpDir)
|
||||
e.debugf("total bundle content size: %s (%d bytes)", humanizeBytes(manifest.TotalSizeBytes), manifest.TotalSizeBytes)
|
||||
@@ -434,8 +509,8 @@ func (e *Exporter) GetDebugInfo() map[string]interface{} {
|
||||
defer e.mu.Unlock()
|
||||
|
||||
info := map[string]interface{}{
|
||||
"debug_enabled": e.debug,
|
||||
"version": e.version,
|
||||
"debug_enabled": e.debug,
|
||||
"version": e.version,
|
||||
"has_active_job": e.activeJob != nil,
|
||||
}
|
||||
|
||||
@@ -565,8 +640,12 @@ func (e *Exporter) dumpDatabase(stackName, dbDir string, manifest *Manifest) boo
|
||||
return true
|
||||
}
|
||||
|
||||
// exportHDDData copies HDD bind mount data for the export.
|
||||
func (e *Exporter) exportHDDData(stackName, dataDir string, manifest *Manifest) {
|
||||
// exportHDDData copies HDD bind mount data for the export. v0.130.0 (C6B-F1 §8): a basename
|
||||
// collision between two mounts is a FATAL error (the manifest keys tars by basename; the old
|
||||
// code silently overwrote the first tar). A non-existent mount is still soft-skipped (honestly
|
||||
// absent from the manifest — the anti-hollow guard catches total emptiness).
|
||||
func (e *Exporter) exportHDDData(req ExportRequest, dataDir string, manifest *Manifest) error {
|
||||
stackName := req.StackName
|
||||
hddDir := filepath.Join(dataDir, "hdd")
|
||||
os.MkdirAll(hddDir, 0755)
|
||||
|
||||
@@ -574,33 +653,137 @@ func (e *Exporter) exportHDDData(stackName, dataDir string, manifest *Manifest)
|
||||
e.debugf("HDD mounts for %s: %v (%d total)", stackName, mounts, len(mounts))
|
||||
if len(mounts) == 0 {
|
||||
e.debugf("no HDD mounts — skipping HDD data export")
|
||||
return
|
||||
return nil
|
||||
}
|
||||
|
||||
// Task 4: the class-scoped plan. Legacy / no-block apps get an EMPTY plan (all mounts kept, root
|
||||
// tar with zero excludes) → byte-identical v0.130.0 capture.
|
||||
plan := e.computeFabPlan(req, mounts)
|
||||
// R-203: a NAMESPACE ROOT, not the drive path (identical on an enrolled drive; one segment short
|
||||
// on the system-data fallback).
|
||||
ud := appbackup.UserdataDir(filepath.Clean(e.provider.GetStackNamespaceRoot(stackName)))
|
||||
|
||||
claimed := make(map[string]string) // subdir → mount that claimed it
|
||||
for _, mount := range mounts {
|
||||
if plan.SkipMounts[filepath.Clean(mount)] {
|
||||
e.debugf("HDD mount %s skipped — not selected (class-scoped plan)", mount)
|
||||
continue
|
||||
}
|
||||
isUserdataRoot := filepath.Clean(mount) == filepath.Clean(ud)
|
||||
if isUserdataRoot && plan.SkipUserdataTar {
|
||||
e.debugf("userdata root %s skipped — no selected userdata bind (Scenario B)", mount)
|
||||
continue
|
||||
}
|
||||
if _, err := os.Stat(mount); os.IsNotExist(err) {
|
||||
e.debugf("HDD mount %s does not exist — skipping", mount)
|
||||
continue
|
||||
}
|
||||
subdir := filepath.Base(mount)
|
||||
manifest.HDDSubdirs = append(manifest.HDDSubdirs, subdir)
|
||||
// C6B-F1 §8 (v0.130.0): the manifest keys HDD tars by BASENAME (the import side maps a
|
||||
// basename back to a path), so two mounts sharing a basename cannot round-trip — the old
|
||||
// code silently overwrote the first tar with the second (silent partial data loss).
|
||||
// Renaming can't help either (the import couldn't map the new name), so the only honest
|
||||
// outcome is a loud failure.
|
||||
if prev, dup := claimed[subdir]; dup {
|
||||
return fmt.Errorf("két adatkönyvtár azonos névvel végződik (%q: %s és %s) — a csomag nem tudná megkülönböztetni őket", subdir, prev, mount)
|
||||
}
|
||||
claimed[subdir] = mount
|
||||
|
||||
tarPath := filepath.Join(hddDir, subdir+".tar")
|
||||
e.debugf("tarring HDD mount: %s → %s", mount, tarPath)
|
||||
var excludes []string
|
||||
if isUserdataRoot {
|
||||
excludes = plan.UserdataExcludeRels // R1-C: exclude-scoped root tar (empty for legacy)
|
||||
}
|
||||
e.debugf("tarring HDD mount: %s → %s (%d exclude(s))", mount, tarPath, len(excludes))
|
||||
tarStart := time.Now()
|
||||
if err := tarDirectory(mount, tarPath); err != nil {
|
||||
e.logger.Printf("[WARN] Export: failed to tar %s: %v", mount, err)
|
||||
if err := tarDirectoryExcluding(mount, tarPath, excludes); err != nil {
|
||||
// v0.125.0: claim the subdir ONLY on success — a claimed-but-absent tar would trip
|
||||
// the packaging assertion; an honestly-skipped mount stays out of the manifest.
|
||||
e.logger.Printf("[WARN] Export: failed to tar %s (excluded from the bundle): %v", mount, err)
|
||||
os.Remove(tarPath)
|
||||
} else {
|
||||
manifest.HDDSubdirs = append(manifest.HDDSubdirs, subdir)
|
||||
if info, _ := os.Stat(tarPath); info != nil {
|
||||
e.debugf("HDD tar complete: %s (%s) in %v", subdir, humanizeBytes(info.Size()), time.Since(tarStart))
|
||||
}
|
||||
}
|
||||
}
|
||||
manifest.HasHDDData = len(manifest.HDDSubdirs) > 0
|
||||
return nil
|
||||
}
|
||||
|
||||
// exportVolumeData exports Docker named volumes for apps without HDD storage.
|
||||
func (e *Exporter) exportVolumeData(stackName, dataDir string, manifest *Manifest) {
|
||||
// dockerExec is the docker-CLI seam (v0.125.0): runs `docker args...` with optional
|
||||
// stdin/stdout STREAMING and returns captured stderr (truncated). The volume legs stream tars
|
||||
// over the docker API (docker cp) — NEVER via `docker run -v <controller-path>` host mounts,
|
||||
// which the daemon resolves against the GUEST filesystem and silently strands the tar when the
|
||||
// controller itself runs containerized (the v0.124.0 HIGH finding). Package var so unit tests
|
||||
// inject a recorder (no docker on test boxes).
|
||||
var dockerExec = func(ctx context.Context, stdin io.Reader, stdout io.Writer, args ...string) (string, error) {
|
||||
cmd := exec.CommandContext(ctx, "docker", args...)
|
||||
cmd.Stdin = stdin
|
||||
cmd.Stdout = stdout
|
||||
var errBuf bytes.Buffer
|
||||
cmd.Stderr = &errBuf
|
||||
err := cmd.Run()
|
||||
stderr := strings.TrimSpace(errBuf.String())
|
||||
if len(stderr) > 500 {
|
||||
stderr = stderr[:500] + "..."
|
||||
}
|
||||
return stderr, err
|
||||
}
|
||||
|
||||
// withVolumeHelper creates a stopped helper container pinning volName at /vol, runs fn(cid),
|
||||
// and ALWAYS force-removes the helper — including on fn failure (no leaked alpine containers).
|
||||
func (e *Exporter) withVolumeHelper(volName string, fn func(cid string) error) error {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
var cidBuf bytes.Buffer
|
||||
stderr, err := dockerExec(ctx, nil, &cidBuf, "create", "-v", volName+":/vol", "alpine", "true")
|
||||
cancel()
|
||||
if err != nil {
|
||||
return fmt.Errorf("creating helper container for volume %s: %s — %w", volName, stderr, err)
|
||||
}
|
||||
cid := strings.TrimSpace(cidBuf.String())
|
||||
defer func() {
|
||||
rmCtx, rmCancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer rmCancel()
|
||||
if _, rmErr := dockerExec(rmCtx, nil, nil, "rm", "-f", cid); rmErr != nil {
|
||||
e.logger.Printf("[WARN] appexport: helper container %s cleanup failed: %v", cid, rmErr)
|
||||
}
|
||||
}()
|
||||
return fn(cid)
|
||||
}
|
||||
|
||||
// exportVolumeTar streams one volume's content into tarPath via `docker cp <cid>:/vol/. -`
|
||||
// (tar on stdout — zero shared paths; live-probed 2026-07-13: content, subdirs, symlinks,
|
||||
// empty files and uid/gid all round-trip).
|
||||
func (e *Exporter) exportVolumeTar(volName, tarPath string) error {
|
||||
return e.withVolumeHelper(volName, func(cid string) error {
|
||||
f, err := os.Create(tarPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("creating %s: %w", tarPath, err)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
||||
defer cancel()
|
||||
stderr, err := dockerExec(ctx, nil, f, "cp", cid+":/vol/.", "-")
|
||||
closeErr := f.Close()
|
||||
if err != nil {
|
||||
os.Remove(tarPath)
|
||||
return fmt.Errorf("streaming volume %s: %s — %w", volName, stderr, err)
|
||||
}
|
||||
if closeErr != nil {
|
||||
os.Remove(tarPath)
|
||||
return fmt.Errorf("flushing %s: %w", tarPath, closeErr)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
}
|
||||
|
||||
// exportVolumeData exports the app's Docker named volumes. v0.130.0 (C6B-F1): runs for EVERY
|
||||
// app — needs_hdd apps hold state in named volumes too (sonarr_config = the whole app DB); the
|
||||
// pre-fix else-branch silently dropped them. v0.125.0: a failed volume export is FATAL (export
|
||||
// must never report success on a hollow bundle — the pre-fix WARN+continue is exactly how the
|
||||
// data-loss bundles were born).
|
||||
func (e *Exporter) exportVolumeData(stackName, dataDir string, manifest *Manifest) error {
|
||||
volDir := filepath.Join(dataDir, "volumes")
|
||||
os.MkdirAll(volDir, 0755)
|
||||
|
||||
@@ -608,28 +791,15 @@ func (e *Exporter) exportVolumeData(stackName, dataDir string, manifest *Manifes
|
||||
e.debugf("Docker volumes for %s: %v (%d total)", stackName, volumes, len(volumes))
|
||||
if len(volumes) == 0 {
|
||||
e.debugf("no Docker volumes — skipping volume data export")
|
||||
return
|
||||
return nil
|
||||
}
|
||||
|
||||
for _, volName := range volumes {
|
||||
tarPath := filepath.Join(volDir, volName+".tar")
|
||||
e.debugf("exporting volume %s via docker run alpine tar...", volName)
|
||||
e.debugf("exporting volume %s via docker cp streaming...", volName)
|
||||
volStart := time.Now()
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
||||
cmd := exec.CommandContext(ctx, "docker", "run", "--rm",
|
||||
"-v", volName+":/vol:ro",
|
||||
"-v", volDir+":/out",
|
||||
"alpine", "tar", "cf", "/out/"+volName+".tar", "-C", "/vol", ".")
|
||||
out, err := cmd.CombinedOutput()
|
||||
cancel()
|
||||
|
||||
if err != nil {
|
||||
e.logger.Printf("[WARN] Export: volume %s export failed: %s — %v",
|
||||
volName, strings.TrimSpace(string(out)), err)
|
||||
e.debugf("volume %s export failed: %s", volName, strings.TrimSpace(string(out)))
|
||||
os.Remove(tarPath)
|
||||
continue
|
||||
if err := e.exportVolumeTar(volName, tarPath); err != nil {
|
||||
return fmt.Errorf("volume %s export failed: %w", volName, err)
|
||||
}
|
||||
if info, _ := os.Stat(tarPath); info != nil {
|
||||
e.debugf("volume %s exported: %s in %v", volName, humanizeBytes(info.Size()), time.Since(volStart))
|
||||
@@ -637,6 +807,36 @@ func (e *Exporter) exportVolumeData(stackName, dataDir string, manifest *Manifes
|
||||
manifest.VolumeNames = append(manifest.VolumeNames, volName)
|
||||
}
|
||||
manifest.HasVolumeData = len(manifest.VolumeNames) > 0
|
||||
return nil
|
||||
}
|
||||
|
||||
// assertBundleDataComplete is the fail-loud post-export guard (v0.125.0, scenario B): every
|
||||
// manifest-CLAIMED data tar must exist non-empty in the staging tree before packaging. A
|
||||
// mismatch aborts the export — yesterday's outcome ("success" with a hollow bundle) is the
|
||||
// one this exists to make impossible. v0.130.0 (C6B-F1 cause 3): also refuses a needs_hdd
|
||||
// bundle that claims NO data at all — the claimed-tar checks pass trivially on 0 claims, which
|
||||
// is how a discovery gap shipped hollow bundles right past the v0.125.0 net.
|
||||
func assertBundleDataComplete(tmpDir string, manifest *Manifest) error {
|
||||
for _, v := range manifest.VolumeNames {
|
||||
fi, err := os.Stat(filepath.Join(tmpDir, "data", "volumes", v+".tar"))
|
||||
if err != nil || fi.Size() == 0 {
|
||||
return fmt.Errorf("bundle assertion: volume %q is claimed by the manifest but its tar is missing or empty", v)
|
||||
}
|
||||
}
|
||||
for _, s := range manifest.HDDSubdirs {
|
||||
fi, err := os.Stat(filepath.Join(tmpDir, "data", "hdd", s+".tar"))
|
||||
if err != nil || fi.Size() == 0 {
|
||||
return fmt.Errorf("bundle assertion: HDD subdir %q is claimed by the manifest but its tar is missing or empty", s)
|
||||
}
|
||||
}
|
||||
// C6B-F1 cause 3 (v0.130.0): the claimed-tar checks above pass TRIVIALLY when discovery finds
|
||||
// nothing (0 claims → 0 checks) — exactly how a 4.17 GB app shipped as a 2308-byte config-only
|
||||
// bundle. A needs_hdd app with NO data of any kind is a hollow bundle by definition; refuse it
|
||||
// loudly so a future discovery gap can never again ship silently.
|
||||
if manifest.NeedsHDD && !manifest.HasHDDData && !manifest.HasVolumeData {
|
||||
return fmt.Errorf("a mentés nem tartalmaz alkalmazásadatot (0 adatkönyvtár, 0 kötet egy adattárolós alkalmazásnál)")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// createTarGz creates a gzipped tar archive of a directory.
|
||||
@@ -691,51 +891,10 @@ func createTarGz(outputPath, sourceDir string) error {
|
||||
})
|
||||
}
|
||||
|
||||
// tarDirectory creates a tar (not gzipped) of a directory's contents.
|
||||
// tarDirectory creates a tar (not gzipped) of a directory's contents. Thin wrapper over
|
||||
// tarDirectoryExcluding (Task 4) with no excludes — its existing callers are unchanged.
|
||||
func tarDirectory(sourceDir, outputPath string) error {
|
||||
outFile, err := os.Create(outputPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer outFile.Close()
|
||||
|
||||
tw := tar.NewWriter(outFile)
|
||||
defer tw.Close()
|
||||
|
||||
return filepath.Walk(sourceDir, func(path string, info os.FileInfo, err error) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
relPath, err := filepath.Rel(sourceDir, path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if relPath == "." {
|
||||
return nil
|
||||
}
|
||||
|
||||
header, err := tar.FileInfoHeader(info, "")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
header.Name = relPath
|
||||
|
||||
if err := tw.WriteHeader(header); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if info.IsDir() {
|
||||
return nil
|
||||
}
|
||||
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
_, err = io.Copy(tw, f)
|
||||
return err
|
||||
})
|
||||
return tarDirectoryExcluding(sourceDir, outputPath, nil)
|
||||
}
|
||||
|
||||
// gzipFile compresses a file with gzip.
|
||||
|
||||
@@ -0,0 +1,306 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"compress/gzip"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// C6B-F1 (v0.130.0) — the additive-export tests. A needs_hdd app bundles BOTH its HDD mounts
|
||||
// and its named volumes (the old either/or dropped every needs_hdd app's volumes); a basename
|
||||
// collision between mounts fails LOUDLY (the old code silently overwrote the first tar); the
|
||||
// HDD round-trip places a "userdata" tar back at <HDD_PATH>/userdata through the untouched
|
||||
// import mapping.
|
||||
|
||||
// listFabEntries returns the entry names inside an unencrypted .fab (tar.gz).
|
||||
func listFabEntries(t *testing.T, fabPath string) map[string]int64 {
|
||||
t.Helper()
|
||||
f, err := os.Open(fabPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
gz, err := gzip.NewReader(f)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
tr := tar.NewReader(gz)
|
||||
entries := map[string]int64{}
|
||||
for {
|
||||
hdr, err := tr.Next()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
entries[filepath.ToSlash(hdr.Name)] = hdr.Size
|
||||
}
|
||||
return entries
|
||||
}
|
||||
|
||||
func findFab(t *testing.T, drive string) string {
|
||||
t.Helper()
|
||||
entries, _ := os.ReadDir(ExportDir(drive))
|
||||
for _, en := range entries {
|
||||
if strings.HasSuffix(en.Name(), ".fab") {
|
||||
return filepath.Join(ExportDir(drive), en.Name())
|
||||
}
|
||||
}
|
||||
t.Fatal("no .fab produced")
|
||||
return ""
|
||||
}
|
||||
|
||||
// Scenario A (§7): the sonarr shape — a needs_hdd app with a populated userdata mount AND a
|
||||
// named volume. The bundle must contain BOTH tars; the manifest must claim both.
|
||||
// RED-PROOF (either/or): revert executeExport to the else-only volume branch → the volume tar
|
||||
// is absent and has_volume_data=false → this test fails.
|
||||
// RED-PROOF (discovery): with the pre-fix ${HDD_PATH}-only adapter the mount list is empty →
|
||||
// has_hdd_data=false → this test fails (proven at the adapter level in stacks/export_mounts_test.go).
|
||||
func TestExport_NeedsHDDBundlesBothUserdataAndVolumes(t *testing.T) {
|
||||
swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
switch c.args[0] {
|
||||
case "create":
|
||||
fmt.Fprint(stdout, "cid-123\n")
|
||||
return "", nil
|
||||
case "cp":
|
||||
// stream a plausible non-empty tar for the volume
|
||||
fmt.Fprint(stdout, strings.Repeat("VOLTAR", 100))
|
||||
return "", nil
|
||||
case "rm":
|
||||
return "", nil
|
||||
}
|
||||
return "", fmt.Errorf("unexpected docker call: %v", c.args)
|
||||
})
|
||||
|
||||
srcStack := t.TempDir()
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"),
|
||||
[]byte("services:\n hdd-app:\n image: alpine\n"), 0644)
|
||||
|
||||
hdd := t.TempDir()
|
||||
ud := filepath.Join(hdd, "userdata")
|
||||
os.MkdirAll(filepath.Join(ud, "media", "tv"), 0755)
|
||||
os.WriteFile(filepath.Join(ud, "media", "tv", "marker.bin"), []byte("USERDATA-MARKER-7"), 0644)
|
||||
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true,
|
||||
volumes: []string{"hdd-app_config"}},
|
||||
mounts: []string{ud}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
|
||||
if err := e.StartExport(ExportRequest{StackName: "hdd-app", DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
if msg := jobErr(job); msg != "" {
|
||||
t.Fatalf("export failed: %s", msg)
|
||||
}
|
||||
|
||||
fabPath := findFab(t, drive)
|
||||
man, err := ReadManifestFromFAB(fabPath)
|
||||
if err != nil {
|
||||
t.Fatalf("manifest: %v", err)
|
||||
}
|
||||
if !man.HasHDDData {
|
||||
t.Error("has_hdd_data=false — the userdata mount was dropped (C6B-F1 cause 2)")
|
||||
}
|
||||
if !man.HasVolumeData {
|
||||
t.Error("has_volume_data=false — the named volume was dropped (C6B-F1 cause 1, the either/or)")
|
||||
}
|
||||
|
||||
entries := listFabEntries(t, fabPath)
|
||||
if sz, ok := entries["data/hdd/userdata.tar"]; !ok || sz == 0 {
|
||||
t.Errorf("bundle is missing a non-empty data/hdd/userdata.tar (entries: %v)", entries)
|
||||
}
|
||||
if sz, ok := entries["data/volumes/hdd-app_config.tar"]; !ok || sz == 0 {
|
||||
t.Errorf("bundle is missing a non-empty data/volumes/hdd-app_config.tar (entries: %v)", entries)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario E (§7): a needs_hdd app whose volume export strands (cp writes nothing) must FAIL the
|
||||
// whole job loudly — never a partial-success bundle with userdata but silently-missing volumes.
|
||||
func TestExport_NeedsHDDVolumeStrandFailsLoud(t *testing.T) {
|
||||
swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
switch c.args[0] {
|
||||
case "create":
|
||||
fmt.Fprint(stdout, "cid-123\n")
|
||||
return "", nil
|
||||
case "cp":
|
||||
return "", nil // stranded: nothing written
|
||||
case "rm":
|
||||
return "", nil
|
||||
}
|
||||
return "", fmt.Errorf("unexpected docker call: %v", c.args)
|
||||
})
|
||||
|
||||
srcStack := t.TempDir()
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"),
|
||||
[]byte("services:\n hdd-app:\n image: alpine\n"), 0644)
|
||||
hdd := t.TempDir()
|
||||
ud := filepath.Join(hdd, "userdata")
|
||||
os.MkdirAll(ud, 0755)
|
||||
os.WriteFile(filepath.Join(ud, "f.bin"), []byte("x"), 0644)
|
||||
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true,
|
||||
volumes: []string{"vol1"}},
|
||||
mounts: []string{ud}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
if err := e.StartExport(ExportRequest{StackName: "hdd-app", DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
if msg := jobErr(job); msg == "" {
|
||||
t.Fatal("a stranded volume tar must FAIL a needs_hdd export too — got success")
|
||||
}
|
||||
entries, _ := os.ReadDir(ExportDir(drive))
|
||||
for _, en := range entries {
|
||||
if strings.HasSuffix(en.Name(), ".fab") {
|
||||
t.Fatalf("a bundle was produced despite the stranded volume: %s", en.Name())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// §8: two mounts sharing a basename cannot round-trip through the basename-keyed manifest —
|
||||
// the export must fail loudly instead of silently overwriting the first tar (the pre-fix
|
||||
// behavior). RED-PROOF: drop the collision check in exportHDDData → this test fails.
|
||||
func TestExport_HDDMountBasenameCollisionFailsLoud(t *testing.T) {
|
||||
srcStack := t.TempDir()
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"),
|
||||
[]byte("services:\n hdd-app:\n image: alpine\n"), 0644)
|
||||
|
||||
hdd := t.TempDir()
|
||||
a := filepath.Join(hdd, "a", "config")
|
||||
b := filepath.Join(hdd, "b", "config")
|
||||
os.MkdirAll(a, 0755)
|
||||
os.MkdirAll(b, 0755)
|
||||
os.WriteFile(filepath.Join(a, "one.txt"), []byte("A"), 0644)
|
||||
os.WriteFile(filepath.Join(b, "two.txt"), []byte("B"), 0644)
|
||||
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true},
|
||||
mounts: []string{a, b}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
if err := e.StartExport(ExportRequest{StackName: "hdd-app", DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
msg := jobErr(job)
|
||||
if msg == "" {
|
||||
t.Fatal("a basename collision must FAIL the export — got success (silent overwrite)")
|
||||
}
|
||||
if !strings.Contains(msg, "config") {
|
||||
t.Errorf("the error must name the colliding basename, got %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario A' — the round-trip placement proof: an exported "userdata" tar restores back to
|
||||
// <HDD_PATH>/userdata through the UNTOUCHED import mapping (basename → <HDD_PATH>/<subdir>
|
||||
// fallback). This is the property that dictated capturing the userdata ROOT rather than
|
||||
// per-bind subpaths.
|
||||
func TestFabRoundTrip_UserdataPlacement(t *testing.T) {
|
||||
const stack = "ud-app"
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
|
||||
hdd := t.TempDir()
|
||||
ud := filepath.Join(hdd, "userdata")
|
||||
os.MkdirAll(filepath.Join(ud, "media", "tv"), 0755)
|
||||
marker := "ROUNDTRIP-MARKER-99"
|
||||
os.WriteFile(filepath.Join(ud, "media", "tv", "show.bin"), []byte(marker), 0644)
|
||||
|
||||
srcStack := t.TempDir()
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"),
|
||||
[]byte("services:\n ud-app:\n image: alpine\n volumes:\n - ${USERDATA_PATH}/media/tv:/tv\n"), 0644)
|
||||
// app.yaml carries HDD_PATH into the bundle — the import derives every restore path from it.
|
||||
os.WriteFile(filepath.Join(srcStack, "app.yaml"),
|
||||
[]byte("deployed: true\nenv:\n HDD_PATH: "+hdd+"\n"), 0644)
|
||||
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true},
|
||||
mounts: []string{ud}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, lg, "test")
|
||||
if err := e.StartExport(ExportRequest{StackName: stack, DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
if msg := jobErr(job); msg != "" {
|
||||
t.Fatalf("export failed: %s", msg)
|
||||
}
|
||||
fabPath := findFab(t, drive)
|
||||
|
||||
// wipe the source userdata — the import must bring it back to the same place
|
||||
if err := os.RemoveAll(ud); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
prov2 := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: false},
|
||||
hddPath: hdd,
|
||||
}
|
||||
e2 := NewExporter(prov2, lg, "test")
|
||||
if err := e2.StartImport(ImportRequest{FABPath: fabPath}); err != nil {
|
||||
t.Fatalf("StartImport: %v", err)
|
||||
}
|
||||
job = waitJob(t, e2)
|
||||
if msg := jobErr(job); msg != "" {
|
||||
t.Fatalf("import failed: %s", msg)
|
||||
}
|
||||
|
||||
got, err := os.ReadFile(filepath.Join(hdd, "userdata", "media", "tv", "show.bin"))
|
||||
if err != nil {
|
||||
t.Fatalf("restored userdata not at <HDD_PATH>/userdata/media/tv/show.bin: %v", err)
|
||||
}
|
||||
if string(got) != marker {
|
||||
t.Fatalf("restored content differs: got %q want %q", got, marker)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario D (§7) — the anti-hollow net (C6B-F1 cause 3): a needs_hdd app where discovery finds
|
||||
// NOTHING (mounts absent on disk, no volumes) must FAIL the export with an honest Hungarian
|
||||
// error — never success-with-a-hollow-bundle. RED-PROOF: remove the needs_hdd&&no-data assertion
|
||||
// from assertBundleDataComplete → this test fails (the pre-fix silent hollow success).
|
||||
func TestExport_NeedsHDDNoDataAtAllRefused(t *testing.T) {
|
||||
srcStack := t.TempDir()
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"),
|
||||
[]byte("services:\n hdd-app:\n image: alpine\n"), 0644)
|
||||
|
||||
hdd := t.TempDir()
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true},
|
||||
// the mount does NOT exist on disk and there are no volumes — total discovery blank
|
||||
mounts: []string{filepath.Join(hdd, "userdata")}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
if err := e.StartExport(ExportRequest{StackName: "hdd-app", DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
msg := jobErr(job)
|
||||
if msg == "" {
|
||||
t.Fatal("a needs_hdd export with ZERO discovered data must FAIL — got success (the hollow bundle)")
|
||||
}
|
||||
if !strings.Contains(msg, "nem tartalmaz alkalmazásadatot") {
|
||||
t.Errorf("expected the honest Hungarian no-app-data error, got %q", msg)
|
||||
}
|
||||
entries, _ := os.ReadDir(ExportDir(drive))
|
||||
for _, en := range entries {
|
||||
if strings.HasSuffix(en.Name(), ".fab") {
|
||||
t.Fatalf("a hollow bundle was produced: %s", en.Name())
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
func fabWrite(t *testing.T, root, rel, content string) {
|
||||
t.Helper()
|
||||
p := filepath.Join(root, filepath.FromSlash(rel))
|
||||
if err := os.MkdirAll(filepath.Dir(p), 0755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(p, []byte(content), 0644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C at the export level: the userdata.tar keeps the mandatory subtree and NOT the siblings,
|
||||
// and the manifest still lists the single `userdata` basename (v1 unchanged). The bundle-level anchor
|
||||
// for the SQ6 fix (mirrors the §13 before/after).
|
||||
func TestFabExport_ExcludeScopedUserdataTar(t *testing.T) {
|
||||
drive := t.TempDir() // hddPath
|
||||
fabWrite(t, drive, "userdata/media/books/a.epub", "BOOK")
|
||||
fabWrite(t, drive, "userdata/media/movies/big.mkv", "MOVIE")
|
||||
fabWrite(t, drive, "userdata/music/s.flac", "SONG")
|
||||
|
||||
ud := appbackup.UserdataDir(filepath.Clean(drive))
|
||||
prov := &fabProv{
|
||||
rtProvider: &rtProvider{}, hddPath: drive, has: true,
|
||||
binds: []appbackup.ClassifiedBind{mUD("media/books"), xUD("media/movies")},
|
||||
mounts: []string{ud},
|
||||
}
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
|
||||
dataDir := t.TempDir()
|
||||
man := &Manifest{}
|
||||
if err := e.exportHDDData(ExportRequest{StackName: "calibre-web"}, dataDir, man); err != nil {
|
||||
t.Fatalf("exportHDDData: %v", err)
|
||||
}
|
||||
|
||||
entries := tarEntries(t, filepath.Join(dataDir, "hdd", "userdata.tar"))
|
||||
if !containsSuffix(entries, "media/books/a.epub") {
|
||||
t.Errorf("mandatory media/books missing from userdata.tar: %v", entries)
|
||||
}
|
||||
for _, sib := range []string{"media/movies/big.mkv", "media/movies", "music/s.flac", "music"} {
|
||||
if containsSuffix(entries, sib) {
|
||||
t.Errorf("sibling %q must NOT ride along (SQ6): %v", sib, entries)
|
||||
}
|
||||
}
|
||||
// v1 manifest: the single `userdata` basename, unchanged.
|
||||
if len(man.HDDSubdirs) != 1 || man.HDDSubdirs[0] != "userdata" {
|
||||
t.Errorf("manifest must list the single v1 `userdata` basename, got %v", man.HDDSubdirs)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario A at the export level: a legacy (no-block) app tars the FULL userdata root (every sibling)
|
||||
// — byte-identical to v0.130.0 (the SQ5 safety net).
|
||||
func TestFabExport_LegacyFullRoot(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
fabWrite(t, drive, "userdata/media/books/a.epub", "BOOK")
|
||||
fabWrite(t, drive, "userdata/media/movies/big.mkv", "MOVIE")
|
||||
|
||||
ud := appbackup.UserdataDir(filepath.Clean(drive))
|
||||
prov := &fabProv{rtProvider: &rtProvider{}, hddPath: drive, has: false, mounts: []string{ud}}
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
|
||||
dataDir := t.TempDir()
|
||||
man := &Manifest{}
|
||||
if err := e.exportHDDData(ExportRequest{StackName: "sonarr"}, dataDir, man); err != nil {
|
||||
t.Fatalf("exportHDDData: %v", err)
|
||||
}
|
||||
entries := tarEntries(t, filepath.Join(dataDir, "hdd", "userdata.tar"))
|
||||
for _, want := range []string{"media/books/a.epub", "media/movies/big.mkv"} {
|
||||
if !containsSuffix(entries, want) {
|
||||
t.Errorf("legacy app must capture the FULL root — %q missing: %v", want, entries)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B at the export level: an all-excluded app produces NO userdata.tar (root tar skipped).
|
||||
func TestFabExport_AllExcludedNoUserdataTar(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
fabWrite(t, drive, "userdata/media/movies/big.mkv", "MOVIE")
|
||||
ud := appbackup.UserdataDir(filepath.Clean(drive))
|
||||
prov := &fabProv{
|
||||
rtProvider: &rtProvider{}, hddPath: drive, has: true,
|
||||
binds: []appbackup.ClassifiedBind{xUD("media/movies")},
|
||||
mounts: []string{ud},
|
||||
}
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
dataDir := t.TempDir()
|
||||
man := &Manifest{}
|
||||
if err := e.exportHDDData(ExportRequest{StackName: "radarr"}, dataDir, man); err != nil {
|
||||
t.Fatalf("exportHDDData: %v", err)
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(dataDir, "hdd", "userdata.tar")); !os.IsNotExist(err) {
|
||||
t.Errorf("all-excluded app must produce NO userdata.tar (Scenario B), stat err=%v", err)
|
||||
}
|
||||
if len(man.HDDSubdirs) != 0 {
|
||||
t.Errorf("no userdata leg → no manifest subdir, got %v", man.HDDSubdirs)
|
||||
}
|
||||
}
|
||||
|
||||
// §7-F: EstimateExport populates the class split for a classified app (both web estimate pipelines
|
||||
// call this shared function, so both surface it). du returns 0 on the Windows test host, so this
|
||||
// asserts the STRUCTURE (HasClassification + item keys), not byte values.
|
||||
func TestEstimateExport_ClassifiedSplit(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
stackDir := t.TempDir()
|
||||
os.WriteFile(filepath.Join(stackDir, "docker-compose.yml"), []byte("services: {}\n"), 0644)
|
||||
prov := &fabProv{
|
||||
rtProvider: &rtProvider{stackDir: stackDir, deployed: true}, hddPath: drive, has: true,
|
||||
binds: []appbackup.ClassifiedBind{mUD("media/books"), oUD("media/comics"), xUD("media/movies")},
|
||||
}
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
est, err := e.EstimateExport("calibre-web", drive)
|
||||
if err != nil {
|
||||
t.Fatalf("EstimateExport: %v", err)
|
||||
}
|
||||
if !est.HasClassification {
|
||||
t.Fatal("classified app estimate must carry HasClassification")
|
||||
}
|
||||
if len(est.MandatoryItems) != 1 || est.MandatoryItems[0].Key != "userdata/media/books" {
|
||||
t.Errorf("MandatoryItems = %+v", est.MandatoryItems)
|
||||
}
|
||||
if len(est.OptionalItems) != 1 || est.OptionalItems[0].Key != "userdata/media/comics" {
|
||||
t.Errorf("OptionalItems = %+v", est.OptionalItems)
|
||||
}
|
||||
if len(est.ExcludedItems) != 1 || est.ExcludedItems[0].Key != "userdata/media/movies" {
|
||||
t.Errorf("ExcludedItems = %+v", est.ExcludedItems)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,311 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// `.fab` class-scoped export plan (Task 4, architecture §2 `.fab` row + SQ5 exclusion-scoping verdict,
|
||||
// R1-C). Mechanics unchanged from v0.130.0: ONE exclude-scoped userdata-root tar + per-mount skip for
|
||||
// non-selected HDD binds. The manifest stays v1 (basename keying) and the import side is untouched.
|
||||
|
||||
// fabPlan is the class-scoped adjustment to the v0.130.0 mount/tar set. Empty (all zero) = the legacy
|
||||
// full capture (Scenario A — a no-block app produces this).
|
||||
type fabPlan struct {
|
||||
SkipMounts map[string]bool // absolute HDD mount paths to skip entirely
|
||||
SkipUserdataTar bool // no selected userdata bind ⇒ the whole root tar is skipped (Scenario B)
|
||||
UserdataExcludeRels []string // rels (relative to the userdata root) excluded from its tar (R1-C, Scenario C)
|
||||
}
|
||||
|
||||
func relKey(root appbackup.BindRoot, rel string) string { return string(root) + "/" + rel }
|
||||
func relKeyOf(cp appbackup.CapturePath) string { return relKey(cp.Root, cp.RelPath) }
|
||||
|
||||
// computeFabPlan resolves the class buckets + the caller's selection into the mount/userdata plan
|
||||
// (§8). Legacy / no-block ⇒ empty plan. The mandatory floor is enforced here: DeselectOptional can
|
||||
// never drop a mandatory path.
|
||||
func (e *Exporter) computeFabPlan(req ExportRequest, mounts []string) fabPlan {
|
||||
binds, has := e.provider.GetStackClassifiedBinds(req.StackName)
|
||||
if !has {
|
||||
return fabPlan{} // legacy: byte-identical v0.130.0 capture
|
||||
}
|
||||
// R-203: the shared resolver's root parameter is a NAMESPACE ROOT — that is what the off-site
|
||||
// side has always passed (ComputeCaptureSet ← offbox_capture.go). This site passed the bare drive
|
||||
// path, so on the system-data fallback the export's classified paths and the backup's capture set
|
||||
// described DIFFERENT directories for the same declared bind. They now agree by construction.
|
||||
nsRoot := filepath.Clean(e.provider.GetStackNamespaceRoot(req.StackName))
|
||||
fb := appbackup.ComputeFabBuckets(binds, has, nsRoot, e.provider.GetImportRoot())
|
||||
|
||||
deselect := sliceSet(req.DeselectOptional)
|
||||
optIn := sliceSet(req.OptInExcluded)
|
||||
|
||||
// Server-side floor: a request naming a mandatory path in DeselectOptional is ignored (loud WARN).
|
||||
for _, cp := range fb.Mandatory {
|
||||
if deselect[relKeyOf(cp)] {
|
||||
e.logger.Printf("[WARN] appexport: %s: request tried to deselect a MANDATORY path %s — ignored (floor enforced)", req.StackName, relKeyOf(cp))
|
||||
}
|
||||
}
|
||||
|
||||
// Resolve the selected set (mandatory always; optional default-in; excluded default-out).
|
||||
var selectedHDD, selectedUD []appbackup.CapturePath
|
||||
add := func(cp appbackup.CapturePath) {
|
||||
if cp.Root == appbackup.RootUserdata {
|
||||
selectedUD = append(selectedUD, cp)
|
||||
} else {
|
||||
selectedHDD = append(selectedHDD, cp)
|
||||
}
|
||||
}
|
||||
for _, cp := range fb.Mandatory {
|
||||
add(cp)
|
||||
}
|
||||
for _, cp := range fb.Optional {
|
||||
if !deselect[relKeyOf(cp)] {
|
||||
add(cp)
|
||||
}
|
||||
}
|
||||
for _, cp := range fb.Excluded {
|
||||
if optIn[relKeyOf(cp)] {
|
||||
add(cp)
|
||||
}
|
||||
}
|
||||
|
||||
// Every classified HDD bind (any class) — a mount matching NONE of these is "unclassified" and kept
|
||||
// (fail toward capture, the C6B-F1 direction).
|
||||
var classifiedHDD []string
|
||||
for _, bucket := range [][]appbackup.CapturePath{fb.Mandatory, fb.Optional, fb.Excluded} {
|
||||
for _, cp := range bucket {
|
||||
if cp.Root == appbackup.RootHDD {
|
||||
classifiedHDD = append(classifiedHDD, cp.Abs)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
plan := fabPlan{SkipMounts: map[string]bool{}}
|
||||
// R-203: UserdataDir takes a NAMESPACE ROOT, not the drive path. Identical on an enrolled drive;
|
||||
// one segment short on the system-data fallback, which is where the export plan then skipped (or
|
||||
// failed to skip) the wrong directory.
|
||||
ud := appbackup.UserdataDir(nsRoot)
|
||||
for _, m := range mounts {
|
||||
mc := filepath.Clean(m)
|
||||
if mc == filepath.Clean(ud) {
|
||||
if len(selectedUD) == 0 {
|
||||
plan.SkipUserdataTar = true // Scenario B: no selected userdata bind → no root tar
|
||||
}
|
||||
continue
|
||||
}
|
||||
// HDD mount. Unmatched by ANY classified bind → keep (fail toward capture), log it.
|
||||
if !relatedToAny(mc, classifiedHDD) {
|
||||
e.logger.Printf("[INFO] appexport: %s: HDD mount %s matches no classified bind — kept (fail toward capture)", req.StackName, mc)
|
||||
continue
|
||||
}
|
||||
// Matched a classified bind: keep iff ancestor-or-descendant of a SELECTED HDD path.
|
||||
if !relatedToAny(mc, absList(selectedHDD)) {
|
||||
plan.SkipMounts[mc] = true
|
||||
}
|
||||
}
|
||||
|
||||
if !plan.SkipUserdataTar && len(selectedUD) > 0 {
|
||||
plan.UserdataExcludeRels = e.fabUserdataExcludes(ud, udRels(selectedUD))
|
||||
}
|
||||
return plan
|
||||
}
|
||||
|
||||
// fabEstimateSplit populates the class-split estimate fields for a classified app (Task 4). Legacy /
|
||||
// no-block apps leave HasClassification=false (the UI shows the plain estimate). BaseBytes = config +
|
||||
// volumes + mandatory; optional/excluded carry per-path sizes for client-side total recomputation.
|
||||
func (e *Exporter) fabEstimateSplit(stackName string, est *ExportEstimate, volumeBytes int64) {
|
||||
binds, has := e.provider.GetStackClassifiedBinds(stackName)
|
||||
if !has {
|
||||
return
|
||||
}
|
||||
nsRoot := filepath.Clean(e.provider.GetStackNamespaceRoot(stackName)) // R-203, as above
|
||||
fb := appbackup.ComputeFabBuckets(binds, has, nsRoot, e.provider.GetImportRoot())
|
||||
est.HasClassification = true
|
||||
|
||||
toItems := func(cps []appbackup.CapturePath) ([]FabItem, int64) {
|
||||
var items []FabItem
|
||||
var sum int64
|
||||
for _, cp := range cps {
|
||||
sz := duBytes(cp.Abs)
|
||||
items = append(items, FabItem{Key: relKeyOf(cp), Root: string(cp.Root), RelPath: cp.RelPath, Bytes: sz, Human: humanizeBytes(sz)})
|
||||
sum += sz
|
||||
}
|
||||
return items, sum
|
||||
}
|
||||
var mandSum int64
|
||||
est.MandatoryItems, mandSum = toItems(fb.Mandatory)
|
||||
est.OptionalItems, _ = toItems(fb.Optional)
|
||||
est.ExcludedItems, _ = toItems(fb.Excluded)
|
||||
est.BaseBytes = est.ConfigSizeBytes + volumeBytes + mandSum
|
||||
est.BaseHuman = humanizeBytes(est.BaseBytes)
|
||||
}
|
||||
|
||||
func sliceSet(ss []string) map[string]bool {
|
||||
m := make(map[string]bool, len(ss))
|
||||
for _, s := range ss {
|
||||
m[s] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func absList(cps []appbackup.CapturePath) []string {
|
||||
out := make([]string, len(cps))
|
||||
for i, cp := range cps {
|
||||
out[i] = cp.Abs
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// udRels returns the userdata rels (relative to ${USERDATA_PATH}) of selected userdata paths.
|
||||
func udRels(cps []appbackup.CapturePath) []string {
|
||||
out := make([]string, 0, len(cps))
|
||||
for _, cp := range cps {
|
||||
out = append(out, cp.RelPath)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// relatedToAny reports whether path p is an ancestor OR descendant (or equal) of any path in set.
|
||||
func relatedToAny(p string, set []string) bool {
|
||||
pc := filepath.Clean(p)
|
||||
for _, s := range set {
|
||||
sc := filepath.Clean(s)
|
||||
if pc == sc || strings.HasPrefix(pc, sc+string(filepath.Separator)) || strings.HasPrefix(sc, pc+string(filepath.Separator)) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// fabRelClass classifies a dir rel (slash-form, relative to the userdata root) against the selected
|
||||
// userdata rels — the R1-C keep-rule (mirrors backup.classifyTier2Rel, copied not imported):
|
||||
// keepInside = a selected rel or inside one (keep, don't descend); keepAncestor = on the path to a
|
||||
// selected rel (keep, descend); else stale (exclude the topmost).
|
||||
type fabRelClass int
|
||||
|
||||
const (
|
||||
fabStale fabRelClass = iota
|
||||
fabKeepInside
|
||||
fabKeepAncestor
|
||||
)
|
||||
|
||||
func classifyFabRel(dirRel string, selectedRels []string) fabRelClass {
|
||||
for _, sr := range selectedRels {
|
||||
if dirRel == sr || strings.HasPrefix(dirRel, sr+"/") {
|
||||
return fabKeepInside
|
||||
}
|
||||
}
|
||||
for _, sr := range selectedRels {
|
||||
if strings.HasPrefix(sr, dirRel+"/") {
|
||||
return fabKeepAncestor
|
||||
}
|
||||
}
|
||||
return fabStale
|
||||
}
|
||||
|
||||
// fabUserdataExcludes walks the userdata root (via the dirLister seam) and returns the topmost rels
|
||||
// (relative to the root, slash-form) that are neither an ancestor nor a descendant of a selected
|
||||
// userdata rel — the exclude list for the root tar (R1-C). Deterministic (sorted).
|
||||
func (e *Exporter) fabUserdataExcludes(udRoot string, selectedRels []string) []string {
|
||||
lister := e.dirLister
|
||||
if lister == nil {
|
||||
lister = realDirLister
|
||||
}
|
||||
var excludes []string
|
||||
var walk func(dirAbs, dirRel string)
|
||||
walk = func(dirAbs, dirRel string) {
|
||||
for _, name := range lister(dirAbs) {
|
||||
childRel := name
|
||||
if dirRel != "" {
|
||||
childRel = dirRel + "/" + name
|
||||
}
|
||||
switch classifyFabRel(childRel, selectedRels) {
|
||||
case fabKeepInside:
|
||||
// selected leg or content inside it — keep, no descent
|
||||
case fabKeepAncestor:
|
||||
walk(filepath.Join(dirAbs, name), childRel)
|
||||
default:
|
||||
excludes = append(excludes, childRel) // topmost neither-ancestor-nor-descendant
|
||||
}
|
||||
}
|
||||
}
|
||||
walk(udRoot, "")
|
||||
sort.Strings(excludes)
|
||||
return excludes
|
||||
}
|
||||
|
||||
func realDirLister(dir string) []string {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var names []string
|
||||
for _, en := range entries {
|
||||
if en.IsDir() {
|
||||
names = append(names, en.Name())
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// tarDirectoryExcluding is tarDirectory with an exclude list: any path whose rel (relative to
|
||||
// sourceDir, slash-form) equals or descends from an exclude rel is skipped (a dir is pruned whole).
|
||||
// Empty excludes == tarDirectory. Anchored to the tar root exactly like tarDirectory's rel names.
|
||||
func tarDirectoryExcluding(sourceDir, outputPath string, excludeRels []string) error {
|
||||
excl := make([]string, len(excludeRels))
|
||||
for i, r := range excludeRels {
|
||||
excl[i] = filepath.ToSlash(r)
|
||||
}
|
||||
outFile, err := os.Create(outputPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer outFile.Close()
|
||||
tw := tar.NewWriter(outFile)
|
||||
defer tw.Close()
|
||||
|
||||
return filepath.Walk(sourceDir, func(path string, info os.FileInfo, err error) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
relPath, err := filepath.Rel(sourceDir, path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if relPath == "." {
|
||||
return nil
|
||||
}
|
||||
rel := filepath.ToSlash(relPath)
|
||||
for _, e := range excl {
|
||||
if rel == e || strings.HasPrefix(rel, e+"/") {
|
||||
if info.IsDir() {
|
||||
return filepath.SkipDir
|
||||
}
|
||||
return nil
|
||||
}
|
||||
}
|
||||
header, err := tar.FileInfoHeader(info, "")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
header.Name = relPath
|
||||
if err := tw.WriteHeader(header); err != nil {
|
||||
return err
|
||||
}
|
||||
if info.IsDir() {
|
||||
return nil
|
||||
}
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
_, err = io.Copy(tw, f)
|
||||
return err
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,270 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"sort"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// fabProv is a minimal ExportStackProvider for the plan tests: configurable classified binds + hddPath.
|
||||
type fabProv struct {
|
||||
*rtProvider
|
||||
hddPath string
|
||||
binds []appbackup.ClassifiedBind
|
||||
has bool
|
||||
mounts []string
|
||||
}
|
||||
|
||||
func (p *fabProv) GetStackHDDPath(string) string { return p.hddPath }
|
||||
func (p *fabProv) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
|
||||
// R-203: these fixtures use ENROLLED drive paths, where the namespace root IS the drive path.
|
||||
// Delegating keeps that identity explicit rather than hardcoding it.
|
||||
func (p *fabProv) GetStackNamespaceRoot(name string) string { return p.GetStackHDDPath(name) }
|
||||
func (p *fabProv) GetStackHDDMounts(string) []string { return p.mounts }
|
||||
func (p *fabProv) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return p.binds, p.has
|
||||
}
|
||||
|
||||
func newFabExporter(binds []appbackup.ClassifiedBind, has bool, hddPath string, tree map[string][]string) *Exporter {
|
||||
e := NewExporter(&fabProv{rtProvider: &rtProvider{}, hddPath: hddPath, binds: binds, has: has}, log.New(io.Discard, "", 0), "test")
|
||||
e.dirLister = func(dir string) []string { return tree[dir] }
|
||||
return e
|
||||
}
|
||||
|
||||
func mHDD(rel string) appbackup.ClassifiedBind {
|
||||
return appbackup.ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootHDD, RelPath: rel}, Class: appbackup.ClassMandatory}
|
||||
}
|
||||
func mUD(rel string) appbackup.ClassifiedBind {
|
||||
return appbackup.ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootUserdata, RelPath: rel}, Class: appbackup.ClassMandatory}
|
||||
}
|
||||
func oUD(rel string) appbackup.ClassifiedBind {
|
||||
return appbackup.ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootUserdata, RelPath: rel}, Class: appbackup.ClassOptional}
|
||||
}
|
||||
func xUD(rel string) appbackup.ClassifiedBind {
|
||||
return appbackup.ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootUserdata, RelPath: rel}, Class: appbackup.ClassExcluded}
|
||||
}
|
||||
|
||||
const hp = "/srv/data" // hddPath (plan is string-only; the lister is injected)
|
||||
|
||||
// udDir builds an OS-native userdata dir path (matches appbackup.UserdataDir + the walk's
|
||||
// filepath.Join, so injected-lister keys line up on any host).
|
||||
func udDir(rel ...string) string {
|
||||
return filepath.Join(append([]string{hp, "userdata"}, rel...)...)
|
||||
}
|
||||
|
||||
// A — legacy app: empty plan (byte-identical v0.130.0 capture).
|
||||
func TestFabPlan_LegacyEmpty(t *testing.T) {
|
||||
e := newFabExporter(nil, false, hp, nil)
|
||||
mounts := []string{hp + "/appdata/app", hp + "/userdata"}
|
||||
plan := e.computeFabPlan(ExportRequest{StackName: "x"}, mounts)
|
||||
if len(plan.SkipMounts) != 0 || plan.SkipUserdataTar || plan.UserdataExcludeRels != nil {
|
||||
t.Errorf("legacy app must yield an EMPTY plan (every mount kept, no excludes), got %+v", plan)
|
||||
}
|
||||
}
|
||||
|
||||
// B — all-excluded app: the userdata root tar is skipped entirely.
|
||||
func TestFabPlan_AllExcludedSkipsUserdataTar(t *testing.T) {
|
||||
binds := []appbackup.ClassifiedBind{xUD("media/movies"), xUD("downloads")}
|
||||
e := newFabExporter(binds, true, hp, map[string][]string{"/srv/data/userdata": {"media", "downloads"}})
|
||||
plan := e.computeFabPlan(ExportRequest{StackName: "radarr"}, []string{hp + "/userdata"})
|
||||
if !plan.SkipUserdataTar {
|
||||
t.Error("no selected userdata bind → the whole root tar must be skipped (Scenario B)")
|
||||
}
|
||||
}
|
||||
|
||||
// C — selected-complement: media/books kept, siblings excluded (R1-C).
|
||||
func TestFabPlan_SelectedComplement(t *testing.T) {
|
||||
binds := []appbackup.ClassifiedBind{mUD("media/books"), xUD("media/movies")}
|
||||
tree := map[string][]string{
|
||||
udDir(): {"media", "music"},
|
||||
udDir("media"): {"books", "movies", "comics"},
|
||||
}
|
||||
e := newFabExporter(binds, true, hp, tree)
|
||||
plan := e.computeFabPlan(ExportRequest{StackName: "calibre-web"}, []string{hp + "/userdata"})
|
||||
if plan.SkipUserdataTar {
|
||||
t.Fatal("a selected userdata bind exists — the root tar must NOT be skipped")
|
||||
}
|
||||
want := []string{"media/comics", "media/movies", "music"}
|
||||
if got := plan.UserdataExcludeRels; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("excludes = %v, want %v (media/books kept, siblings excluded)", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// D — optional default-in / uncheck-out / excluded opt-in.
|
||||
func TestFabPlan_OptionalAndOptIn(t *testing.T) {
|
||||
// A mandatory anchor (data) keeps the root tar always produced, so comics/podcasts inclusion is
|
||||
// exercised via the EXCLUDE LIST (not the whole-tar skip).
|
||||
binds := []appbackup.ClassifiedBind{mUD("data"), oUD("media/comics"), xUD("media/podcasts")}
|
||||
tree := map[string][]string{
|
||||
udDir(): {"data", "media"},
|
||||
udDir("media"): {"comics", "podcasts", "junk"},
|
||||
}
|
||||
// D1 default: optional comics IN → not excluded; podcasts (excluded) + junk (unselected) excluded.
|
||||
e := newFabExporter(binds, true, hp, tree)
|
||||
p1 := e.computeFabPlan(ExportRequest{StackName: "komga"}, []string{hp + "/userdata"})
|
||||
if p1.SkipUserdataTar {
|
||||
t.Fatal("mandatory anchor selected — tar must be produced")
|
||||
}
|
||||
if effExcluded(p1.UserdataExcludeRels, "media/comics") {
|
||||
t.Error("D1: default → optional comics must be INCLUDED (not excluded)")
|
||||
}
|
||||
if !effExcluded(p1.UserdataExcludeRels, "media/podcasts") {
|
||||
t.Error("D1: excluded podcasts must be excluded by default")
|
||||
}
|
||||
// D2 uncheck the optional → excluded (effectively, via a topmost exclude covering it).
|
||||
p2 := e.computeFabPlan(ExportRequest{StackName: "komga", DeselectOptional: []string{"userdata/media/comics"}}, []string{hp + "/userdata"})
|
||||
if !effExcluded(p2.UserdataExcludeRels, "media/comics") {
|
||||
t.Errorf("D2: unchecked optional must be excluded, excludes=%v", p2.UserdataExcludeRels)
|
||||
}
|
||||
// D3 opt-in the excluded → included.
|
||||
p3 := e.computeFabPlan(ExportRequest{StackName: "komga", OptInExcluded: []string{"userdata/media/podcasts"}}, []string{hp + "/userdata"})
|
||||
if effExcluded(p3.UserdataExcludeRels, "media/podcasts") {
|
||||
t.Error("D3: opted-in excluded must be INCLUDED (not excluded)")
|
||||
}
|
||||
}
|
||||
|
||||
// effExcluded reports whether rel (or an ancestor of it) is in the topmost exclude list.
|
||||
func effExcluded(excludes []string, rel string) bool {
|
||||
for _, e := range excludes {
|
||||
if rel == e || len(rel) > len(e) && rel[:len(e)+1] == e+"/" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// D floor — a request deselecting a MANDATORY path is IGNORED (mandatory stays in).
|
||||
func TestFabPlan_MandatoryFloor(t *testing.T) {
|
||||
binds := []appbackup.ClassifiedBind{mUD("media/books")}
|
||||
tree := map[string][]string{udDir(): {"media"}, udDir("media"): {"books"}}
|
||||
e := newFabExporter(binds, true, hp, tree)
|
||||
// client tries to deselect the mandatory path — must be ignored (books NOT excluded).
|
||||
plan := e.computeFabPlan(ExportRequest{StackName: "x", DeselectOptional: []string{"userdata/media/books"}}, []string{hp + "/userdata"})
|
||||
if contains(plan.UserdataExcludeRels, "media/books") || plan.SkipUserdataTar {
|
||||
t.Errorf("mandatory floor breached — media/books must stay in the bundle; plan=%+v", plan)
|
||||
}
|
||||
}
|
||||
|
||||
// §8 — an HDD mount matching NO classified bind is KEPT (fail toward capture).
|
||||
func TestFabPlan_UnmatchedMountKept(t *testing.T) {
|
||||
binds := []appbackup.ClassifiedBind{mHDD("appdata/known")}
|
||||
e := newFabExporter(binds, true, hp, nil)
|
||||
mounts := []string{hp + "/appdata/known", hp + "/appdata/mystery"}
|
||||
plan := e.computeFabPlan(ExportRequest{StackName: "x"}, mounts)
|
||||
if plan.SkipMounts[filepath.Clean(hp+"/appdata/mystery")] {
|
||||
t.Error("an unmatched mount must be KEPT (fail toward capture, C6B-F1)")
|
||||
}
|
||||
if plan.SkipMounts[filepath.Clean(hp+"/appdata/known")] {
|
||||
t.Error("a mandatory-matched mount must be kept")
|
||||
}
|
||||
}
|
||||
|
||||
// §8 — a classified HDD mount that is NOT selected is skipped.
|
||||
func TestFabPlan_UnselectedHDDMountSkipped(t *testing.T) {
|
||||
binds := []appbackup.ClassifiedBind{xHDD("appdata/cache")}
|
||||
e := newFabExporter(binds, true, hp, nil)
|
||||
mounts := []string{hp + "/appdata/cache"}
|
||||
plan := e.computeFabPlan(ExportRequest{StackName: "x"}, mounts)
|
||||
if !plan.SkipMounts[filepath.Clean(hp+"/appdata/cache")] {
|
||||
t.Error("an excluded, un-opted-in HDD mount must be skipped")
|
||||
}
|
||||
}
|
||||
|
||||
func xHDD(rel string) appbackup.ClassifiedBind {
|
||||
return appbackup.ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootHDD, RelPath: rel}, Class: appbackup.ClassExcluded}
|
||||
}
|
||||
|
||||
// classifyFabRel keep-rule truth table (the R1-C core, pure).
|
||||
func TestClassifyFabRel(t *testing.T) {
|
||||
sel := []string{"media/books"}
|
||||
cases := []struct {
|
||||
rel string
|
||||
want fabRelClass
|
||||
}{
|
||||
{"media/books", fabKeepInside},
|
||||
{"media/books/covers", fabKeepInside},
|
||||
{"media", fabKeepAncestor},
|
||||
{"media/movies", fabStale},
|
||||
{"music", fabStale},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := classifyFabRel(c.rel, sel); got != c.want {
|
||||
t.Errorf("classifyFabRel(%q) = %d, want %d", c.rel, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// tarDirectoryExcluding FS-level: excluded subtrees are absent, kept content present.
|
||||
func TestTarDirectoryExcluding(t *testing.T) {
|
||||
src := t.TempDir()
|
||||
write := func(rel, content string) {
|
||||
p := filepath.Join(src, filepath.FromSlash(rel))
|
||||
os.MkdirAll(filepath.Dir(p), 0755)
|
||||
os.WriteFile(p, []byte(content), 0644)
|
||||
}
|
||||
write("media/books/a.epub", "BOOK")
|
||||
write("media/movies/big.mkv", "MOVIE")
|
||||
write("music/song.flac", "SONG")
|
||||
|
||||
out := filepath.Join(t.TempDir(), "userdata.tar")
|
||||
if err := tarDirectoryExcluding(src, out, []string{"media/movies", "music"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got := tarEntries(t, out)
|
||||
if !containsSuffix(got, "media/books/a.epub") {
|
||||
t.Errorf("kept content missing: %v", got)
|
||||
}
|
||||
for _, bad := range []string{"media/movies/big.mkv", "music/song.flac", "media/movies", "music"} {
|
||||
if containsSuffix(got, bad) {
|
||||
t.Errorf("excluded path %q present in tar: %v", bad, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func tarEntries(t *testing.T, tarPath string) []string {
|
||||
t.Helper()
|
||||
f, err := os.Open(tarPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
tr := tar.NewReader(f)
|
||||
var names []string
|
||||
for {
|
||||
h, err := tr.Next()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
names = append(names, filepath.ToSlash(h.Name))
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names
|
||||
}
|
||||
|
||||
func contains(ss []string, want string) bool {
|
||||
for _, s := range ss {
|
||||
if s == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
func containsSuffix(ss []string, suffix string) bool {
|
||||
for _, s := range ss {
|
||||
if s == suffix || filepath.ToSlash(s) == suffix {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -4,6 +4,8 @@
|
||||
// the app to its current state.
|
||||
package appexport
|
||||
|
||||
import "gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
|
||||
// ExportStackProvider provides stack data without circular imports.
|
||||
// Implemented by exportAdapter in main.go (same pattern as backup.StackDataProvider).
|
||||
type ExportStackProvider interface {
|
||||
@@ -15,6 +17,18 @@ type ExportStackProvider interface {
|
||||
GetStackHDDMounts(name string) []string
|
||||
// GetStackHDDPath returns the raw HDD_PATH env var from app.yaml.
|
||||
GetStackHDDPath(name string) string
|
||||
// GetImportRoot returns the CANONICAL drop-zone root (R-75), on the SYSTEM drive. ${IMPORT_PATH}
|
||||
// binds resolve against THIS, never against GetStackHDDPath. Empty when unresolvable.
|
||||
GetImportRoot() string
|
||||
// GetStackNamespaceRoot returns the app's felhom-data NAMESPACE ROOT — the directory that directly
|
||||
// contains backups/ and userdata/. It is NOT GetStackHDDPath: on an enrolled drive the two are the
|
||||
// same, and on the system-data fallback the namespace root has one more segment (R-203). Every
|
||||
// appbackup path helper takes THIS, never the drive path. Empty when the app has no HDD_PATH.
|
||||
GetStackNamespaceRoot(name string) string
|
||||
// GetStackClassifiedBinds returns the app's backup-classified compose binds + whether it carries a
|
||||
// (valid) backup block (Task 2). Drives the `.fab` class-scoped export plan (Task 4); a legacy app
|
||||
// (false) exports the v0.130.0 full-root capture unchanged.
|
||||
GetStackClassifiedBinds(name string) ([]appbackup.ClassifiedBind, bool)
|
||||
// IsStackRunning returns true if the stack has running containers.
|
||||
IsStackRunning(name string) bool
|
||||
// StopStack stops the stack via docker compose down.
|
||||
|
||||
@@ -326,6 +326,16 @@ func (e *Exporter) executeImport(req ImportRequest, job *Job) {
|
||||
job.mu.Unlock()
|
||||
|
||||
e.logger.Printf("[INFO] Import: opening bundle for %s (%s)", manifest.AppName, manifest.DisplayName)
|
||||
|
||||
// v0.125.0 validate-before-destroy (scenario C): every manifest-claimed data tar must be
|
||||
// present and non-empty BEFORE the app is stopped and BEFORE any volume is removed. Hollow
|
||||
// bundles (a containerized v<=0.124.0 exporter stranded the tars host-side) land HERE — the
|
||||
// pre-fix order wiped the volumes first and only then discovered the emptiness.
|
||||
if err := validateBundleData(tmpDir, manifest); err != nil {
|
||||
e.failJob(job, step, fmt.Sprintf("A csomag hiányos — az importálás el sem indult, a meglévő alkalmazás érintetlen. (%v) A csomagot valószínűleg egy régebbi (≤0.124.0), konténerben futó vezérlő exportálta — készíts friss exportot.", err))
|
||||
return
|
||||
}
|
||||
|
||||
e.debugf("step 0 (open bundle) done in %v", time.Since(stepStart))
|
||||
|
||||
job.setStep(step, "done", "")
|
||||
@@ -597,9 +607,9 @@ func (e *Exporter) restoreHDDData(tmpDir string, manifest *Manifest, composePath
|
||||
tarPath := filepath.Join(hddDir, subdir+".tar")
|
||||
tarInfo, err := os.Stat(tarPath)
|
||||
if err != nil {
|
||||
e.logger.Printf("[WARN] Import: HDD tar not found: %s", tarPath)
|
||||
e.debugf("restoreHDDData: tar not found: %s", tarPath)
|
||||
continue
|
||||
// v0.125.0: a claimed-but-absent tar is an ERROR (validateBundleData already refused
|
||||
// it pre-destroy; this is defense in depth, not a reachable soft path).
|
||||
return fmt.Errorf("HDD tar missing from bundle: %s", subdir+".tar")
|
||||
}
|
||||
e.debugf("restoreHDDData: subdir=%s tarSize=%s", subdir, humanizeBytes(tarInfo.Size()))
|
||||
|
||||
@@ -670,7 +680,52 @@ func resolveHDDMounts(composePath string, env map[string]string) []string {
|
||||
return mounts
|
||||
}
|
||||
|
||||
// restoreVolumeData recreates Docker named volumes from bundle tarballs.
|
||||
// validateBundleData asserts every manifest-claimed data tar exists non-empty in the extracted
|
||||
// bundle (v0.125.0, scenario C). Pure read — called before ANY destructive import step.
|
||||
func validateBundleData(tmpDir string, manifest *Manifest) error {
|
||||
for _, v := range manifest.VolumeNames {
|
||||
if err := ValidateSegment("volume_name", v); err != nil {
|
||||
return err
|
||||
}
|
||||
fi, err := os.Stat(filepath.Join(tmpDir, "data", "volumes", v+".tar"))
|
||||
if err != nil || fi.Size() == 0 {
|
||||
return fmt.Errorf("a(z) %q kötet adata hiányzik a csomagból", v)
|
||||
}
|
||||
}
|
||||
for _, s := range manifest.HDDSubdirs {
|
||||
if err := ValidateSegment("hdd_subdir", s); err != nil {
|
||||
return err
|
||||
}
|
||||
fi, err := os.Stat(filepath.Join(tmpDir, "data", "hdd", s+".tar"))
|
||||
if err != nil || fi.Size() == 0 {
|
||||
return fmt.Errorf("a(z) %q adatkönyvtár tartalma hiányzik a csomagból", s)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// importVolumeTar streams tarPath into the (existing) volume via `docker cp - <cid>:/vol` —
|
||||
// zero shared paths, correct on bare metal AND under the containerized controller (the
|
||||
// docker-run -v population was the v0.124.0 strand's import half).
|
||||
func (e *Exporter) importVolumeTar(volName, tarPath string) error {
|
||||
return e.withVolumeHelper(volName, func(cid string) error {
|
||||
f, err := os.Open(tarPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("opening %s: %w", tarPath, err)
|
||||
}
|
||||
defer f.Close()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
||||
defer cancel()
|
||||
if stderr, err := dockerExec(ctx, f, nil, "cp", "-", cid+":/vol"); err != nil {
|
||||
return fmt.Errorf("streaming into volume %s: %s — %w", volName, stderr, err)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
}
|
||||
|
||||
// restoreVolumeData recreates Docker named volumes from bundle tarballs. v0.125.0: a missing
|
||||
// tar is an ERROR (defense in depth behind validateBundleData), and population streams via
|
||||
// docker cp instead of a docker-run -v host mount.
|
||||
func (e *Exporter) restoreVolumeData(tmpDir string, manifest *Manifest) error {
|
||||
volDir := filepath.Join(tmpDir, "data", "volumes")
|
||||
|
||||
@@ -683,34 +738,24 @@ func (e *Exporter) restoreVolumeData(tmpDir string, manifest *Manifest) error {
|
||||
tarPath := filepath.Join(volDir, volName+".tar")
|
||||
tarInfo, err := os.Stat(tarPath)
|
||||
if err != nil {
|
||||
e.logger.Printf("[WARN] Import: volume tar not found: %s", tarPath)
|
||||
e.debugf("restoreVolumeData: tar not found: %s", tarPath)
|
||||
continue
|
||||
return fmt.Errorf("volume tar missing from bundle: %s", volName+".tar")
|
||||
}
|
||||
e.debugf("restoreVolumeData: volume=%s tarSize=%s", volName, humanizeBytes(tarInfo.Size()))
|
||||
|
||||
// Create the Docker volume
|
||||
e.debugf("restoreVolumeData: creating docker volume %s", volName)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
out, err := exec.CommandContext(ctx, "docker", "volume", "create", volName).CombinedOutput()
|
||||
stderr, err := dockerExec(ctx, nil, nil, "volume", "create", volName)
|
||||
cancel()
|
||||
if err != nil {
|
||||
return fmt.Errorf("creating volume %s: %s — %w", volName, strings.TrimSpace(string(out)), err)
|
||||
return fmt.Errorf("creating volume %s: %s — %w", volName, stderr, err)
|
||||
}
|
||||
e.debugf("restoreVolumeData: volume %s created: %s", volName, strings.TrimSpace(string(out)))
|
||||
|
||||
// Populate volume from tar
|
||||
// Populate volume from tar (docker cp streaming)
|
||||
e.logger.Printf("[INFO] Import: populating volume %s", volName)
|
||||
e.debugf("restoreVolumeData: populating %s via docker run alpine tar xf...", volName)
|
||||
popStart := time.Now()
|
||||
ctx, cancel = context.WithTimeout(context.Background(), 10*time.Minute)
|
||||
out, err = exec.CommandContext(ctx, "docker", "run", "--rm",
|
||||
"-v", volName+":/vol",
|
||||
"-v", volDir+":/in:ro",
|
||||
"alpine", "tar", "xf", "/in/"+volName+".tar", "-C", "/vol").CombinedOutput()
|
||||
cancel()
|
||||
if err != nil {
|
||||
return fmt.Errorf("populating volume %s: %s — %w", volName, strings.TrimSpace(string(out)), err)
|
||||
if err := e.importVolumeTar(volName, tarPath); err != nil {
|
||||
return fmt.Errorf("populating volume %s: %w", volName, err)
|
||||
}
|
||||
e.debugf("restoreVolumeData: volume %s populated in %v", volName, time.Since(popStart))
|
||||
}
|
||||
|
||||
@@ -0,0 +1,190 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// v0.124.0 Part 3 (§7D) — the .fab loop's proof at unit level: a real export through
|
||||
// executeExport produces a bundle that a real executeImport restores to identical content, and
|
||||
// a corrupted downloaded copy is REFUSED (the .fab's integrity is the gzip CRC + the manifest
|
||||
// segment validation — there is no per-file checksum; corruption breaks the extract, and the
|
||||
// import must fail loudly, not restore garbage).
|
||||
|
||||
// rtProvider is a filesystem-only fake: a config-only app (no HDD, no volumes, no DB) so the
|
||||
// whole loop runs without docker.
|
||||
type rtProvider struct {
|
||||
stackDir string
|
||||
stacksDir string
|
||||
deployed bool
|
||||
running bool
|
||||
volumes []string
|
||||
started bool
|
||||
stopped int
|
||||
removed int
|
||||
savedEnv map[string]string
|
||||
}
|
||||
|
||||
func (p *rtProvider) GetStackDir(string) (string, bool) { return p.stackDir, true }
|
||||
func (p *rtProvider) GetStackComposePath(string) (string, bool) {
|
||||
return filepath.Join(p.stackDir, "docker-compose.yml"), true
|
||||
}
|
||||
func (p *rtProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *rtProvider) GetStackHDDPath(string) string { return "" }
|
||||
func (p *rtProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
|
||||
// R-203: these fixtures use ENROLLED drive paths, where the namespace root IS the drive path.
|
||||
// Delegating keeps that identity explicit rather than hardcoding it.
|
||||
func (p *rtProvider) GetStackNamespaceRoot(name string) string { return p.GetStackHDDPath(name) }
|
||||
func (p *rtProvider) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
func (p *rtProvider) IsStackRunning(string) bool { return p.running }
|
||||
func (p *rtProvider) StopStack(string) error { p.stopped++; return nil }
|
||||
func (p *rtProvider) StartStack(string) error { p.started = true; return nil }
|
||||
func (p *rtProvider) GetStackDisplayName(n string) string { return "RT " + n }
|
||||
func (p *rtProvider) GetStackNeedsHDD(string) bool { return false }
|
||||
func (p *rtProvider) GetDockerVolumes(string) []string { return p.volumes }
|
||||
func (p *rtProvider) IsStackDeployed(string) bool { return p.deployed }
|
||||
func (p *rtProvider) GetDecryptedEnv(string) map[string]string { return nil }
|
||||
func (p *rtProvider) GetStacksBaseDir() string { return p.stacksDir }
|
||||
func (p *rtProvider) RefreshStacks() error { return nil }
|
||||
func (p *rtProvider) RemoveStackVolumes(string) error { p.removed++; return nil }
|
||||
func (p *rtProvider) SaveEncryptedAppConfig(stackDir string, env map[string]string) error {
|
||||
p.savedEnv = env
|
||||
return nil
|
||||
}
|
||||
|
||||
func waitJob(t *testing.T, e *Exporter) *Job {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(30 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
job := e.GetActiveJob()
|
||||
if job != nil {
|
||||
job.mu.RLock()
|
||||
done := job.Done
|
||||
job.mu.RUnlock()
|
||||
if done {
|
||||
return job
|
||||
}
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("job did not finish in time")
|
||||
return nil
|
||||
}
|
||||
|
||||
func jobErr(j *Job) string {
|
||||
j.mu.RLock()
|
||||
defer j.mu.RUnlock()
|
||||
if j.Error != "" {
|
||||
return j.Error
|
||||
}
|
||||
for _, s := range j.Steps {
|
||||
if s.Status == "failed" {
|
||||
return s.Error
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func TestFabRoundTrip_ExportImportContentEquality(t *testing.T) {
|
||||
const stack = "rt-app"
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
|
||||
// Source stack: a compose file + a marker config with known content.
|
||||
srcStack := t.TempDir()
|
||||
compose := "services:\n rt-app:\n image: alpine\n"
|
||||
marker := "MARKER-CONTENT-42\n"
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"), []byte(compose), 0644)
|
||||
os.WriteFile(filepath.Join(srcStack, "settings.conf"), []byte(marker), 0644)
|
||||
|
||||
drive := t.TempDir()
|
||||
prov := &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true}
|
||||
e := NewExporter(prov, lg, "test")
|
||||
|
||||
// --- export (the REAL pipeline; same producer as a drive export) ---
|
||||
if err := e.StartExport(ExportRequest{StackName: stack, DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
if msg := jobErr(job); msg != "" {
|
||||
t.Fatalf("export failed: %s", msg)
|
||||
}
|
||||
entries, _ := os.ReadDir(ExportDir(drive))
|
||||
var fabPath string
|
||||
for _, en := range entries {
|
||||
if strings.HasSuffix(en.Name(), ".fab") {
|
||||
fabPath = filepath.Join(ExportDir(drive), en.Name())
|
||||
}
|
||||
}
|
||||
if fabPath == "" {
|
||||
t.Fatal("no .fab produced")
|
||||
}
|
||||
|
||||
// The manifest is readable and names the app (what /api/export/manifest shows pre-import).
|
||||
man, err := ReadManifestFromFAB(fabPath)
|
||||
if err != nil {
|
||||
t.Fatalf("manifest: %v", err)
|
||||
}
|
||||
if man.AppName != stack {
|
||||
t.Fatalf("manifest app = %q", man.AppName)
|
||||
}
|
||||
|
||||
// --- corrupted copy must be REFUSED (assert the refusal, §10 red-proof of the loop) ---
|
||||
corrupt := filepath.Join(t.TempDir(), "corrupt.fab")
|
||||
raw, _ := os.ReadFile(fabPath)
|
||||
mid := len(raw) / 2
|
||||
raw[mid] ^= 0xFF
|
||||
raw[mid+1] ^= 0xFF
|
||||
os.WriteFile(corrupt, raw, 0644)
|
||||
|
||||
prov2 := &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: false}
|
||||
e2 := NewExporter(prov2, lg, "test")
|
||||
if err := e2.StartImport(ImportRequest{FABPath: corrupt}); err != nil {
|
||||
t.Fatalf("StartImport(corrupt) should start (refusal is async): %v", err)
|
||||
}
|
||||
job = waitJob(t, e2)
|
||||
if msg := jobErr(job); msg == "" {
|
||||
t.Fatal("a corrupted bundle must FAIL the import (gzip CRC), got success")
|
||||
}
|
||||
if prov2.started {
|
||||
t.Fatal("a refused import must not start the app")
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(prov2.stacksDir, stack, "settings.conf")); !os.IsNotExist(err) {
|
||||
t.Fatal("a refused import must not restore content")
|
||||
}
|
||||
|
||||
// --- the clean bundle round-trips: restored content is byte-identical ---
|
||||
prov3 := &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: false}
|
||||
e3 := NewExporter(prov3, lg, "test")
|
||||
if err := e3.StartImport(ImportRequest{FABPath: fabPath}); err != nil {
|
||||
t.Fatalf("StartImport: %v", err)
|
||||
}
|
||||
job = waitJob(t, e3)
|
||||
if msg := jobErr(job); msg != "" {
|
||||
t.Fatalf("import failed: %s", msg)
|
||||
}
|
||||
restoredStack := filepath.Join(prov3.stacksDir, stack)
|
||||
for name, want := range map[string]string{
|
||||
"docker-compose.yml": compose,
|
||||
"settings.conf": marker,
|
||||
} {
|
||||
got, err := os.ReadFile(filepath.Join(restoredStack, name))
|
||||
if err != nil {
|
||||
t.Fatalf("restored %s missing: %v", name, err)
|
||||
}
|
||||
if string(got) != want {
|
||||
t.Errorf("restored %s differs:\n got %q\nwant %q", name, got, want)
|
||||
}
|
||||
}
|
||||
if !prov3.started {
|
||||
t.Error("import must start the restored app")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,234 @@
|
||||
package appexport
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// v0.125.0 — the containerized-.fab volume-strand fix (IA finding 1, HIGH). These tests pin:
|
||||
// (B) export can no longer lie — a stranded/empty tar aborts the export, no bundle;
|
||||
// (C) import validates BEFORE it destroys — a hollow bundle is refused with the app untouched;
|
||||
// the docker-cp command construction (image, mount shape, cp direction) and the always-remove
|
||||
// helper-container rule. The real docker legs are covered by the §3 live probe + §13 round-trip.
|
||||
|
||||
// dockerCall records one dockerExec invocation.
|
||||
type dockerCall struct {
|
||||
args []string
|
||||
stdin bool
|
||||
}
|
||||
|
||||
// swapDockerExec installs a scripted fake for the package seam and restores it on cleanup.
|
||||
func swapDockerExec(t *testing.T, fn func(call dockerCall, stdin io.Reader, stdout io.Writer) (string, error)) *[]dockerCall {
|
||||
t.Helper()
|
||||
var mu sync.Mutex
|
||||
calls := &[]dockerCall{}
|
||||
orig := dockerExec
|
||||
dockerExec = func(ctx context.Context, stdin io.Reader, stdout io.Writer, args ...string) (string, error) {
|
||||
mu.Lock()
|
||||
c := dockerCall{args: append([]string{}, args...), stdin: stdin != nil}
|
||||
*calls = append(*calls, c)
|
||||
mu.Unlock()
|
||||
return fn(c, stdin, stdout)
|
||||
}
|
||||
t.Cleanup(func() { dockerExec = orig })
|
||||
return calls
|
||||
}
|
||||
|
||||
func volTestExporter(t *testing.T, volumes []string) (*Exporter, *rtProvider, string) {
|
||||
t.Helper()
|
||||
srcStack := t.TempDir()
|
||||
os.WriteFile(filepath.Join(srcStack, "docker-compose.yml"), []byte("services:\n vol-app:\n image: alpine\n"), 0644)
|
||||
prov := &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true, volumes: volumes}
|
||||
drive := t.TempDir()
|
||||
return NewExporter(prov, log.New(io.Discard, "", 0), "test"), prov, drive
|
||||
}
|
||||
|
||||
// Scenario B: a volume tar that fails to materialize (cp "succeeds" but writes nothing — the
|
||||
// strand's signature) must FAIL the export with the volume named, and NO bundle may exist.
|
||||
// Red-proof: remove the assertBundleDataComplete call → this test fails (hollow success).
|
||||
func TestExport_HollowVolumeTarAbortsExport(t *testing.T) {
|
||||
swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
switch c.args[0] {
|
||||
case "create":
|
||||
fmt.Fprint(stdout, "cid-123\n")
|
||||
return "", nil
|
||||
case "cp":
|
||||
return "", nil // writes NOTHING to stdout — the stranded-tar signature
|
||||
case "rm":
|
||||
return "", nil
|
||||
}
|
||||
return "", fmt.Errorf("unexpected docker call: %v", c.args)
|
||||
})
|
||||
|
||||
e, _, drive := volTestExporter(t, []string{"vol1"})
|
||||
if err := e.StartExport(ExportRequest{StackName: "vol-app", DestDrive: drive}); err != nil {
|
||||
t.Fatalf("StartExport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
msg := jobErr(job)
|
||||
if msg == "" {
|
||||
t.Fatal("a hollow volume tar must FAIL the export — got success")
|
||||
}
|
||||
if !strings.Contains(msg, "vol1") {
|
||||
t.Errorf("the error must NAME the missing volume, got %q", msg)
|
||||
}
|
||||
entries, _ := os.ReadDir(ExportDir(drive))
|
||||
for _, en := range entries {
|
||||
if strings.HasSuffix(en.Name(), ".fab") {
|
||||
t.Fatalf("a bundle was produced despite the hollow tar: %s", en.Name())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C: a bundle whose manifest claims a volume without its tar is refused BEFORE any
|
||||
// destructive step — the app is not stopped, no volume is removed or recreated, zero docker
|
||||
// calls happen. Red-proof: disable the pre-flight (pre-fix order: wipe first, discover later)
|
||||
// → the zero-destruction assertions fail.
|
||||
func TestImport_HollowBundleRefusedBeforeDestroy(t *testing.T) {
|
||||
calls := swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
return "", nil
|
||||
})
|
||||
|
||||
// Handcraft a hollow bundle: manifest CLAIMS volume data, data/volumes is empty.
|
||||
tree := t.TempDir()
|
||||
os.MkdirAll(filepath.Join(tree, "config"), 0755)
|
||||
os.MkdirAll(filepath.Join(tree, "data", "volumes"), 0755)
|
||||
os.WriteFile(filepath.Join(tree, "config", "docker-compose.yml"), []byte("services: {}\n"), 0644)
|
||||
man := &Manifest{
|
||||
Version: ManifestVersion, AppName: "vol-app", DisplayName: "Vol App",
|
||||
HasVolumeData: true, VolumeNames: []string{"vol1"},
|
||||
ConfigFiles: []string{"docker-compose.yml"},
|
||||
}
|
||||
data, err := man.Marshal()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
os.WriteFile(filepath.Join(tree, "manifest.json"), data, 0644)
|
||||
fab := filepath.Join(t.TempDir(), "hollow.fab")
|
||||
if err := createTarGz(fab, tree); err != nil {
|
||||
t.Fatalf("createTarGz: %v", err)
|
||||
}
|
||||
|
||||
prov := &rtProvider{stackDir: t.TempDir(), stacksDir: t.TempDir(), deployed: true, running: true}
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
if err := e.StartImport(ImportRequest{FABPath: fab}); err != nil {
|
||||
t.Fatalf("StartImport: %v", err)
|
||||
}
|
||||
job := waitJob(t, e)
|
||||
msg := jobErr(job)
|
||||
if msg == "" {
|
||||
t.Fatal("a hollow bundle must be REFUSED — got success")
|
||||
}
|
||||
if !strings.Contains(msg, "érintetlen") {
|
||||
t.Errorf("refusal copy must state the app is untouched, got %q", msg)
|
||||
}
|
||||
// THE exact non-effects: nothing was stopped, wiped, recreated or started.
|
||||
if prov.stopped != 0 || prov.removed != 0 {
|
||||
t.Fatalf("refusal happened AFTER destruction: stopped=%d removedVolumes=%d", prov.stopped, prov.removed)
|
||||
}
|
||||
if prov.started {
|
||||
t.Fatal("a refused import must not start the app")
|
||||
}
|
||||
if len(*calls) != 0 {
|
||||
t.Fatalf("a refused import must make ZERO docker calls, got %v", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// Command construction + helper hygiene: the export leg uses create/cp/rm with the exact arg
|
||||
// shapes the §3 probe validated, and the helper container is force-removed EVEN when cp fails.
|
||||
func TestExportVolumeTar_CommandShapesAndHelperCleanup(t *testing.T) {
|
||||
t.Run("happy path shapes", func(t *testing.T) {
|
||||
calls := swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
switch c.args[0] {
|
||||
case "create":
|
||||
fmt.Fprint(stdout, "cid-abc\n")
|
||||
case "cp":
|
||||
fmt.Fprint(stdout, "TARBYTES")
|
||||
}
|
||||
return "", nil
|
||||
})
|
||||
e, _, _ := volTestExporter(t, nil)
|
||||
tarPath := filepath.Join(t.TempDir(), "v.tar")
|
||||
if err := e.exportVolumeTar("vol1", tarPath); err != nil {
|
||||
t.Fatalf("exportVolumeTar: %v", err)
|
||||
}
|
||||
got, _ := os.ReadFile(tarPath)
|
||||
if string(got) != "TARBYTES" {
|
||||
t.Fatalf("tar content = %q", got)
|
||||
}
|
||||
want := [][]string{
|
||||
{"create", "-v", "vol1:/vol", "alpine", "true"},
|
||||
{"cp", "cid-abc:/vol/.", "-"},
|
||||
{"rm", "-f", "cid-abc"},
|
||||
}
|
||||
assertCalls(t, *calls, want)
|
||||
})
|
||||
|
||||
t.Run("helper removed on cp failure", func(t *testing.T) {
|
||||
calls := swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
switch c.args[0] {
|
||||
case "create":
|
||||
fmt.Fprint(stdout, "cid-err\n")
|
||||
return "", nil
|
||||
case "cp":
|
||||
return "boom", fmt.Errorf("cp failed")
|
||||
}
|
||||
return "", nil
|
||||
})
|
||||
e, _, _ := volTestExporter(t, nil)
|
||||
tarPath := filepath.Join(t.TempDir(), "v.tar")
|
||||
if err := e.exportVolumeTar("vol1", tarPath); err == nil {
|
||||
t.Fatal("cp failure must surface")
|
||||
}
|
||||
if _, err := os.Stat(tarPath); !os.IsNotExist(err) {
|
||||
t.Error("a failed export must not leave a partial tar")
|
||||
}
|
||||
last := (*calls)[len(*calls)-1]
|
||||
if strings.Join(last.args, " ") != "rm -f cid-err" {
|
||||
t.Fatalf("helper container must be force-removed on failure, last call: %v", last.args)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("import leg shapes", func(t *testing.T) {
|
||||
calls := swapDockerExec(t, func(c dockerCall, stdin io.Reader, stdout io.Writer) (string, error) {
|
||||
if c.args[0] == "create" {
|
||||
fmt.Fprint(stdout, "cid-imp\n")
|
||||
}
|
||||
if c.args[0] == "cp" && !c.stdin {
|
||||
t.Error("import cp must stream the tar on stdin")
|
||||
}
|
||||
return "", nil
|
||||
})
|
||||
e, _, _ := volTestExporter(t, nil)
|
||||
tarPath := filepath.Join(t.TempDir(), "v.tar")
|
||||
os.WriteFile(tarPath, []byte("TAR"), 0644)
|
||||
if err := e.importVolumeTar("vol1", tarPath); err != nil {
|
||||
t.Fatalf("importVolumeTar: %v", err)
|
||||
}
|
||||
want := [][]string{
|
||||
{"create", "-v", "vol1:/vol", "alpine", "true"},
|
||||
{"cp", "-", "cid-imp:/vol"},
|
||||
{"rm", "-f", "cid-imp"},
|
||||
}
|
||||
assertCalls(t, *calls, want)
|
||||
})
|
||||
}
|
||||
|
||||
func assertCalls(t *testing.T, got []dockerCall, want [][]string) {
|
||||
t.Helper()
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("docker calls = %d, want %d (%v)", len(got), len(want), got)
|
||||
}
|
||||
for i := range want {
|
||||
if strings.Join(got[i].args, " ") != strings.Join(want[i], " ") {
|
||||
t.Errorf("call %d = %v, want %v", i, got[i].args, want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,252 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ── Backup admission (R-181) ─────────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// WHAT WAS WRONG. B2's capture floor (v0.192.0, R-165) shipped as the deliberate replacement for the
|
||||
// bulkhead the `mp1` partition used to give, and it was consulted in exactly ONE place —
|
||||
// `captureAllRecoveryUnits`, which writes a manifest and three compose files: a few KB. The two legs
|
||||
// that write the BULK into the same `backups/primary/<app>` tree — the database dump and the volume
|
||||
// dump — ran FIRST and unguarded. Measured on demo-hp 2026-08-03 06:40:03: opengist's volume dump
|
||||
// wrote 2.0 GB with no check, free fell to 1.0 GB, and the floor then refused the cheap write it had
|
||||
// already lost the argument to. Its refusal message said *"the previous unit is untouched"*, which
|
||||
// was false by then — that app's tar had gone 182,272 B → 2,147,666,432 B under a stale manifest.
|
||||
//
|
||||
// WHAT THIS IS. ONE verdict per app per run, taken before that app's FIRST write of the run, covering
|
||||
// all three legs. The three write under one per-app root (`appbackup.RecoveryUnitPath`), which is
|
||||
// exactly why one verdict can honestly cover them — and why the message may now claim what it claims.
|
||||
//
|
||||
// WHY IT IS DECIDED LAZILY AND NOT ONCE AT THE START OF THE RUN. Space changes during a run: app A's
|
||||
// 2 GB dump can put app B under the reserve. A verdict taken at run start would wave B through on a
|
||||
// reading that was true before the disk filled — the same class of mistake as the one being fixed,
|
||||
// moved one level up.
|
||||
//
|
||||
// WHY IT IS REMEMBERED AND NOT RE-DECIDED PER LEG. Re-deciding between an app's own legs reintroduces
|
||||
// the split this closes: the DB leg admitted, the volume leg admitted, the capture refused — with the
|
||||
// bulk already written. Decide once, remember, reuse; reset per run, because a set carried between
|
||||
// runs is a wrong answer with a confident face.
|
||||
//
|
||||
// IT REFUSES; IT NEVER DELETES. Unchanged from B2 and load-bearing: nothing on this filesystem is
|
||||
// generational (one unit per app at one fixed path, refreshed in place), so "prune the oldest" could
|
||||
// only mean destroying a DIFFERENT app's only local copy. `pruneStalePrimaryDirs` removes ORPHANED
|
||||
// dirs an app left on a drive it moved off — it has no notion of age or of the current app — and must
|
||||
// never be repurposed for headroom.
|
||||
|
||||
// floorReason records WHICH term bound, so the operator can tell "the disk is full" from "this app's
|
||||
// backup is too big for what is left". An alert that says only "refused" sends them to read code.
|
||||
type floorReason int
|
||||
|
||||
const (
|
||||
floorAdmit floorReason = iota // admitted — no term binds
|
||||
floorHeadroom // the filesystem is ALREADY at/below the reserve
|
||||
floorSize // there is room now, but this app's own write would cross the reserve
|
||||
)
|
||||
|
||||
func (r floorReason) String() string {
|
||||
switch r {
|
||||
case floorHeadroom:
|
||||
return "headroom"
|
||||
case floorSize:
|
||||
return "size"
|
||||
default:
|
||||
return "admitted"
|
||||
}
|
||||
}
|
||||
|
||||
// admissionVerdict is one app's decision for one run. It carries everything the alert needs, so the
|
||||
// alert is rendered once from the same value every leg consults.
|
||||
type admissionVerdict struct {
|
||||
admitted bool
|
||||
reason floorReason
|
||||
usage *UnitSpace
|
||||
estGiB float64 // the estimated write in GiB — the arithmetic unit, matching the reserve's terms
|
||||
estBytes int64 // the same estimate in bytes — the RENDERING unit; see floorRefusal
|
||||
hasEst bool // whether an estimate was available at all (§8.2: distinct from "estimated 0")
|
||||
err error // the refusal, nil when admitted
|
||||
}
|
||||
|
||||
// admissionSet is the per-RUN memo. Deliberately not a field with a lifetime of its own: it is
|
||||
// created by beginAdmissionRun and cleared by the returned func, so an absent set means "no run is in
|
||||
// flight" rather than "a stale answer from last night".
|
||||
type admissionSet struct {
|
||||
v map[string]admissionVerdict
|
||||
}
|
||||
|
||||
// beginAdmissionRun opens the per-run admission scope and returns the closer. Called once at the top
|
||||
// of runDBDumpsInternal — which is the single orchestrator of all three legs — so the DB dump, the
|
||||
// volume dump and the capture of one app all consult the SAME verdict.
|
||||
//
|
||||
// A second call while a set is live REPLACES it and the returned closer restores the previous one, so
|
||||
// nesting cannot silently drop a caller's scope.
|
||||
func (m *Manager) beginAdmissionRun() func() {
|
||||
m.admissionMu.Lock()
|
||||
prev := m.admission
|
||||
m.admission = &admissionSet{v: map[string]admissionVerdict{}}
|
||||
m.admissionMu.Unlock()
|
||||
return func() {
|
||||
m.admissionMu.Lock()
|
||||
m.admission = prev
|
||||
m.admissionMu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// admitApp is THE gate. It returns true when this app may write, false when the reserve refuses it.
|
||||
//
|
||||
// On the first refusal for an app it logs and fires EXACTLY ONE operator alert; every later leg in
|
||||
// the same run reads the memo and stays silent, so a refused app produces one email and not three.
|
||||
//
|
||||
// With no run scope open (the periodic status refresh calls captureAllRecoveryUnits directly) it
|
||||
// decides fresh. That is not a gap: each app appears once in that sweep, so "once per app" still
|
||||
// holds — there is simply nothing to remember it across.
|
||||
func (m *Manager) admitApp(stackName string) bool {
|
||||
m.admissionMu.Lock()
|
||||
defer m.admissionMu.Unlock()
|
||||
|
||||
if set := m.admission; set != nil {
|
||||
if v, ok := set.v[stackName]; ok {
|
||||
return v.admitted // already decided this run — do NOT re-decide, do NOT re-alert
|
||||
}
|
||||
}
|
||||
|
||||
v := m.decideAdmission(stackName)
|
||||
if set := m.admission; set != nil {
|
||||
set.v[stackName] = v
|
||||
}
|
||||
if v.admitted {
|
||||
return true
|
||||
}
|
||||
|
||||
// The claim below is now literally true, and that is the whole point of R-181: the verdict is
|
||||
// taken before the FIRST of the three writes, so at this moment nothing under
|
||||
// backups/primary/<app> has been touched by this run. TestAdmission_RefusedAppsTreeIsByteIdentical
|
||||
// pins the consequence by checksumming the tree, not by reading this line.
|
||||
m.logger.Printf("[WARN] [backup] App backup REFUSED for %s (%s) — %v; NO database dump, NO volume "+
|
||||
"dump and NO recovery-unit capture was written for it, the previous unit is untouched and "+
|
||||
"NOTHING was deleted", stackName, v.reason, v.err)
|
||||
if m.unitNotify != nil {
|
||||
m.unitNotify(stackName, v.err, v.usage)
|
||||
}
|
||||
// R-182: the digest entry is recorded HERE, where the verdict is taken — once per app per run.
|
||||
// Not at the three call sites that consult the memo: R-181's whole contract is that ONE verdict
|
||||
// covers all three legs, so noting it per leg listed a single refused app three times and
|
||||
// produced counts like "2 of 1 apps failed". The leg name says what actually happened, which is
|
||||
// that nothing was attempted at all.
|
||||
m.noteFailure(stackName, "whole app (refused before any write)", v.err.Error())
|
||||
return false
|
||||
}
|
||||
|
||||
// decideAdmission applies the floor to a fresh reading plus this app's estimated write.
|
||||
func (m *Manager) decideAdmission(stackName string) admissionVerdict {
|
||||
estBytes, hasEst := m.estimatedWriteBytes(stackName)
|
||||
estGiB := float64(estBytes) / (1024 * 1024 * 1024)
|
||||
usage, reason := m.floorVerdict(m.readUnitSpace(stackName), estGiB)
|
||||
v := admissionVerdict{
|
||||
admitted: reason == floorAdmit,
|
||||
reason: reason,
|
||||
usage: usage,
|
||||
estGiB: estGiB,
|
||||
estBytes: estBytes,
|
||||
hasEst: hasEst,
|
||||
}
|
||||
if !v.admitted {
|
||||
v.err = floorRefusal(reason, usage, estBytes, hasEst)
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// floorRefusal renders the refusal an operator reads. It names the reserve (not an I/O error — this
|
||||
// is a deliberate hold, not broken machinery), says WHICH term bound, and states plainly when the
|
||||
// decision was headroom-only because the app has no previous backup to estimate from (§8.2).
|
||||
//
|
||||
// THE ESTIMATE IS RENDERED IN BYTES-HUMANIZED, NOT GiB, and that is not cosmetic. Fixed to two
|
||||
// decimal GiB, every app under ~10 MB prints `0.00 GiB` — which reads as "no estimate was available"
|
||||
// and is the opposite of what happened. Observed on the live proof run: opengist's real 178 KB
|
||||
// estimate rendered as `estimated 0.00 GiB write`. The arithmetic stays in GiB (the reserve's own
|
||||
// unit); only the rendering changes.
|
||||
func floorRefusal(reason floorReason, usage *UnitSpace, estBytes int64, hasEst bool) error {
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "%%w (reserve: %.0f%%%% used or %.1f GiB free", FloorUsedPercent, FloorFreeGiB)
|
||||
switch {
|
||||
case reason == floorSize:
|
||||
fmt.Fprintf(&b, "; this app's last backup was %s and writing it again would cross the reserve", humanizeBytes(estBytes))
|
||||
case hasEst:
|
||||
fmt.Fprintf(&b, "; the filesystem is already below it, before this app's estimated %s write", humanizeBytes(estBytes))
|
||||
default:
|
||||
b.WriteString("; this app has no previous backup on disk, so only current headroom was considered")
|
||||
}
|
||||
b.WriteString(") — %s")
|
||||
return fmt.Errorf(b.String(), ErrCaptureFloor, usage)
|
||||
}
|
||||
|
||||
// estimatedWriteBytes estimates what this app's three legs are about to write, from what the PREVIOUS
|
||||
// run left in its unit: the `.sql` dumps and the `.tar` volume archives already on disk for this app.
|
||||
//
|
||||
// WHY THIS ESTIMATOR. It is free — two ReadDirs of a directory the caller is about to write into — and
|
||||
// the next write is usually close to the last one. The alternative, a container-based `du` of every
|
||||
// named volume, was measured on the demo box before being rejected; the figure is in REPORT.md §6.
|
||||
//
|
||||
// NO HISTORY → (0, false), and the caller falls back to headroom-only. Refusing an app because it has
|
||||
// never been backed up would make the first backup the one that can never happen (Scenario E).
|
||||
//
|
||||
// It reads the app's CURRENT unit root, so an app that moved drives estimates from its new (probably
|
||||
// empty) location and is treated as history-less — conservative in the admitting direction, which is
|
||||
// the right way round for an estimate that only ever tightens a threshold.
|
||||
func (m *Manager) estimatedWriteBytes(stackName string) (int64, bool) {
|
||||
drivePath := m.GetAppDrivePath(stackName)
|
||||
if drivePath == "" {
|
||||
return 0, false
|
||||
}
|
||||
nsRoot := m.namespaceRoot(drivePath)
|
||||
var total int64
|
||||
var found bool
|
||||
for _, d := range []struct {
|
||||
dir string
|
||||
ext string
|
||||
}{
|
||||
{AppDBDumpPath(nsRoot, stackName), ".sql"},
|
||||
{AppVolumeDumpPath(nsRoot, stackName), ".tar"},
|
||||
} {
|
||||
n, ok := sumFileSizes(d.dir, d.ext)
|
||||
total += n
|
||||
found = found || ok
|
||||
}
|
||||
if !found {
|
||||
return 0, false
|
||||
}
|
||||
return total, true
|
||||
}
|
||||
|
||||
// sumFileSizes totals the sizes of files with the given suffix in dir. The bool reports whether ANY
|
||||
// such file was seen — distinct from a zero total, because a 0-byte dump is history (a real, if
|
||||
// alarming, previous result) while an absent directory is not.
|
||||
//
|
||||
// A stat error on one entry is skipped rather than aborting the sum: an estimate built from the
|
||||
// readable files is worth more than no estimate, and the entry that could not be read is logged
|
||||
// nowhere because this is a hint, not a measurement — it can only tighten a threshold, never relax
|
||||
// one below what the headroom term already enforces.
|
||||
func sumFileSizes(dir, suffix string) (int64, bool) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
var total int64
|
||||
var found bool
|
||||
for _, e := range entries {
|
||||
if e.IsDir() || !strings.HasSuffix(e.Name(), suffix) {
|
||||
continue
|
||||
}
|
||||
fi, err := os.Stat(filepath.Join(dir, e.Name()))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
total += fi.Size()
|
||||
}
|
||||
return total, found
|
||||
}
|
||||
@@ -0,0 +1,741 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-181 — the reserve guards the write that fills the disk, and its promise is true.
|
||||
//
|
||||
// WHAT THESE ASSERT, AND WHY IT IS THE TREE AND NOT THE LOG. The defect being closed is precisely a
|
||||
// log line that claimed something the filesystem contradicted: B2 printed *"the previous unit is
|
||||
// untouched"* while the volume leg had already rewritten that unit's tar 182,272 B → 2,147,666,432 B.
|
||||
// So a test that reads the message and believes it would have passed against the broken code. Every
|
||||
// refusal test here checksums the whole `backups/primary` tree before and after and compares.
|
||||
|
||||
// ── Harness ──────────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
// admissionProvider records the two acts a refused app must never suffer: its recovery info being
|
||||
// read (a capture that was ATTEMPTED) and its stack being stopped (which DumpAppVolumesSafe does as
|
||||
// its first act, before any check of its own).
|
||||
type admissionProvider struct {
|
||||
stacks []string
|
||||
volumes map[string][]string
|
||||
hdd map[string]string // per-app drive path, for the drive-state skip tests
|
||||
dir string
|
||||
infoHits []string
|
||||
stopped []string
|
||||
}
|
||||
|
||||
func (p *admissionProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *admissionProvider) ListDeployedStacks() []StackSummary {
|
||||
out := make([]StackSummary, 0, len(p.stacks))
|
||||
for _, s := range p.stacks {
|
||||
out = append(out, StackSummary{Name: s})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *admissionProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *admissionProvider) GetStackHDDPath(n string) string { return p.hdd[n] }
|
||||
func (p *admissionProvider) GetImportRoot() string { return "" }
|
||||
func (p *admissionProvider) GetDockerVolumes(name string) []string {
|
||||
if p.volumes == nil {
|
||||
return []string{name + "_data"} // every app is volume-bearing unless told otherwise
|
||||
}
|
||||
return p.volumes[name]
|
||||
}
|
||||
func (p *admissionProvider) StopStack(name string) error {
|
||||
p.stopped = append(p.stopped, name)
|
||||
return nil
|
||||
}
|
||||
func (p *admissionProvider) StartStack(string) error { return nil }
|
||||
func (p *admissionProvider) RefreshAndIsRunning(string) bool { return true }
|
||||
func (p *admissionProvider) GetStackRecoveryInfo(name string) (RecoveryInfo, bool) {
|
||||
p.infoHits = append(p.infoHits, name)
|
||||
return RecoveryInfo{StackDir: filepath.Join(p.dir, "stacks", name)}, true
|
||||
}
|
||||
func (p *admissionProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *admissionProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *admissionProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *admissionProvider) GetStackClassifiedBinds(string) ([]ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
type admissionHarness struct {
|
||||
m *Manager
|
||||
prov *admissionProvider
|
||||
events []unitEvent
|
||||
usage map[string]*UnitSpace
|
||||
dir string
|
||||
logs *bytes.Buffer
|
||||
volDumped []string
|
||||
}
|
||||
|
||||
func newAdmissionHarness(t *testing.T, stacks ...string) *admissionHarness {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
h := &admissionHarness{
|
||||
prov: &admissionProvider{stacks: stacks, dir: dir, hdd: map[string]string{}},
|
||||
usage: map[string]*UnitSpace{},
|
||||
dir: dir,
|
||||
logs: &bytes.Buffer{},
|
||||
}
|
||||
h.m = &Manager{
|
||||
logger: log.New(h.logs, "", 0),
|
||||
systemDataPath: dir,
|
||||
stackProvider: h.prov,
|
||||
unitSpaceFn: func(name string) *UnitSpace { return h.usage[name] },
|
||||
}
|
||||
// The volume-dump seam records the leg that writes the BULK — the one B2 never gated. A refused
|
||||
// app must not reach it.
|
||||
h.m.dumpVolumesSafe = func(name string) error {
|
||||
h.volDumped = append(h.volDumped, name)
|
||||
// Write what the real leg writes, so an ungated call is visible in the tree checksum too.
|
||||
dumpDir := AppVolumeDumpPath(h.nsRoot(), name)
|
||||
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(filepath.Join(dumpDir, name+"_data.tar"), []byte("FRESH TAR FROM THIS RUN"), 0o644)
|
||||
}
|
||||
h.m.SetUnitNotify(func(name string, err error, u *UnitSpace) {
|
||||
h.events = append(h.events, unitEvent{app: name, err: err.Error(), usage: u})
|
||||
})
|
||||
return h
|
||||
}
|
||||
|
||||
// markDisconnected / markDecommissioned put a real settings row behind the drive-state skips, so
|
||||
// Scenario F exercises the production guards rather than a stub of them.
|
||||
func (h *admissionHarness) markDisconnected(app string) {
|
||||
h.driveState(app, true, false)
|
||||
}
|
||||
|
||||
func (h *admissionHarness) markDecommissioned(app string) {
|
||||
h.driveState(app, false, true)
|
||||
}
|
||||
|
||||
func (h *admissionHarness) driveState(app string, disconnected, decommissioned bool) {
|
||||
if h.m.settings == nil {
|
||||
sett, err := settings.Load(filepath.Join(h.dir, "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
h.m.settings = sett
|
||||
}
|
||||
// Each such app gets its OWN drive path, or marking one would skip them all.
|
||||
p := filepath.Join(h.dir, "drives", app)
|
||||
if err := os.MkdirAll(p, 0o755); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
h.prov.hdd[app] = p
|
||||
if err := h.m.settings.AddStoragePath(settings.StoragePath{Path: p, Label: app}); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if disconnected {
|
||||
if err := h.m.settings.SetDisconnected(p, true, nil); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
}
|
||||
if decommissioned {
|
||||
if err := h.m.settings.SetDecommissioned(p, ""); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (h *admissionHarness) nsRoot() string { return filepath.Join(h.dir, "felhom-data") }
|
||||
|
||||
// setSpace states the filesystem's occupancy as a test INPUT — the whole point of the unitSpaceFn
|
||||
// seam, so no test has to manufacture disk pressure on a real disk.
|
||||
func (h *admissionHarness) setSpace(app string, usedPct, availGB, totalGB float64) {
|
||||
h.usage[app] = &UnitSpace{
|
||||
Path: h.dir, UsedPercent: usedPct, AvailGB: availGB,
|
||||
TotalGB: totalGB, UsedGB: totalGB * usedPct / 100,
|
||||
}
|
||||
}
|
||||
|
||||
// seedUnit writes a previous recovery unit for an app: a manifest, a captured app.yaml, a DB dump and
|
||||
// a volume tar of the given size. The tar is SPARSE (Truncate), so a 2 GiB "previous backup" costs no
|
||||
// disk — the estimator reads st_size, which is what the next write will actually cost.
|
||||
func (h *admissionHarness) seedUnit(t *testing.T, app string, tarBytes int64) {
|
||||
t.Helper()
|
||||
ns := h.nsRoot()
|
||||
for _, d := range []string{
|
||||
RecoveryUnitComposePath(ns, app),
|
||||
AppDBDumpPath(ns, app),
|
||||
AppVolumeDumpPath(ns, app),
|
||||
} {
|
||||
if err := os.MkdirAll(d, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write := func(p string, b []byte, mode os.FileMode) {
|
||||
if err := os.WriteFile(p, b, mode); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write(RecoveryUnitManifestPath(ns, app), []byte(`{"app_name":"`+app+`","created_at":"2026-08-02T00:00:00Z"}`), 0o644)
|
||||
write(filepath.Join(RecoveryUnitComposePath(ns, app), "app.yaml"), []byte("deployed: true\nenv:\n A: previous-good-value\n"), 0o600)
|
||||
write(filepath.Join(AppDBDumpPath(ns, app), app+"-postgres.sql"), []byte("-- previous good dump\n"), 0o644)
|
||||
|
||||
tar := filepath.Join(AppVolumeDumpPath(ns, app), app+"_data.tar")
|
||||
f, err := os.Create(tar)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := f.WriteString("PREVIOUS GOOD TAR"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if tarBytes > 0 {
|
||||
if err := f.Truncate(tarBytes); err != nil { // sparse — st_size is the estimate, blocks are not spent
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// runOneBackupRun performs exactly the sequence runDBDumpsInternal performs for the two legs that can
|
||||
// be driven without Docker: the admission scope is opened, the volume leg runs, then the capture leg.
|
||||
// The DB leg's wiring is pinned structurally by TestAdmission_IsWiredIntoEveryProductionWriteLeg,
|
||||
// because DiscoverDatabases shells out to `docker` and cannot honestly run here.
|
||||
func (h *admissionHarness) runOneBackupRun() {
|
||||
done := h.m.beginAdmissionRun()
|
||||
defer done()
|
||||
h.m.runVolumeDumps()
|
||||
h.m.captureAllRecoveryUnits()
|
||||
}
|
||||
|
||||
// ── The instrument: a checksum of the whole backup tree ──────────────────────────────────────────
|
||||
|
||||
// treeFingerprint walks every file under backups/primary and returns "relpath mode sha256" lines,
|
||||
// sorted. It is the ONLY honest way to check the refusal's claim: it detects a rewritten payload, an
|
||||
// added file and a deleted one alike, which a log line and an exit code both fail to do.
|
||||
func treeFingerprint(t *testing.T, root string) string {
|
||||
t.Helper()
|
||||
var lines []string
|
||||
err := filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
if fi.IsDir() {
|
||||
return nil
|
||||
}
|
||||
f, err := os.Open(p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
sum := sha256.New()
|
||||
if _, err := io.Copy(sum, f); err != nil {
|
||||
return err
|
||||
}
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
lines = append(lines, fmt.Sprintf("%s %o %d %s", rel, fi.Mode().Perm(), fi.Size(), hex.EncodeToString(sum.Sum(nil))))
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("fingerprinting %s: %v", root, err)
|
||||
}
|
||||
sort.Strings(lines)
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
// treeStatFingerprint is the instrument for trees holding a multi-GiB fixture, where hashing every
|
||||
// byte costs more than it proves: name + mode + SIZE. It still catches the act being tested — the
|
||||
// volume leg replacing a 2 GiB tar with a freshly written one — because that changes the size, and it
|
||||
// catches an added or deleted file by name. Content-identical-but-different-bytes is the one thing it
|
||||
// cannot see, which is why the small-tree tests use treeFingerprint instead.
|
||||
func treeStatFingerprint(t *testing.T, root string) string {
|
||||
t.Helper()
|
||||
var lines []string
|
||||
_ = filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err != nil || fi.IsDir() {
|
||||
return nil
|
||||
}
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
lines = append(lines, fmt.Sprintf("%s %o %d", rel, fi.Mode().Perm(), fi.Size()))
|
||||
return nil
|
||||
})
|
||||
sort.Strings(lines)
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
// treeFileList is the weaker instrument used for Scenario F: names only, so the assertion is
|
||||
// specifically about DELETION and cannot be satisfied or broken by a content change.
|
||||
func treeFileList(t *testing.T, root string) []string {
|
||||
t.Helper()
|
||||
var names []string
|
||||
_ = filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err != nil || fi.IsDir() {
|
||||
return nil
|
||||
}
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
names = append(names, rel)
|
||||
return nil
|
||||
})
|
||||
sort.Strings(names)
|
||||
return names
|
||||
}
|
||||
|
||||
func (h *admissionHarness) primaryRoot() string {
|
||||
return PrimaryBackupPath(h.nsRoot())
|
||||
}
|
||||
|
||||
// ── Scenario A — one decision, taken before the first byte ───────────────────────────────────────
|
||||
|
||||
func TestAdmission_RefusedAppWritesNothingAndIsNotStopped(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "privatebin", "opengist", "homebox")
|
||||
h.setSpace("privatebin", 40, 60, 100)
|
||||
h.setSpace("opengist", 98, 0.4, 70) // below the reserve on BOTH terms
|
||||
h.setSpace("homebox", 40, 60, 100)
|
||||
h.seedUnit(t, "opengist", 0)
|
||||
|
||||
// Scoped to the REFUSED app's own unit: its two siblings are admitted and legitimately write
|
||||
// theirs, so a whole-tree fingerprint would change for the right reason and prove nothing here.
|
||||
// Scenario F below takes the whole-tree view, where every app is refused.
|
||||
refusedUnit := RecoveryUnitPath(h.nsRoot(), "opengist")
|
||||
before := treeFingerprint(t, refusedUnit)
|
||||
if before == "" {
|
||||
t.Fatal("the fixture seeded no previous unit, so 'byte-identical' would be vacuously true")
|
||||
}
|
||||
h.runOneBackupRun()
|
||||
after := treeFingerprint(t, refusedUnit)
|
||||
|
||||
// 1. NOT ONE of the three legs ran for the refused app.
|
||||
for _, got := range h.volDumped {
|
||||
if got == "opengist" {
|
||||
t.Fatal("the VOLUME leg ran for a refused app — this is the R-181 defect exactly: the leg " +
|
||||
"that writes the bulk was never gated, so the reserve it protects was consumed by the " +
|
||||
"very step it exists to bound")
|
||||
}
|
||||
}
|
||||
for _, got := range h.prov.infoHits {
|
||||
if got == "opengist" {
|
||||
t.Fatal("the CAPTURE leg was attempted for a refused app — the verdict must be taken before " +
|
||||
"any write is prepared, not partway through one")
|
||||
}
|
||||
}
|
||||
|
||||
// 2. The tree is byte-identical. This is the assertion the broken code could not pass.
|
||||
if after != before {
|
||||
t.Fatalf("the backup tree CHANGED across a refusal.\n--- before ---\n%s\n--- after ---\n%s\n"+
|
||||
"A refusal that has already rewritten the payload is the defect, not the fix", before, after)
|
||||
}
|
||||
|
||||
// 3. The app was never stopped. DumpAppVolumesSafe stops the stack as its FIRST act, so a gate
|
||||
// placed inside it would bounce the app it is refusing to back up.
|
||||
for _, got := range h.prov.stopped {
|
||||
if got == "opengist" {
|
||||
t.Fatal("the refused app was STOPPED — the reserve check has drifted behind the stop")
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Exactly ONE alert, for that app, carrying the space figures. Three legs must not mean three
|
||||
// emails about one disk.
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1 (one app refused, three legs): %+v", len(h.events), h.events)
|
||||
}
|
||||
if h.events[0].app != "opengist" {
|
||||
t.Fatalf("alert names %q, want opengist", h.events[0].app)
|
||||
}
|
||||
if h.events[0].usage == nil || h.events[0].usage.AvailGB != 0.4 {
|
||||
t.Fatalf("the alert carries no/incorrect space figures: %+v", h.events[0].usage)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario B — the other apps are unaffected ───────────────────────────────────────────────────
|
||||
|
||||
func TestAdmission_SiblingAppsProceedAndOnlyTheRefusedOneAlerts(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "privatebin", "opengist", "homebox")
|
||||
h.setSpace("privatebin", 40, 60, 100)
|
||||
h.setSpace("opengist", 99, 0.2, 70)
|
||||
h.setSpace("homebox", 40, 60, 100)
|
||||
|
||||
h.runOneBackupRun()
|
||||
|
||||
for _, app := range []string{"privatebin", "homebox"} {
|
||||
if !hasStr(h.volDumped, app) {
|
||||
t.Errorf("%s was not volume-dumped (dumped=%v) — one app's refusal silenced its siblings", app, h.volDumped)
|
||||
}
|
||||
if !hasStr(h.prov.infoHits, app) {
|
||||
t.Errorf("%s was not captured (attempted=%v) — the loop did not continue past the refusal", app, h.prov.infoHits)
|
||||
}
|
||||
if _, err := os.Stat(RecoveryUnitManifestPath(h.nsRoot(), app)); err != nil {
|
||||
t.Errorf("%s has no manifest after the run: %v — an admitted app must be backed up normally", app, err)
|
||||
}
|
||||
}
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1: %+v", len(h.events), h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario C — the promise is true ─────────────────────────────────────────────────────────────
|
||||
|
||||
// Every claim the shipped message makes is checked against the tree it describes. The wording is NOT
|
||||
// weakened to fit the behaviour; the behaviour was moved so the wording became true (§8.3).
|
||||
func TestAdmission_EveryClaimInTheRefusalMessageHoldsAgainstTheTree(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
h.setSpace("opengist", 98, 0.5, 70)
|
||||
h.seedUnit(t, "opengist", 0)
|
||||
|
||||
beforeFP := treeFingerprint(t, h.primaryRoot())
|
||||
beforeList := treeFileList(t, h.primaryRoot())
|
||||
h.runOneBackupRun()
|
||||
msg := h.logs.String()
|
||||
|
||||
if !strings.Contains(msg, "REFUSED for opengist") {
|
||||
t.Fatalf("no refusal was logged for opengist; log was:\n%s", msg)
|
||||
}
|
||||
|
||||
// Claim 1: "NO database dump, NO volume dump and NO recovery-unit capture was written for it".
|
||||
for _, claim := range []string{"NO database dump", "NO volume dump", "NO recovery-unit capture"} {
|
||||
if !strings.Contains(msg, claim) {
|
||||
t.Fatalf("the message no longer claims %q — if a leg cannot be brought under the verdict the "+
|
||||
"wording must be narrowed deliberately and the gap named, not dropped silently.\n%s", claim, msg)
|
||||
}
|
||||
}
|
||||
if len(h.volDumped) != 0 || len(h.prov.infoHits) != 0 {
|
||||
t.Fatalf("the message claims no leg ran, but volume=%v capture=%v", h.volDumped, h.prov.infoHits)
|
||||
}
|
||||
|
||||
// Claim 2: "the previous unit is untouched" — the claim that was MEASURED FALSE in R-181.
|
||||
if !strings.Contains(msg, "the previous unit is untouched") {
|
||||
t.Fatalf("the message dropped the untouched claim: %s", msg)
|
||||
}
|
||||
if got := treeFingerprint(t, h.primaryRoot()); got != beforeFP {
|
||||
t.Fatalf("the message says the previous unit is untouched; the tree says otherwise.\n"+
|
||||
"--- before ---\n%s\n--- after ---\n%s", beforeFP, got)
|
||||
}
|
||||
|
||||
// Claim 3: "NOTHING was deleted".
|
||||
if !strings.Contains(msg, "NOTHING was deleted") {
|
||||
t.Fatalf("the message dropped the no-deletion claim: %s", msg)
|
||||
}
|
||||
if got := treeFileList(t, h.primaryRoot()); !equalStrs(got, beforeList) {
|
||||
t.Fatalf("files disappeared across a refusal: before=%v after=%v", beforeList, got)
|
||||
}
|
||||
|
||||
// Claim 4: the reason is named, so the operator can tell which term bound.
|
||||
if !strings.Contains(msg, "headroom") {
|
||||
t.Fatalf("the message does not name WHICH term bound — an operator cannot tell 'the disk is "+
|
||||
"full' from 'this app's backup is too big for what is left':\n%s", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario D — size-aware, not just headroom-aware ─────────────────────────────────────────────
|
||||
|
||||
// The live R-181 sequence, reproduced as a unit: the filesystem is ABOVE the reserve on both terms
|
||||
// when the run reaches the app, and the app's own write is what crosses it. Under B2 this app was
|
||||
// admitted at 96% and then allowed to write 2 GB.
|
||||
func TestAdmission_SizeTermRefusesAnAppWhoseOwnWriteWouldCrossTheReserve(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
// 96% used of 70 GiB, 3.0 GiB free — BOTH reserve terms deliberately still clear (97% / 1.0 GiB),
|
||||
// exactly as on demo-hp at 06:40:03, so a headroom-only rule starts the run.
|
||||
h.setSpace("opengist", 96, 3.0, 70)
|
||||
if _, r := h.m.floorVerdict(h.usage["opengist"], 0); r != floorAdmit {
|
||||
t.Fatalf("fixture is wrong: the headroom term already refuses (%v), so this test would pass "+
|
||||
"without a size term and prove nothing", r)
|
||||
}
|
||||
h.seedUnit(t, "opengist", 2<<30) // its last backup was 2 GiB — the figure measured live
|
||||
|
||||
before := treeStatFingerprint(t, h.primaryRoot())
|
||||
h.runOneBackupRun()
|
||||
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want 1 — the app was admitted at 96%% and would have been allowed to "+
|
||||
"write 2 GiB, which is the R-181 sequence: %+v", len(h.events), h.events)
|
||||
}
|
||||
if !strings.Contains(h.logs.String(), "(size)") {
|
||||
t.Fatalf("the refusal was not attributed to the SIZE term:\n%s", h.logs.String())
|
||||
}
|
||||
if !strings.Contains(h.events[0].err, "last backup was 2.0 GB") {
|
||||
t.Fatalf("the alert does not carry the estimate that produced the refusal: %q", h.events[0].err)
|
||||
}
|
||||
if len(h.volDumped) != 0 {
|
||||
t.Fatalf("the volume leg ran anyway: %v", h.volDumped)
|
||||
}
|
||||
if got := treeStatFingerprint(t, h.primaryRoot()); got != before {
|
||||
t.Fatalf("the tree changed despite the size-term refusal.\nbefore=%s\nafter =%s", before, got)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario E — a first-ever backup is not blocked by having no history ─────────────────────────
|
||||
|
||||
func TestAdmission_FirstEverBackupIsAdmitted(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "brandnew")
|
||||
h.setSpace("brandnew", 40, 600, 1000) // ample room, and NO previous unit on disk
|
||||
|
||||
if est, ok := h.m.estimatedWriteBytes("brandnew"); ok || est != 0 {
|
||||
t.Fatalf("estimatedWriteBytes = (%v, %v) for an app with no history, want (0, false)", est, ok)
|
||||
}
|
||||
h.runOneBackupRun()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("a brand-new app was refused: %+v — refusing every app that has no size to estimate "+
|
||||
"from would make the FIRST backup the one that can never happen", h.events)
|
||||
}
|
||||
if !hasStr(h.volDumped, "brandnew") || !hasStr(h.prov.infoHits, "brandnew") {
|
||||
t.Fatalf("the app was not backed up (volume=%v capture=%v)", h.volDumped, h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario F — the reserve still never deletes ─────────────────────────────────────────────────
|
||||
|
||||
func TestAdmission_NothingUnderBackupsIsEverRemoved(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "privatebin", "opengist", "homebox")
|
||||
for _, app := range []string{"privatebin", "opengist", "homebox"} {
|
||||
h.setSpace(app, 99, 0.1, 70) // every app refused — maximum pressure to "make room"
|
||||
h.seedUnit(t, app, 0)
|
||||
}
|
||||
|
||||
before := treeFileList(t, h.primaryRoot())
|
||||
h.runOneBackupRun()
|
||||
after := treeFileList(t, h.primaryRoot())
|
||||
|
||||
if !equalStrs(before, after) {
|
||||
t.Fatalf("the file list changed under the reserve.\nbefore=%v\nafter =%v\n"+
|
||||
"Nothing here is generational — a unit is ONE fixed path per app — so 'prune the oldest' "+
|
||||
"could only mean destroying a DIFFERENT app's only local recovery unit", before, after)
|
||||
}
|
||||
if len(before) == 0 {
|
||||
t.Fatal("the fixture seeded no files, so this test would pass against code that deleted everything")
|
||||
}
|
||||
}
|
||||
|
||||
// ── §8.1 — one verdict per app per run, and it resets between runs ───────────────────────────────
|
||||
|
||||
// The verdict must not be re-taken between an app's own legs. Re-deciding is how the split this fixes
|
||||
// came about: DB leg admitted, volume leg admitted, capture refused — with the bulk already written.
|
||||
func TestAdmission_VerdictIsTakenOncePerAppPerRunAndNotRedecidedBetweenLegs(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
reads := 0
|
||||
h.m.unitSpaceFn = func(string) *UnitSpace {
|
||||
reads++
|
||||
if reads == 1 {
|
||||
return &UnitSpace{Path: h.dir, UsedPercent: 99, AvailGB: 0.1, TotalGB: 70, UsedGB: 69.3}
|
||||
}
|
||||
// The disk "recovers" mid-run. A re-decided verdict would admit the capture leg here — which
|
||||
// is precisely the split R-181 closes, arriving from the other direction.
|
||||
return &UnitSpace{Path: h.dir, UsedPercent: 10, AvailGB: 60, TotalGB: 70, UsedGB: 7}
|
||||
}
|
||||
|
||||
h.runOneBackupRun()
|
||||
|
||||
if reads != 1 {
|
||||
t.Fatalf("the filesystem was read %d times for ONE app in ONE run — the verdict is being "+
|
||||
"re-decided between legs, which reintroduces the split (bulk written, capture refused)", reads)
|
||||
}
|
||||
if len(h.prov.infoHits) != 0 {
|
||||
t.Fatal("the capture leg ran after the app was refused earlier in the same run")
|
||||
}
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1 per app per run: %+v", len(h.events), h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// A set carried between runs is a wrong answer with a confident face: tonight's question answered
|
||||
// with last night's disk.
|
||||
func TestAdmission_TheRememberedSetResetsBetweenRuns(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
h.setSpace("opengist", 99, 0.1, 70)
|
||||
h.runOneBackupRun()
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("run 1: want 1 alert, got %+v", h.events)
|
||||
}
|
||||
|
||||
h.setSpace("opengist", 20, 55, 70) // space freed between runs
|
||||
h.runOneBackupRun()
|
||||
|
||||
if !hasStr(h.volDumped, "opengist") {
|
||||
t.Fatal("the second run still refused the app — the previous run's verdict was carried over, " +
|
||||
"so freeing space could never take effect")
|
||||
}
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("the second (admitted) run alerted again: %+v", h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// ── §8.4 — a nil reading neither refuses nor warns, across ALL THREE legs ────────────────────────
|
||||
|
||||
// Unchanged behaviour, re-pinned because the decision now governs three legs instead of one: an
|
||||
// unreadable filesystem must not silently stop an app being backed up at all.
|
||||
func TestAdmission_UnreadableFilesystemAdmitsEveryLegAndDoesNotWarn(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
// No usage entry → the reader returns nil, which is what system.GetDiskUsage does on error.
|
||||
|
||||
h.runOneBackupRun()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("an unreadable filesystem produced %d alert(s): %+v — that is the drive gate's "+
|
||||
"business and has its own alert", len(h.events), h.events)
|
||||
}
|
||||
if !hasStr(h.volDumped, "opengist") {
|
||||
t.Fatal("the VOLUME leg was refused on an unreadable read — a drive that merely blipped would " +
|
||||
"now stop the bulk of the backup, not just the capture")
|
||||
}
|
||||
if !hasStr(h.prov.infoHits, "opengist") {
|
||||
t.Fatal("the CAPTURE leg was refused on an unreadable read")
|
||||
}
|
||||
}
|
||||
|
||||
// ── The estimator, through the production path (no seam) ─────────────────────────────────────────
|
||||
|
||||
func TestEstimatedWriteBytes_SumsTheAppsPreviousDumpsFromRealFiles(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
h.seedUnit(t, "opengist", 3<<30) // 3 GiB sparse tar + a small .sql
|
||||
|
||||
est, ok := h.m.estimatedWriteBytes("opengist")
|
||||
if !ok {
|
||||
t.Fatal("history on disk was not recognised as history")
|
||||
}
|
||||
if est < 3<<30 || est > (3<<30)+4096 {
|
||||
t.Fatalf("estimate = %d B, want ~%d (the .tar plus the small .sql)", est, int64(3)<<30)
|
||||
}
|
||||
|
||||
// An app whose unit exists but holds no dumps yet is history-LESS, not a zero-byte estimate.
|
||||
other := AppVolumeDumpPath(h.nsRoot(), "empty")
|
||||
if err := os.MkdirAll(other, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if est, ok := h.m.estimatedWriteBytes("empty"); ok || est != 0 {
|
||||
t.Fatalf("an empty unit reported history (%v, %v) — an absent dump is not a 0-byte one", est, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// ── The seam is WIRED — walked as an AST, not grepped ────────────────────────────────────────────
|
||||
|
||||
// FOUR mechanisms in this project have been built and left disconnected (REUSE.md's seam register).
|
||||
// The behavioural tests above drive the two legs that can run without Docker; the DB leg cannot, so
|
||||
// its gate is pinned HERE, structurally. `strings.Contains` is deliberately not used: a commented-out
|
||||
// call still contains the string, and so does a call inside dead code.
|
||||
func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
file, err := parser.ParseFile(fset, "backup.go", nil, 0) // comments dropped — only real calls survive
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
calls := map[string][]string{} // enclosing func → called names, in source order
|
||||
var current string
|
||||
ast.Inspect(file, func(n ast.Node) bool {
|
||||
switch v := n.(type) {
|
||||
case *ast.FuncDecl:
|
||||
current = v.Name.Name
|
||||
case *ast.CallExpr:
|
||||
name := ""
|
||||
switch fn := v.Fun.(type) {
|
||||
case *ast.Ident:
|
||||
name = fn.Name
|
||||
case *ast.SelectorExpr:
|
||||
name = fn.Sel.Name
|
||||
}
|
||||
if name != "" && current != "" {
|
||||
calls[current] = append(calls[current], name)
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
// 1. The run scope is opened by the orchestrator of all three legs.
|
||||
if !hasStr(calls["runDBDumpsInternal"], "beginAdmissionRun") {
|
||||
t.Fatal("runDBDumpsInternal does not open the admission scope — without it every leg decides " +
|
||||
"independently and the per-run memo never exists, which is the pre-R-181 behaviour")
|
||||
}
|
||||
|
||||
// 2. The DB leg consults it BEFORE the dump. Order is the whole point: a gate after the write is
|
||||
// the defect, relocated.
|
||||
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "DumpOne",
|
||||
"the DATABASE leg dumps before consulting the reserve")
|
||||
|
||||
// 3. The volume leg consults it BEFORE the dump seam — which stops the stack as its first act.
|
||||
assertGateBefore(t, calls["runVolumeDumps"], "admitApp", "dump",
|
||||
"the VOLUME leg — the one that writes the bulk, and the one B2 never gated — dumps before "+
|
||||
"consulting the reserve")
|
||||
|
||||
// 4. The capture leg, in its own file.
|
||||
rfset := token.NewFileSet()
|
||||
rfile, err := parser.ParseFile(rfset, "recovery_unit.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
capCalls := map[string][]string{}
|
||||
current = ""
|
||||
ast.Inspect(rfile, func(n ast.Node) bool {
|
||||
switch v := n.(type) {
|
||||
case *ast.FuncDecl:
|
||||
current = v.Name.Name
|
||||
case *ast.CallExpr:
|
||||
if sel, ok := v.Fun.(*ast.SelectorExpr); ok && current != "" {
|
||||
capCalls[current] = append(capCalls[current], sel.Sel.Name)
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "CaptureRecoveryUnit",
|
||||
"the CAPTURE leg captures before consulting the reserve")
|
||||
}
|
||||
|
||||
// assertGateBefore checks that `gate` appears in the call list before `act`.
|
||||
func assertGateBefore(t *testing.T, calls []string, gate, act, why string) {
|
||||
t.Helper()
|
||||
gi, ai := -1, -1
|
||||
for i, c := range calls {
|
||||
if c == gate && gi < 0 {
|
||||
gi = i
|
||||
}
|
||||
if c == act && ai < 0 {
|
||||
ai = i
|
||||
}
|
||||
}
|
||||
if gi < 0 {
|
||||
t.Fatalf("%s: %q is never called there at all (calls=%v)", why, gate, calls)
|
||||
}
|
||||
if ai < 0 {
|
||||
t.Fatalf("fixture drift: %q is no longer called in that function (calls=%v) — this test can no "+
|
||||
"longer see the act it is ordering the gate against", act, calls)
|
||||
}
|
||||
if gi > ai {
|
||||
t.Fatalf("%s: %q first appears at %d, after %q at %d", why, gate, gi, act, ai)
|
||||
}
|
||||
}
|
||||
|
||||
func hasStr(hay []string, needle string) bool {
|
||||
for _, s := range hay {
|
||||
if s == needle {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func equalStrs(a, b []string) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for i := range a {
|
||||
if a[i] != b[i] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
@@ -27,6 +27,7 @@ type AppBackupInfo = appbackup.AppBackupInfo
|
||||
type AppDataPath = appbackup.AppDataPath
|
||||
type AppDockerVolume = appbackup.AppDockerVolume
|
||||
type RecoveryInfo = appbackup.RecoveryInfo
|
||||
type ClassifiedBind = appbackup.ClassifiedBind
|
||||
|
||||
// --- type aliases (dbdump) ---
|
||||
|
||||
@@ -43,6 +44,14 @@ const (
|
||||
DBTypeMariaDB = appbackup.DBTypeMariaDB
|
||||
)
|
||||
|
||||
// Backup-classification class constants (Task 3-core) — aliased so the tier engines can switch on
|
||||
// class without importing appbackup directly.
|
||||
const (
|
||||
ClassMandatory = appbackup.ClassMandatory
|
||||
ClassOptional = appbackup.ClassOptional
|
||||
ClassExcluded = appbackup.ClassExcluded
|
||||
)
|
||||
|
||||
// FelhomDataDir is the namespace directory on storage drives for all felhom-managed data.
|
||||
const FelhomDataDir = appbackup.FelhomDataDir
|
||||
|
||||
@@ -91,6 +100,12 @@ func ParseComposeImages(composePath string) []string {
|
||||
return appbackup.ParseComposeImages(composePath)
|
||||
}
|
||||
|
||||
// DBServiceNames forwards to appbackup.DBServiceNames — the compose SERVICE names holding a database,
|
||||
// i.e. the argument list for the DB-only bring-up both restore paths use before a dump replay (R-47).
|
||||
func DBServiceNames(composePath string) ([]string, error) {
|
||||
return appbackup.DBServiceNames(composePath)
|
||||
}
|
||||
|
||||
// humanizeBytes forwards to appbackup.HumanizeBytes; kept unexported so the
|
||||
// many in-package call sites (backup.go, crossdrive.go, restore code) need no edit.
|
||||
func humanizeBytes(b int64) string {
|
||||
@@ -106,6 +121,11 @@ func NamespaceRoot(drivePath string, inGuestDrive bool) string {
|
||||
return appbackup.NamespaceRoot(drivePath, inGuestDrive)
|
||||
}
|
||||
|
||||
// NamespaceRootFor re-exports the ONE drive-kind-aware resolver (R-203).
|
||||
func NamespaceRootFor(drivePath, systemDataPath string) string {
|
||||
return appbackup.NamespaceRootFor(drivePath, systemDataPath)
|
||||
}
|
||||
|
||||
func PrimaryBackupPath(nsRoot string) string {
|
||||
return appbackup.PrimaryBackupPath(nsRoot)
|
||||
}
|
||||
@@ -133,3 +153,11 @@ func RecoveryUnitManifestPath(nsRoot, stackName string) string {
|
||||
func AppDataDir(nsRoot, stackName string) string {
|
||||
return appbackup.AppDataDir(nsRoot, stackName)
|
||||
}
|
||||
|
||||
func AppDataDirNames(hddPath, stackName string, hddMounts []string) []string {
|
||||
return appbackup.AppDataDirNames(hddPath, stackName, hddMounts)
|
||||
}
|
||||
|
||||
func AppDataBindsPresent(hddPath string, hddMounts []string) bool {
|
||||
return appbackup.AppDataBindsPresent(hddPath, hddMounts)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,229 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-174 — the app-stop guard's crash recovery must not start an app onto a MISSING drive.
|
||||
//
|
||||
// The defect these pin, found by review on 2026-08-02 in code shipped 2026-08-01 (v0.189.0):
|
||||
// `appStopGuard.SetStarter(stackMgr)` handed Recover the raw stack manager, whose `StartStack` has
|
||||
// no drive gate. Recover runs at STARTUP — exactly when an external drive may not have come back —
|
||||
// so a backup that stopped an app, followed by a power cut and a drive that did not remount, ended
|
||||
// with the app started onto a missing drive. R-171 one path over.
|
||||
//
|
||||
// THE SEAM UNDER TEST IS THE STARTER, not the gate: `internal/backup` must not import `stacks` or
|
||||
// `settings`, so the production gate lives in `cmd/controller`. What is pinned here is the contract
|
||||
// between them — that a starter returning ErrStartRefused produces a REFUSAL (marker kept, no alarm)
|
||||
// and not a FAILURE. The production wiring itself is pinned by TestMainWiresGatedAppStopStarter.
|
||||
|
||||
// gatingStarter is a starter whose gate refuses a named set of apps, in the shape the production
|
||||
// `gatedAppStopStarter` uses: refuse BEFORE calling through, and wrap ErrStartRefused with a reason.
|
||||
type gatingStarter struct {
|
||||
inner *fakeStarter
|
||||
refuse map[string]string // app → reason
|
||||
refused []string
|
||||
}
|
||||
|
||||
func (s *gatingStarter) StartStack(name string) error {
|
||||
if why, ok := s.refuse[name]; ok {
|
||||
s.refused = append(s.refused, name)
|
||||
return fmt.Errorf("%w: %s", ErrStartRefused, why)
|
||||
}
|
||||
return s.inner.StartStack(name)
|
||||
}
|
||||
|
||||
func newGatedGuard(t *testing.T, dir string, refuse map[string]string) (*AppStopGuard, *gatingStarter) {
|
||||
t.Helper()
|
||||
s := &gatingStarter{inner: &fakeStarter{}, refuse: refuse}
|
||||
g := NewAppStopGuard(filepath.Join(dir, "appstop-state.json"), log.New(io.Discard, "", 0))
|
||||
g.SetStarter(s)
|
||||
return g, s
|
||||
}
|
||||
|
||||
// --- Scenario A — the guard does not start an app onto a missing drive ---------------------------
|
||||
|
||||
func TestRecover_DriveAbsent_RefusesTheStartAndKEEPSTheMarker(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
|
||||
// process 1: a volume dump stops immich, then the box loses power. No End(), no defer.
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatalf("Begin: %v", err)
|
||||
}
|
||||
// <power cut> — and immich's drive does NOT come back.
|
||||
|
||||
// process 2: a fresh controller starts. The drive is absent.
|
||||
g2, starter := newGatedGuard(t, dir, map[string]string{
|
||||
"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint",
|
||||
})
|
||||
res := g2.Recover()
|
||||
|
||||
if len(starter.inner.starts) != 0 {
|
||||
t.Fatalf("started %v — the app was started onto a MISSING drive, which is the whole defect",
|
||||
starter.inner.starts)
|
||||
}
|
||||
if res == nil {
|
||||
t.Fatal("Recover returned nil — the refusal is invisible to the caller, so nothing can report it")
|
||||
}
|
||||
if len(res.Refused) != 1 || res.Refused[0] != "immich" {
|
||||
t.Fatalf("refused=%v, want [immich]", res.Refused)
|
||||
}
|
||||
if len(res.Failed) != 0 {
|
||||
t.Fatalf("failed=%v — a deliberate hold was recorded as a FAILURE. That bucket reaches "+
|
||||
"NotifyBackupFailed, which is customer-enabled by default, so the customer would be "+
|
||||
"emailed \"A biztonsági mentés sikertelen!\" about an app nothing is wrong with (R-171's "+
|
||||
"false-alarm shape one path over)", res.Failed)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was CLEARED after a refused start — the operation is genuinely " +
|
||||
"unfinished, and clearing it erases the only durable record that immich is owed a restart")
|
||||
}
|
||||
// The refusal must name the app AND the reason, or an operator cannot act on it.
|
||||
if d := res.Detail(); !strings.Contains(d, "held_by_drive") || !strings.Contains(d, "immich") {
|
||||
t.Fatalf("detail %q does not name the held app", d)
|
||||
}
|
||||
if msg := res.Message(); !strings.Contains(msg, "HELD") || !strings.Contains(msg, "drive") {
|
||||
t.Fatalf("operator message %q does not say the app is held by an absent drive", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// A refusal-only recovery MUST NOT alarm. This is the assertion that keeps the fix from being the
|
||||
// bug it fixes: the drive gate doing its job is not a backup failure.
|
||||
func TestRecover_RefusalOnly_IsNotAlarming(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g2, _ := newGatedGuard(t, dir, map[string]string{"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint"})
|
||||
res := g2.Recover()
|
||||
|
||||
if res.Alarming() {
|
||||
t.Fatal("a recovery that only REFUSED starts reports as alarming — main.go would push it " +
|
||||
"through NotifyBackupFailed and email the customer about a working drive gate")
|
||||
}
|
||||
}
|
||||
|
||||
// A genuine failure alongside a refusal still alarms, and the two stay in different buckets.
|
||||
func TestRecover_FailureAlongsideRefusal_StillAlarmsAndKeepsThemApart(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:batch", ReasonVolumeDump, []string{"immich", "nextcloud", "homebox"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
g2, starter := newGatedGuard(t, dir, map[string]string{"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint"})
|
||||
starter.inner.failWith = map[string]error{"nextcloud": errors.New("compose up: no such image")}
|
||||
res := g2.Recover()
|
||||
|
||||
if len(res.Refused) != 1 || res.Refused[0] != "immich" {
|
||||
t.Fatalf("refused=%v, want [immich]", res.Refused)
|
||||
}
|
||||
if len(res.Failed) != 1 || res.Failed[0] != "nextcloud" {
|
||||
t.Fatalf("failed=%v, want [nextcloud]", res.Failed)
|
||||
}
|
||||
if len(res.Restarted) != 1 || res.Restarted[0] != "homebox" {
|
||||
t.Fatalf("restarted=%v, want [homebox] — neither a refusal nor a failure may abort the loop",
|
||||
res.Restarted)
|
||||
}
|
||||
if !res.Alarming() {
|
||||
t.Fatal("a genuine restart FAILURE alongside a refusal no longer alarms — the refusal " +
|
||||
"swallowed a real fault")
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared with work still owed")
|
||||
}
|
||||
// The message must not let the held app inflate the failure count.
|
||||
msg := res.Message()
|
||||
if !strings.Contains(msg, "1 of 2 app(s) could NOT be restarted") {
|
||||
t.Fatalf("operator message %q miscounts: the held app must not be counted as a failure", msg)
|
||||
}
|
||||
if !strings.Contains(msg, "not counted as failures") {
|
||||
t.Fatalf("operator message %q does not disclose the held app at all", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario B — a live drive still recovers normally, byte-identical to before -----------------
|
||||
|
||||
func TestRecover_DriveLive_RecoversExactlyAsBefore(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich", "nextcloud"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Nothing refused — the gate says yes for both.
|
||||
g2, starter := newGatedGuard(t, dir, nil)
|
||||
res := g2.Recover()
|
||||
|
||||
if len(starter.inner.starts) != 2 {
|
||||
t.Fatalf("started %v, want both apps — the new gate refused a LEGITIMATE recovery",
|
||||
starter.inner.starts)
|
||||
}
|
||||
if len(res.Refused) != 0 || len(res.Failed) != 0 {
|
||||
t.Fatalf("refused=%v failed=%v, want neither on a live drive", res.Refused, res.Failed)
|
||||
}
|
||||
if len(res.Restarted) != 2 {
|
||||
t.Fatalf("restarted=%v, want both", res.Restarted)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker survived a fully successful recovery — the next boot would restart the apps again")
|
||||
}
|
||||
if !res.Alarming() {
|
||||
t.Fatal("a successful recovery no longer reports to the operator — the interrupted operation " +
|
||||
"itself is what §2.4 wants reported, and it went silent")
|
||||
}
|
||||
}
|
||||
|
||||
// The next startup, with the drive back, completes the recovery and clears the marker. This is what
|
||||
// makes "keep the marker" a recovery rather than a leak.
|
||||
func TestRecover_HeldAppIsRestartedOnceTheDriveReturns(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Boot 1 — drive absent: refused, marker kept.
|
||||
g2, _ := newGatedGuard(t, dir, map[string]string{"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint"})
|
||||
if res := g2.Recover(); len(res.Refused) != 1 {
|
||||
t.Fatalf("boot 1 refused=%v, want [immich]", res.Refused)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("boot 1 cleared the marker — boot 2 has nothing to act on and immich stays down forever")
|
||||
}
|
||||
|
||||
// Boot 2 — the drive is back.
|
||||
g3, starter := newGatedGuard(t, dir, nil)
|
||||
res := g3.Recover()
|
||||
if len(starter.inner.starts) != 1 || starter.inner.starts[0] != "immich" {
|
||||
t.Fatalf("boot 2 started %v, want [immich] — the held app was never picked up again",
|
||||
starter.inner.starts)
|
||||
}
|
||||
if len(res.Restarted) != 1 {
|
||||
t.Fatalf("boot 2 restarted=%v, want [immich]", res.Restarted)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("boot 2 kept the marker after a fully successful recovery")
|
||||
}
|
||||
}
|
||||
|
||||
// ErrStartRefused must be matched with errors.Is, i.e. it survives wrapping. A starter that returns
|
||||
// a bare string reason would land in Failed and alarm — the exact collapse this type prevents.
|
||||
func TestErrStartRefused_SurvivesWrapping(t *testing.T) {
|
||||
err := fmt.Errorf("%w: drive /mnt/felhom-drives/hdd_1 is not a live mountpoint", ErrStartRefused)
|
||||
if !errors.Is(err, ErrStartRefused) {
|
||||
t.Fatal("a wrapped ErrStartRefused is no longer matched by errors.Is — every refusal would " +
|
||||
"be recorded as a restart failure and alarm the customer")
|
||||
}
|
||||
if errors.Is(errors.New("compose up: no such image"), ErrStartRefused) {
|
||||
t.Fatal("an ordinary restart failure matches ErrStartRefused — real faults would go silent")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,365 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ── The app-stop marker (R-166 part 2, decision D-b "in-flight operations") ───────────────────────
|
||||
//
|
||||
// Several operations stop a customer's app, do something to its data, and start it again. Between
|
||||
// the stop and the start, NOTHING ON DISK RECORDED THAT AN APP WAS OWED A RESTART. A controller that
|
||||
// died in that window left the app down with no explanation anywhere — and because a stopped app has
|
||||
// zero containers, the boot reconciler read it as a deliberate customer stop and deliberately left
|
||||
// it alone. Silently, indefinitely.
|
||||
//
|
||||
// A `defer` is NOT the fix and must never be described as one. Campaign 8 fault 10 established this
|
||||
// on live hardware: a SIGKILL runs no deferred function, and what brought the quiesce loop's stacks
|
||||
// back was its persisted marker read by Recover() one second after restart. The defer covers the
|
||||
// graceful exits; the marker covers the hard crash and the power cut. This file is that marker for
|
||||
// the app-data path, modelled directly on internal/quiesce's.
|
||||
//
|
||||
// WHY ITS OWN FILE, not quiesce's: one file, one writer. Quiesce's marker records a whole-guest
|
||||
// backup window and is written by the quiesce loop; this one records an app-data operation and is
|
||||
// written by the backup manager and the exporter. Sharing the file would give it two writers with
|
||||
// two lifetimes, and one clearing the other's record is a stranded app by a different route.
|
||||
//
|
||||
// SAFETY (D-b's binding rule): losing this file must never be worse than not having it. A lost or
|
||||
// corrupt marker means the app is not auto-restarted by THIS mechanism — which is precisely the
|
||||
// pre-v0.189.0 position, not a new hazard. It never deletes, restores, or touches a backup artifact.
|
||||
|
||||
// AppStopReason names WHY an app was stopped, so the recovery log tells an operator which operation
|
||||
// was interrupted rather than merely that something was.
|
||||
type AppStopReason string
|
||||
|
||||
const (
|
||||
// ReasonVolumeDump — DumpAppVolumesSafe: stop, tar the volumes consistently, start.
|
||||
ReasonVolumeDump AppStopReason = "volume_dump"
|
||||
// ReasonOffboxReconstitute — a full offsite restore overwriting the app's files.
|
||||
ReasonOffboxReconstitute AppStopReason = "offbox_reconstitute"
|
||||
// ReasonAppExport — a .fab export taken with "stop the app first".
|
||||
ReasonAppExport AppStopReason = "app_export"
|
||||
)
|
||||
|
||||
// humanReason is the operator-facing phrasing for each reason.
|
||||
func (r AppStopReason) humanReason() string {
|
||||
switch r {
|
||||
case ReasonVolumeDump:
|
||||
return "an app-data backup (volume dump)"
|
||||
case ReasonOffboxReconstitute:
|
||||
return "an off-site restore"
|
||||
case ReasonAppExport:
|
||||
return "an app export"
|
||||
default:
|
||||
return string(r)
|
||||
}
|
||||
}
|
||||
|
||||
// AppStopMarker is the persisted "these apps were stopped by an operation that has not reported
|
||||
// finishing — they are owed a restart" note.
|
||||
type AppStopMarker struct {
|
||||
Active bool `json:"active"`
|
||||
OpID string `json:"op_id"`
|
||||
Reason AppStopReason `json:"reason"`
|
||||
Stacks []string `json:"stacks"`
|
||||
StartedAt time.Time `json:"started_at"`
|
||||
}
|
||||
|
||||
// AppStopStarter is the one thing recovery needs: the ability to start a stack. StartStack must be
|
||||
// idempotent (it is — `compose up -d` on a running stack is a no-op).
|
||||
//
|
||||
// R-174: production MUST pass a GATED starter, never the raw stack manager. Recover runs at STARTUP —
|
||||
// exactly when an external drive may not have come back — and `Manager.StartStack` has no drive gate
|
||||
// of its own. See `gatedAppStopStarter` in cmd/controller/main.go.
|
||||
type AppStopStarter interface {
|
||||
StartStack(name string) error
|
||||
}
|
||||
|
||||
// ErrStartRefused is what a gated starter returns when a DELIBERATE HOLDER — today the drive gate —
|
||||
// says an app must not be started. Wrap it (`fmt.Errorf("%w: …", ErrStartRefused)`) so the reason
|
||||
// survives; Recover matches with errors.Is.
|
||||
//
|
||||
// IT IS NOT A FAILURE, AND THE DISTINCTION IS THE WHOLE POINT OF THE TYPE. A refusal means the
|
||||
// holder is doing its job and owns the restart; a failure means the restart was attempted and broke.
|
||||
// Collapsing the two would put a deliberately-held app into `Failed`, which main.go reports through
|
||||
// `NotifyBackupFailed` — a type that is customer-enabled by default (`settings.DefaultEnabledEvents`)
|
||||
// and carries the Hungarian "A biztonsági mentés sikertelen!". That is R-171's defect one path over:
|
||||
// a false alarm about an app the drive gate is deliberately holding. Both buckets keep the marker;
|
||||
// only `Failed` alarms.
|
||||
var ErrStartRefused = errors.New("start refused by a deliberate holder")
|
||||
|
||||
// AppStopGuard owns one marker file. Construct with NewAppStopGuard; the zero value is inert (every
|
||||
// method is a no-op on a nil guard), so a caller that was never wired degrades to pre-v0.189.0
|
||||
// behaviour instead of panicking.
|
||||
type AppStopGuard struct {
|
||||
path string
|
||||
logger *log.Logger
|
||||
now func() time.Time
|
||||
// starter is only needed by Recover; Begin/End work without one.
|
||||
starter AppStopStarter
|
||||
}
|
||||
|
||||
// AppStopRecovery is what Recover found and did. Returned rather than pushed through a notifier
|
||||
// seam, because of a hard ordering constraint: Recover must COMPLETE before the boot reconciler is
|
||||
// launched (§8.4, main.go:236) and the hub notifier is not constructed until main.go:307. A seam
|
||||
// wired after the fact would be a seam that never fires — the "built but never wired" shape this
|
||||
// project has now hit four times. Returning the outcome lets main.go report it the moment the
|
||||
// notifier exists, and makes the reporting decision visible at the call site instead of buried here.
|
||||
type AppStopRecovery struct {
|
||||
Reason AppStopReason
|
||||
OpID string
|
||||
StartedAt time.Time
|
||||
Restarted []string // apps started again by this recovery
|
||||
Failed []string // apps whose restart was ATTEMPTED and broke (the marker was kept for these)
|
||||
// Refused are apps a deliberate holder said must not start — today, an absent data drive
|
||||
// (R-174). The marker is kept for these too, but they are NOT a fault and MUST NOT alarm: the
|
||||
// holder owns the restart. Separate from Failed for the reason recorded on ErrStartRefused.
|
||||
Refused []string
|
||||
}
|
||||
|
||||
// Alarming reports whether this recovery is worth paging an operator about. A recovery that only
|
||||
// REFUSED starts is the drive gate working as designed, and reporting it through the customer-enabled
|
||||
// `backup_failed` type would be the R-171 false alarm one path over.
|
||||
func (r *AppStopRecovery) Alarming() bool {
|
||||
if r == nil {
|
||||
return false
|
||||
}
|
||||
return len(r.Failed) > 0 || len(r.Restarted) > 0
|
||||
}
|
||||
|
||||
// Message is the operator-facing headline for an interrupted operation.
|
||||
func (r *AppStopRecovery) Message() string {
|
||||
if r == nil {
|
||||
return ""
|
||||
}
|
||||
if len(r.Failed) > 0 {
|
||||
m := fmt.Sprintf("%s was interrupted by a controller restart and %d of %d app(s) could NOT be restarted",
|
||||
r.Reason.humanReason(), len(r.Failed), len(r.Restarted)+len(r.Failed))
|
||||
if len(r.Refused) > 0 {
|
||||
m += fmt.Sprintf(" (a further %d are held by an absent drive and are not counted as failures)", len(r.Refused))
|
||||
}
|
||||
return m
|
||||
}
|
||||
if len(r.Refused) > 0 && len(r.Restarted) == 0 {
|
||||
return fmt.Sprintf("%s was interrupted by a controller restart — %d app(s) are left stopped and HELD: their data drive is not available, so the drive gate restarts them when it returns",
|
||||
r.Reason.humanReason(), len(r.Refused))
|
||||
}
|
||||
m := fmt.Sprintf("%s was interrupted by a controller restart — %d app(s) were left stopped and have been restarted",
|
||||
r.Reason.humanReason(), len(r.Restarted))
|
||||
if len(r.Refused) > 0 {
|
||||
m += fmt.Sprintf("; %d more are held by an absent drive", len(r.Refused))
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Detail is the machine-readable tail. App/stack NAMES only — never env values (§9.5).
|
||||
func (r *AppStopRecovery) Detail() string {
|
||||
if r == nil {
|
||||
return ""
|
||||
}
|
||||
d := fmt.Sprintf("op=%s reason=%s started_at=%s restarted=%v", r.OpID, r.Reason,
|
||||
r.StartedAt.UTC().Format(time.RFC3339), r.Restarted)
|
||||
if len(r.Failed) > 0 {
|
||||
d += fmt.Sprintf(" restart_failed=%v", r.Failed)
|
||||
}
|
||||
if len(r.Refused) > 0 {
|
||||
d += fmt.Sprintf(" held_by_drive=%v", r.Refused)
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
// NewAppStopGuard builds a guard over the given marker path.
|
||||
func NewAppStopGuard(path string, logger *log.Logger) *AppStopGuard {
|
||||
if logger == nil {
|
||||
logger = log.Default()
|
||||
}
|
||||
return &AppStopGuard{path: path, logger: logger, now: time.Now}
|
||||
}
|
||||
|
||||
// SetStarter wires the stack-start seam used by Recover. INIT-ONLY — call once at startup, before
|
||||
// Recover. Separate from the constructor because the guard is built alongside the backup manager,
|
||||
// which learns its stack provider later (the same shape as SetStackProvider).
|
||||
func (g *AppStopGuard) SetStarter(s AppStopStarter) {
|
||||
if g == nil {
|
||||
return
|
||||
}
|
||||
g.starter = s
|
||||
}
|
||||
|
||||
// Begin records that `stacks` are about to be stopped by `reason`. It MUST be called BEFORE the
|
||||
// first stop — an error here means the marker could not be written, and the caller must not proceed
|
||||
// to stop an app it cannot promise to restart.
|
||||
func (g *AppStopGuard) Begin(opID string, reason AppStopReason, stackNames []string) error {
|
||||
if g == nil || g.path == "" {
|
||||
return nil // not wired — pre-v0.189.0 behaviour, never a hard failure
|
||||
}
|
||||
if len(stackNames) == 0 {
|
||||
return nil
|
||||
}
|
||||
return g.write(AppStopMarker{
|
||||
Active: true,
|
||||
OpID: opID,
|
||||
Reason: reason,
|
||||
Stacks: append([]string(nil), stackNames...),
|
||||
StartedAt: g.now(),
|
||||
})
|
||||
}
|
||||
|
||||
// End clears the marker after a successful restart. Best-effort by contract: a failure to clear is
|
||||
// logged, never returned as the operation's error — a stale marker costs one idempotent StartStack
|
||||
// on the next boot, which is exactly D-b's "worst acceptable outcome" and far cheaper than failing
|
||||
// a backup that actually succeeded.
|
||||
func (g *AppStopGuard) End() {
|
||||
if g == nil || g.path == "" {
|
||||
return
|
||||
}
|
||||
if err := os.Remove(g.path); err != nil && !os.IsNotExist(err) {
|
||||
g.logger.Printf("[ERROR] [appstop] could not clear the app-stop marker at %s: %v (a stale marker costs one idempotent restart at next startup)", g.path, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Recover restarts any apps left stopped by an operation that died before restarting them, then
|
||||
// clears the marker. Call ONCE at startup, and — critically — call it to COMPLETION before the boot
|
||||
// reconciler is launched, so an app this marker explains is not also reported as an unexplained boot
|
||||
// orphan (§8.4).
|
||||
//
|
||||
// Idempotent: StartStack on a running stack is tolerated, and an absent or inactive marker is a
|
||||
// no-op. On a restart FAILURE the marker is deliberately LEFT IN PLACE — the next startup retries,
|
||||
// and in the meantime the app is down with desired_state:running, so the boot reconciler sees it as
|
||||
// an orphan and the dead-app alarm owns it. Clearing a marker whose restart failed would erase the
|
||||
// only durable record that an app is owed one.
|
||||
//
|
||||
// Returns nil when there was nothing to recover — so "no interrupted operation" and "the recovery
|
||||
// never ran" are distinguishable to the caller, not only in a log (standing rule 3).
|
||||
func (g *AppStopGuard) Recover() *AppStopRecovery {
|
||||
if g == nil || g.path == "" {
|
||||
return nil
|
||||
}
|
||||
m, ok := g.read()
|
||||
if !ok || !m.Active || len(m.Stacks) == 0 {
|
||||
return nil
|
||||
}
|
||||
if g.starter == nil {
|
||||
g.logger.Printf("[ERROR] [appstop] crash recovery: %d app(s) were stopped by %s and are owed a restart, but no stack starter is wired — leaving the marker for the next startup: %v",
|
||||
len(m.Stacks), m.Reason.humanReason(), m.Stacks)
|
||||
return nil
|
||||
}
|
||||
|
||||
g.logger.Printf("[WARN] [appstop] crash recovery: %s (op %q) was interrupted and left %d app(s) stopped — restarting them: %v",
|
||||
m.Reason.humanReason(), m.OpID, len(m.Stacks), m.Stacks)
|
||||
|
||||
res := &AppStopRecovery{Reason: m.Reason, OpID: m.OpID, StartedAt: m.StartedAt}
|
||||
for _, name := range m.Stacks {
|
||||
if err := g.starter.StartStack(name); err != nil {
|
||||
// R-174: a REFUSAL is not a failure. The starter's gate has said this app must not be
|
||||
// started (an absent data drive), so the app is left down deliberately and the holder
|
||||
// owns the restart. Logged at WARN with the reason, and kept out of Failed so it never
|
||||
// reaches the customer-enabled backup_failed alarm — see ErrStartRefused.
|
||||
if errors.Is(err, ErrStartRefused) {
|
||||
g.logger.Printf("[WARN] [appstop] crash recovery: NOT restarting %s — %v; the marker is KEPT and the holder owns the restart", name, err)
|
||||
res.Refused = append(res.Refused, name)
|
||||
continue
|
||||
}
|
||||
g.logger.Printf("[ERROR] [appstop] crash recovery: restart %s failed: %v", name, err)
|
||||
res.Failed = append(res.Failed, name)
|
||||
continue
|
||||
}
|
||||
g.logger.Printf("[INFO] [appstop] crash recovery: restarted %s after the interrupted %s", name, m.Reason.humanReason())
|
||||
res.Restarted = append(res.Restarted, name)
|
||||
}
|
||||
sort.Strings(res.Failed)
|
||||
sort.Strings(res.Refused)
|
||||
sort.Strings(res.Restarted)
|
||||
|
||||
// The marker is kept for BOTH unfinished outcomes, for the same reason and with different
|
||||
// urgency: a failed restart is retried next startup, and a refused one is genuinely unfinished
|
||||
// until its drive returns. Clearing it in either case would erase the only durable record that
|
||||
// an app is owed a restart.
|
||||
if len(res.Failed) > 0 {
|
||||
g.logger.Printf("[ERROR] [appstop] crash recovery: %d app(s) could not be restarted — KEEPING the marker so the next startup retries; the dead-app alarm owns them meanwhile: %v",
|
||||
len(res.Failed), res.Failed)
|
||||
return res
|
||||
}
|
||||
if len(res.Refused) > 0 {
|
||||
g.logger.Printf("[WARN] [appstop] crash recovery: %d app(s) were deliberately NOT restarted (drive absent) — KEEPING the marker; this is the gate working, not a fault: %v",
|
||||
len(res.Refused), res.Refused)
|
||||
return res
|
||||
}
|
||||
g.End()
|
||||
return res
|
||||
}
|
||||
|
||||
// HeldStacks returns the stacks an app-data operation is CURRENTLY holding down, or nil.
|
||||
//
|
||||
// Read-only and nil-safe. It exists for the boot reconciler (§8.2): once R-157 mechanism A widened
|
||||
// the boot window, the sweep could overlap a running volume dump or export and "recover" an app that
|
||||
// is deliberately stopped mid-operation — restarting it under a tar, which is the inconsistency the
|
||||
// stop was taken to avoid. Recover() has already run to completion by then, so a marker seen through
|
||||
// this method belongs to an operation running NOW, not to a crashed one.
|
||||
func (g *AppStopGuard) HeldStacks() []string {
|
||||
if g == nil || g.path == "" {
|
||||
return nil
|
||||
}
|
||||
m, ok := g.read()
|
||||
if !ok || !m.Active {
|
||||
return nil
|
||||
}
|
||||
return append([]string(nil), m.Stacks...)
|
||||
}
|
||||
|
||||
// ---- marker persistence (atomic, 0600) — the quiesce shape ------------------------------------
|
||||
|
||||
func (g *AppStopGuard) write(m AppStopMarker) error {
|
||||
data, err := json.MarshalIndent(m, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(g.path), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
tmp := g.path + ".tmp"
|
||||
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := f.Write(data); err != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
// fsync before rename: the whole point is surviving a power cut, and a rename that lands ahead
|
||||
// of the bytes it points at is a marker that reads as corrupt at exactly the wrong moment.
|
||||
if err := f.Sync(); err != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp, g.path)
|
||||
}
|
||||
|
||||
func (g *AppStopGuard) read() (AppStopMarker, bool) {
|
||||
data, err := os.ReadFile(g.path)
|
||||
if err != nil {
|
||||
return AppStopMarker{}, false
|
||||
}
|
||||
var m AppStopMarker
|
||||
if err := json.Unmarshal(data, &m); err != nil {
|
||||
// Never a silent skip (§9.4): a corrupt marker is LOUD and the bad file is quarantined, so a
|
||||
// genuinely interrupted operation leaves a trace instead of vanishing. Still returns false —
|
||||
// "no usable marker ⇒ no recovery" is the correct contract, and matches quiesce's.
|
||||
g.logger.Printf("[WARN] [appstop] the app-stop marker at %s is corrupt (%v) — quarantining; apps are NOT auto-restarted from it", g.path, err)
|
||||
_ = os.Rename(g.path, fmt.Sprintf("%s.corrupt-%d", g.path, g.now().Unix()))
|
||||
return AppStopMarker{}, false
|
||||
}
|
||||
return m, true
|
||||
}
|
||||
@@ -0,0 +1,395 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-166 part 2 — the app-stop crash marker.
|
||||
//
|
||||
// THE DISCIPLINE THAT MATTERS HERE (§10): a `defer` is not crash-safety, so a test that lets the
|
||||
// deferred cleanup run proves nothing about a crash. Every "interrupted" test below simulates a
|
||||
// SIGKILL by never reaching the restart — the marker is written, the process conceptually dies, and
|
||||
// a FRESH guard over the SAME file does the recovering. That is exactly what Campaign 8 fault 10
|
||||
// established on live hardware: a SIGKILL runs no deferred function, and what brought the stacks
|
||||
// back was the marker read at startup.
|
||||
|
||||
type fakeStarter struct {
|
||||
starts []string
|
||||
failWith map[string]error
|
||||
}
|
||||
|
||||
func (f *fakeStarter) StartStack(name string) error {
|
||||
f.starts = append(f.starts, name)
|
||||
if err := f.failWith[name]; err != nil {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func newGuard(t *testing.T, dir string) (*AppStopGuard, *fakeStarter) {
|
||||
t.Helper()
|
||||
s := &fakeStarter{}
|
||||
g := NewAppStopGuard(filepath.Join(dir, "appstop-state.json"), log.New(io.Discard, "", 0))
|
||||
g.SetStarter(s)
|
||||
return g, s
|
||||
}
|
||||
|
||||
func markerPath(dir string) string { return filepath.Join(dir, "appstop-state.json") }
|
||||
|
||||
func markerExists(t *testing.T, dir string) bool {
|
||||
t.Helper()
|
||||
_, err := os.Stat(markerPath(dir))
|
||||
if err != nil && !os.IsNotExist(err) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// --- Scenario E — a crash mid-backup brings the app back -----------------------------------------
|
||||
|
||||
func TestRecover_InterruptedVolumeDump_RestartsTheAppAndClearsTheMarker(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
|
||||
// --- process 1: an operation stops the app and is KILLED. No End(), no defer, no cleanup. ---
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatalf("Begin: %v", err)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("Begin did not write a marker — nothing would survive the kill")
|
||||
}
|
||||
// <SIGKILL here> — g1 is abandoned deliberately; nothing else is called on it.
|
||||
|
||||
// --- process 2: a fresh controller starts and recovers from the file alone. ---
|
||||
g2, starter := newGuard(t, dir)
|
||||
res := g2.Recover()
|
||||
|
||||
if len(starter.starts) != 1 || starter.starts[0] != "immich" {
|
||||
t.Fatalf("started %v, want exactly [immich] — the app was left stranded by the interrupted backup", starter.starts)
|
||||
}
|
||||
if res == nil || len(res.Restarted) != 1 || res.Restarted[0] != "immich" {
|
||||
t.Fatalf("recovery result = %+v, want immich restarted", res)
|
||||
}
|
||||
if res.Reason != ReasonVolumeDump {
|
||||
t.Fatalf("reason = %q, want %q — the operator must be told WHICH operation was interrupted", res.Reason, ReasonVolumeDump)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker survived a successful recovery — the next boot would restart the app again")
|
||||
}
|
||||
// The operator-facing text must name the interruption, not merely report a restart.
|
||||
if msg := res.Message(); msg == "" || !strings.Contains(msg, "interrupted") {
|
||||
t.Fatalf("operator message %q does not say the operation was interrupted", msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_NoMarker_IsASilentNoOp(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g, starter := newGuard(t, dir)
|
||||
if res := g.Recover(); res != nil {
|
||||
t.Fatalf("Recover reported %+v on a box with no marker", res)
|
||||
}
|
||||
if len(starter.starts) != 0 {
|
||||
t.Fatalf("started %v with no marker present", starter.starts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_FailedRestart_KEEPSTheMarkerForTheNextStartup(t *testing.T) {
|
||||
// The single most important failure behaviour: clearing a marker whose restart failed would
|
||||
// erase the only durable record that an app is owed one. The app is genuinely still down.
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich", "nextcloud"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
g2, starter := newGuard(t, dir)
|
||||
starter.failWith = map[string]error{"immich": errors.New("compose up: no such image")}
|
||||
res := g2.Recover()
|
||||
|
||||
if len(res.Failed) != 1 || res.Failed[0] != "immich" {
|
||||
t.Fatalf("failed=%v, want [immich]", res.Failed)
|
||||
}
|
||||
if len(res.Restarted) != 1 || res.Restarted[0] != "nextcloud" {
|
||||
t.Fatalf("restarted=%v, want [nextcloud] — one app failing must not abort the others", res.Restarted)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared even though a restart FAILED — the next startup would not retry")
|
||||
}
|
||||
if msg := res.Message(); !strings.Contains(msg, "NOT be restarted") {
|
||||
t.Fatalf("operator message %q does not report the failure", msg)
|
||||
}
|
||||
if d := res.Detail(); !strings.Contains(d, "restart_failed") || !strings.Contains(d, "immich") {
|
||||
t.Fatalf("detail %q does not name which app failed", d)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_IsIdempotentAcrossRepeatedStartups(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("op", ReasonOffboxReconstitute, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g2, s2 := newGuard(t, dir)
|
||||
g2.Recover()
|
||||
g3, s3 := newGuard(t, dir)
|
||||
g3.Recover()
|
||||
|
||||
if len(s2.starts) != 1 {
|
||||
t.Fatalf("first recovery started %v", s2.starts)
|
||||
}
|
||||
if len(s3.starts) != 0 {
|
||||
t.Fatalf("a SECOND startup restarted %v again — the marker was not cleared", s3.starts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_CorruptMarkerIsQuarantinedNotSilentlySkipped(t *testing.T) {
|
||||
// §9.4: never a silent skip. A corrupt marker cannot be acted on, but it must leave a trace —
|
||||
// otherwise a genuinely interrupted operation vanishes without evidence.
|
||||
dir := t.TempDir()
|
||||
if err := os.WriteFile(markerPath(dir), []byte("{not json"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g, starter := newGuard(t, dir)
|
||||
if res := g.Recover(); res != nil {
|
||||
t.Fatalf("a corrupt marker produced a recovery result %+v", res)
|
||||
}
|
||||
if len(starter.starts) != 0 {
|
||||
t.Fatalf("apps were started from a corrupt marker: %v", starter.starts)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the corrupt marker was left in place — it would be re-read forever")
|
||||
}
|
||||
quarantined, _ := filepath.Glob(markerPath(dir) + ".corrupt-*")
|
||||
if len(quarantined) != 1 {
|
||||
t.Fatalf("the corrupt marker was not quarantined (found %d) — it was silently dropped", len(quarantined))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_NoStarterWiredKeepsTheMarker(t *testing.T) {
|
||||
// D-b's safety rule: never worse than not having the file. With no starter the guard cannot act,
|
||||
// so it must keep the record for a startup that can, rather than clear it and lose the app.
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("op", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g2 := NewAppStopGuard(markerPath(dir), log.New(io.Discard, "", 0)) // deliberately no SetStarter
|
||||
if res := g2.Recover(); res != nil {
|
||||
t.Fatalf("recovered without a starter: %+v", res)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared with no starter wired — the app would never come back")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNilGuardIsInert(t *testing.T) {
|
||||
// A caller that was never wired must degrade to pre-v0.189.0 behaviour, not panic.
|
||||
var g *AppStopGuard
|
||||
if err := g.Begin("op", ReasonVolumeDump, []string{"x"}); err != nil {
|
||||
t.Fatalf("nil guard Begin returned %v", err)
|
||||
}
|
||||
g.End()
|
||||
if res := g.Recover(); res != nil {
|
||||
t.Fatalf("nil guard recovered %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkerContentsAreDiagnosable(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g, _ := newGuard(t, dir)
|
||||
if err := g.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, err := os.ReadFile(markerPath(dir))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var m AppStopMarker
|
||||
if err := json.Unmarshal(raw, &m); err != nil {
|
||||
t.Fatalf("the marker on disk is not readable JSON: %v", err)
|
||||
}
|
||||
if !m.Active || m.OpID != "volume-dump:immich" || m.Reason != ReasonVolumeDump ||
|
||||
len(m.Stacks) != 1 || m.Stacks[0] != "immich" || m.StartedAt.IsZero() {
|
||||
t.Fatalf("the marker does not record enough to diagnose the interruption: %+v", m)
|
||||
}
|
||||
// 0600 — it names customer apps.
|
||||
fi, err := os.Stat(markerPath(dir))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if fi.Mode().Perm() != 0o600 {
|
||||
t.Fatalf("marker mode = %v, want 0600", fi.Mode().Perm())
|
||||
}
|
||||
}
|
||||
|
||||
func TestBeginWithNoStacksWritesNothing(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g, _ := newGuard(t, dir)
|
||||
if err := g.Begin("op", ReasonVolumeDump, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("a marker was written for an operation that stops nothing")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenarios E/F — DumpAppVolumesSafe, the primary site ----------------------------------------
|
||||
|
||||
// inspectingProvider is the StackDataProvider slice DumpAppVolumesSafe touches. It records whether
|
||||
// the marker file EXISTED at each step — the positive observable for the ordering property. An
|
||||
// absent log line is not evidence (standing rule 3); the file's presence at the moment of the stop
|
||||
// is.
|
||||
//
|
||||
// GetDockerVolumes returns nothing, so the dump itself is a no-op and no Docker is involved — the
|
||||
// stop/start bracket around it is what is under test.
|
||||
type inspectingProvider struct {
|
||||
StackDataProvider
|
||||
markerFile string
|
||||
events []string
|
||||
stopErr error
|
||||
startErr error
|
||||
markerPresentAtStop bool
|
||||
markerAtStartCall bool
|
||||
// panicOnVolumes simulates a hard abort (SIGKILL/power cut) at the point the dump begins: the
|
||||
// unwind skips the restart statement, exactly as a kill would.
|
||||
panicOnVolumes bool
|
||||
}
|
||||
|
||||
func (p *inspectingProvider) GetDockerVolumes(string) []string {
|
||||
if p.panicOnVolumes {
|
||||
panic("simulated hard abort mid-dump")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (p *inspectingProvider) StopStack(name string) error {
|
||||
_, err := os.Stat(p.markerFile)
|
||||
p.markerPresentAtStop = err == nil
|
||||
p.events = append(p.events, "stop:"+name)
|
||||
return p.stopErr
|
||||
}
|
||||
|
||||
func (p *inspectingProvider) StartStack(name string) error {
|
||||
_, err := os.Stat(p.markerFile)
|
||||
p.markerAtStartCall = err == nil
|
||||
p.events = append(p.events, "start:"+name)
|
||||
return p.startErr
|
||||
}
|
||||
|
||||
func newDumpManager(t *testing.T, dir string, p *inspectingProvider) *Manager {
|
||||
t.Helper()
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
m := &Manager{logger: lg, stackProvider: p, systemDataPath: dir}
|
||||
m.appStop = NewAppStopGuard(markerPath(dir), lg)
|
||||
return m
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_MarkerCoversTheWholeStopStartWindow(t *testing.T) {
|
||||
// Scenario F, the happy path: the marker is on disk BEFORE the stop, still on disk for the whole
|
||||
// time the app is down, and GONE once the restart succeeds.
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir)}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
if err := m.DumpAppVolumesSafe("immich"); err != nil {
|
||||
t.Fatalf("DumpAppVolumesSafe: %v", err)
|
||||
}
|
||||
|
||||
if !p.markerPresentAtStop {
|
||||
t.Fatal("the marker was NOT on disk when the app was stopped — a crash one instruction later " +
|
||||
"strands the app, which is the entire failure this marker exists to prevent")
|
||||
}
|
||||
if !p.markerAtStartCall {
|
||||
t.Fatal("the marker was already gone while the app was still down")
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker survived a dump whose restart succeeded — the next boot would restart the app again")
|
||||
}
|
||||
if len(p.events) != 2 || p.events[0] != "stop:immich" || p.events[1] != "start:immich" {
|
||||
t.Fatalf("events=%v, want [stop:immich start:immich]", p.events)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_Interrupted_RecoveryBringsTheAppBack(t *testing.T) {
|
||||
// Scenario E end-to-end THROUGH THE PRODUCTION PATH, and WITHOUT running any cleanup.
|
||||
//
|
||||
// The abort is real: GetDockerVolumes panics, which unwinds out of DumpAppVolumesSafe AFTER the
|
||||
// marker was written and the app stopped, and BEFORE the restart statement — and because that
|
||||
// restart is a plain statement, not a defer, it never runs. That is the shape of a hard kill.
|
||||
//
|
||||
// The earlier version of this test called m.appStop.Begin itself, which meant it proved the
|
||||
// marker type worked and NOT that DumpAppVolumesSafe uses it — it survived the red-proof that
|
||||
// deleted the production Begin call. Driving the real function is what makes the proof bite.
|
||||
//
|
||||
// RED-PROOF: delete the `m.appStop.Begin(...)` call from DumpAppVolumesSafe and this test fails —
|
||||
// nothing is written, so nothing is recovered. Demonstrated in REPORT.md §5.
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir), panicOnVolumes: true}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
func() {
|
||||
defer func() {
|
||||
if recover() == nil {
|
||||
t.Error("the simulated abort did not fire — this test proves nothing")
|
||||
}
|
||||
}()
|
||||
_ = m.DumpAppVolumesSafe("immich")
|
||||
}()
|
||||
|
||||
if !p.markerPresentAtStop {
|
||||
t.Fatal("the app was stopped before any marker existed")
|
||||
}
|
||||
if p.markerAtStartCall {
|
||||
t.Fatal("the restart ran despite the abort — the simulation is wrong, not the code")
|
||||
}
|
||||
|
||||
// <the controller is gone> — a fresh one starts and recovers from the file alone.
|
||||
g, starter := newGuard(t, dir)
|
||||
res := g.Recover()
|
||||
|
||||
if len(starter.starts) != 1 || starter.starts[0] != "immich" {
|
||||
t.Fatalf("started %v — the app stopped by the interrupted dump was not brought back", starter.starts)
|
||||
}
|
||||
if res == nil || res.Reason != ReasonVolumeDump {
|
||||
t.Fatalf("recovery did not name the volume dump as the interrupted operation: %+v", res)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker was not cleared after a successful recovery")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_FailedRestartKeepsTheMarker(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir), startErr: errors.New("compose up failed")}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
if err := m.DumpAppVolumesSafe("immich"); err == nil {
|
||||
t.Fatal("a failed restart must surface as an error")
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared even though the restart FAILED — the app is still down and " +
|
||||
"nothing records that it is owed a restart")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_FailedStopClearsTheMarker(t *testing.T) {
|
||||
// Nothing was stopped, so nothing is owed a restart. A stranded marker here would cost a
|
||||
// spurious restart at the next startup AND a false "a backup was interrupted" alert.
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir), stopErr: errors.New("stack is protected")}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
if err := m.DumpAppVolumesSafe("traefik"); err == nil {
|
||||
t.Fatal("a failed stop must surface as an error")
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("a marker was left behind for an app that was never stopped")
|
||||
}
|
||||
}
|
||||
@@ -9,10 +9,12 @@ import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||||
)
|
||||
|
||||
// Manager orchestrates app-data backups: database dumps and Docker-volume tars.
|
||||
@@ -31,9 +33,104 @@ type Manager struct {
|
||||
// tier2Notify, if set, is called after each Tier 2 copy (success: err==nil) for notifications.
|
||||
tier2Notify func(stackName, destLabel string, dur time.Duration, err error)
|
||||
|
||||
// unitNotify (R-158 / R-167), if set, is called ONCE PER APP whose Tier-1 recovery-unit capture
|
||||
// FAILED, and the capture loop continues to the next app. Wired in cmd/controller/main.go.
|
||||
//
|
||||
// WHY IT EXISTS. `/backups/apps` is the page a person opens to ask whether ONE app is backed up,
|
||||
// and until now it was the one page that never said: a per-app capture failure was a `[WARN]`
|
||||
// line and went no further. The manager had three notify seams and none for the unit capture —
|
||||
// the FIFTH instance in this project of a mechanism built and left disconnected.
|
||||
//
|
||||
// IT CARRIES THE SPACE FIGURES DELIBERATELY. The overwhelmingly likely cause is a full
|
||||
// filesystem, and an operator who has the used/free bytes at the moment of failure can act
|
||||
// without logging in. It is the same pair of numbers the customer-facing fill warning reports,
|
||||
// which is why the two ship together.
|
||||
//
|
||||
// OPERATOR-TIER. Routed to a hub event type that is in `notify.operatorOnlyEvents` — a customer
|
||||
// can take no action on a capture failure. Deliberately NOT `backup_failed`, which is
|
||||
// customer-enabled by default and would email them in Hungarian about it (D-c).
|
||||
//
|
||||
// NO CONTROLLER-SIDE COOLDOWN — the hub owns cooldown, per the offboxEnlargeBlockedNotify
|
||||
// precedent.
|
||||
unitNotify func(stackName string, err error, usage *UnitSpace)
|
||||
|
||||
// unitSpaceFn (R-165 / B2), if set, replaces the real statfs behind the capture floor so a test
|
||||
// can state a filesystem's occupancy as an input. Nil in production → `unitTargetSpace`.
|
||||
unitSpaceFn func(stackName string) *UnitSpace
|
||||
|
||||
// admission (R-181) is the per-RUN memo of the reserve's per-app verdict, guarded by admissionMu.
|
||||
// Non-nil only for the duration of a backup run (beginAdmissionRun → its closer). One verdict per
|
||||
// app covers all THREE write legs — DB dump, volume dump, unit capture — because all three write
|
||||
// under one per-app root; see admission.go for why it is decided lazily and never re-decided.
|
||||
admissionMu sync.Mutex
|
||||
admission *admissionSet
|
||||
|
||||
// summary (R-182) is the per-RUN digest collector, guarded by summaryMu. Same lifetime as
|
||||
// `admission` and for the same reason: an absent collector means "no run in flight", never a
|
||||
// stale answer from last night. runSummaryNotify is the operator digest seam, wired in main.go.
|
||||
summaryMu sync.Mutex
|
||||
summary *runSummary
|
||||
runSummaryNotify func(RunSummary)
|
||||
// manualRun tags the NEXT run as operator-triggered (cleared as the run starts), so the digest
|
||||
// can say which kind it was and the hub can decline to collapse a manual run into a nightly one.
|
||||
manualRun atomic.Bool
|
||||
|
||||
// appStop (R-166) is the crash marker for operations that stop an app, work on its data, and
|
||||
// start it again. Written BEFORE the stop and cleared AFTER the restart, so a SIGKILL or a power
|
||||
// cut in that window leaves a durable record that Recover honours at the next startup. Built in
|
||||
// NewManager from cfg.Paths.DataDir — see appstop_marker.go for why it is not quiesce's file.
|
||||
appStop *AppStopGuard
|
||||
|
||||
// offbox (Part B): the restic-SFTP exec seam (nil → real restic) + the failure→operator-alert hook.
|
||||
offboxRunner offboxRunner
|
||||
offboxNotify func(dur time.Duration, snapshots int, err error)
|
||||
// offboxStreamRunner + offboxProgress: the MANUAL run's live progress (v0.147.0, 4c). The stream
|
||||
// seam scans restic's `--json` stdout line-by-line; the state is what the page polls. Both are
|
||||
// inert on the nightly path — the sink is installed only for the duration of a manual run.
|
||||
offboxStreamRunner offboxStreamRunner
|
||||
offboxProgress offboxProgressState
|
||||
// offboxOrphanEvent (v0.142.0), if set, pushes a hub event on offsite-repo continuity transitions
|
||||
// ("offbox_repo_orphaned" / "offbox_repo_reset"); renamedTo names the move-aside path (reset only).
|
||||
// Wired in main.go to the notifier. Nil-safe.
|
||||
offboxOrphanEvent func(eventType, renamedTo string)
|
||||
// offboxGapNotify (R-203) fires when a COMPLETED offsite run could not capture a directory an
|
||||
// app declares MANDATORY — a coverage gap, not a failed run. nil → no signal.
|
||||
offboxGapNotify func(gaps map[string][]string)
|
||||
// offboxSSH (v0.142.0) is the raw-ssh exec seam for the orphaned-repo move-aside (restic has no
|
||||
// rename); tests inject a fake. Nil → the real ssh invocation (defaultOffboxSSH).
|
||||
offboxSSH func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error)
|
||||
|
||||
// offboxSizer (3a) — the mandatory-set byte estimator for the pre-push enlargement gate, overridable
|
||||
// in tests so the gate is unit-testable without a real du. Nil → the real dirSizeBytes (du -sb).
|
||||
offboxSizer func(path string) int64
|
||||
// offboxNow (v0.206.0, R-241) is the abandonment countdown's clock. Nil → time.Now.
|
||||
//
|
||||
// IT EXISTS SO THE TERMINAL STEP IS TESTABLE WITHOUT SHORTENING A LIVE TIMER (§7.4). The sweep is
|
||||
// the only thing in the product that deletes a customer's off-site history; driving it with a
|
||||
// clock keeps that step exercised on every run of the suite instead of once, on real data, by an
|
||||
// operator who then has to hope.
|
||||
offboxNow func() time.Time
|
||||
// offboxEnlargeBlockedNotify (3a), if set, is called ONCE per app that NEWLY enters the
|
||||
// quota-blocked (enlargement-refused) state — edge-triggered against the persisted EnlargedBlocked
|
||||
// set so a nightly schedule can't re-notify a persistently-blocked app (the hub owns cooldown; the
|
||||
// controller must not add a timer). Wired in cmd/controller/main.go.
|
||||
offboxEnlargeBlockedNotify func(stack string, estBytes int64, usedGB, quotaGB int)
|
||||
// offboxPlaceCopier (3a) — the place-to-live missing-only merge seam (nil → rsyncRestoreMissing,
|
||||
// the `-a --ignore-existing` additive copy). Never rsyncMirror (--delete trap).
|
||||
offboxPlaceCopier func(src, dst string) (int, error)
|
||||
// offboxFullPlaceCopier (R-43, v0.148.0) — the FULL-restore overwrite seam (nil →
|
||||
// rsyncRestoreOverwrite: `-a` with NO --ignore-existing and NO --delete). Distinct from
|
||||
// offboxPlaceCopier on purpose: the two have opposite semantics for an existing file.
|
||||
offboxFullPlaceCopier func(src, dst string) (int, error)
|
||||
// safetyDumpFn (R-43) — the pre-restore safety-dump seam (nil → the real DumpOne), so the
|
||||
// "never replay without an undo on disk" refusal is unit-testable without Docker.
|
||||
safetyDumpFn func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult
|
||||
// offsitePreDumpFn (R-44) — the offsite dump pre-phase seam (nil → runDBDumpsInternal), so the
|
||||
// dumps-strictly-before-capture ordering is observable in a test without Docker or restic.
|
||||
offsitePreDumpFn func(ctx context.Context) error
|
||||
// offboxFreeFn (3a) — the free-space probe for the restore headroom gate, overridable in tests (the
|
||||
// Windows `go test` host has no `df`). Nil → the real diskFreeBytes (df --output=avail).
|
||||
offboxFreeFn func(path string) int64
|
||||
|
||||
// F17 restore seams — overridable in tests so the .sql re-import orchestration can be unit-tested
|
||||
// without Docker. Default to the real DiscoverDatabases / ImportDump (lazy-init in reimportDBDumps).
|
||||
@@ -44,6 +141,16 @@ type Manager struct {
|
||||
// disconnected) can be unit-tested without Docker. Nil → the real DumpAppVolumesSafe.
|
||||
dumpVolumesSafe func(stackName string) error
|
||||
|
||||
// F7 tar seam — the ONE docker exec inside DumpAppVolumes, overridable so the atomic-write
|
||||
// behaviour (tmp+fsync+rename; the last good `.tar` survives a mid-write failure) is unit-testable
|
||||
// without Docker. It must write the tar to `<dumpDir>/<volName>.tar.tmp` and return the combined
|
||||
// output + error. Nil → the real `docker run … alpine tar cf …tar.tmp`.
|
||||
tarVolume func(volName, dumpDir string) ([]byte, error)
|
||||
|
||||
// F6 per-app tier-2 seam — overridable so RunAllTier2's app SELECTION (now incl. volume-only apps)
|
||||
// is unit-testable without rsync/du. Nil → the real RunTier2.
|
||||
perAppTier2 func(stackName string) error
|
||||
|
||||
// generateSecret (O4), if set, produces a replacement value for a RESETTABLE secret that could
|
||||
// not be recovered during restore-from-unit (wired to stacks.Manager.GenerateSecretForField in
|
||||
// main.go). Nil / ok=false → the secret stays absent and the restore proceeds with a loud WARN.
|
||||
@@ -54,6 +161,38 @@ type Manager struct {
|
||||
// the orchestration never shells out. Nil → the real rsyncRestoreMissing (additive-only).
|
||||
restoreFilesCopier func(src, dst string) (filesRestored int, err error)
|
||||
|
||||
// tier2Mirror (F-S2) — the Tier-2 backup mirror seam (both rsync legs in RunTier2), overridable
|
||||
// so the resolve→mirror→record flow is unit-testable without rsync. Nil → the real rsyncMirror
|
||||
// (`-a --delete`, contents-of-src semantics).
|
||||
tier2Mirror func(src, dst string) error
|
||||
|
||||
// sharesPassdbCapture (R-7b) — the samba passdb capture seam (a `docker exec … tar cf -`),
|
||||
// overridable so the shares payload builder is unit-testable without docker. Nil → the real
|
||||
// defaultSharesPassdbCapture. Best-effort by contract: an error yields a manifest-only payload.
|
||||
sharesPassdbCapture func() ([]byte, error)
|
||||
|
||||
// sharesPassdbRestore (R-7b) — the mirror seam for putting a captured passdb archive BACK into the
|
||||
// samba named volume (`docker exec -i … tar xf -`). Nil → the real defaultSharesPassdbRestore.
|
||||
sharesPassdbRestore func(tar []byte) error
|
||||
|
||||
// sharesReconcile (R-7b), if set, re-renders and applies the samba stack after a shares restore
|
||||
// re-adds definitions to the registry (wired in main.go to stacks.Manager.ReconcileSamba). It is a
|
||||
// SEAM rather than a direct call because the backup package must not depend on the stacks package.
|
||||
// Nil → the registry is updated and a WARN says smb.conf will catch up on the next health tick.
|
||||
sharesReconcile func() error
|
||||
|
||||
// tier2SSDFits (3b) — the SSD-headroom predicate seam, overridable in tests (system.GetDiskUsage is
|
||||
// Linux-only → nil on the Windows test host, which would always refuse the SSD branch). Nil → the
|
||||
// real tier2FitsSystemDrive.
|
||||
tier2SSDFits func(sys string, sizeBytes int64) bool
|
||||
|
||||
// samePhysicalDevice — the off-drive identity predicate behind every Tier-2 "is this really a
|
||||
// SECOND disk?" guard, overridable in tests. The real check is `st_dev` equality, so on a host
|
||||
// where every `t.TempDir()` lands on one filesystem the fixture's "two drives" are indistinguishable
|
||||
// and Tier-2 correctly refuses them — which makes the off-drive tests unrunnable rather than wrong.
|
||||
// Nil → the real system.SamePhysicalDevice (production always takes this path).
|
||||
samePhysicalDevice func(a, b string) bool
|
||||
|
||||
// migrationRunning, if set, reports whether a data migration is in progress. The scheduled
|
||||
// backup paths skip when it returns true (Change 3 — backup ↔ migration mutual exclusion), so a
|
||||
// nightly dump/Tier-2 can't race a migration copy/cleanup on the same drive.
|
||||
@@ -63,6 +202,13 @@ type Manager struct {
|
||||
lastDBDump *DBDumpStatus
|
||||
running bool
|
||||
|
||||
// R-43/R-44 (v0.148.0) — the coherence stamp of the offsite run in flight, read by
|
||||
// CaptureRecoveryUnit so each unit records WHICH run took the dumps sitting beside its files.
|
||||
// Set for the duration of the dump pre-phase + capture, cleared after; "" means "no offsite run
|
||||
// is establishing coherence right now" (the periodic refresh and the local 02:30 dump leg).
|
||||
offsiteRunID string
|
||||
offsiteRunDumpAt string
|
||||
|
||||
// Restore op-status (Part B, opstatus.go) — display-only async-restore progress, under `mu`.
|
||||
opRunning bool
|
||||
opName string
|
||||
@@ -95,8 +241,21 @@ type FullBackupStatus struct {
|
||||
// Flash messages (set by handlers, passed through redirect)
|
||||
FlashSuccess string
|
||||
FlashError string
|
||||
|
||||
// SingleCopyWarning (F6, CAMPAIGN-3) is a non-empty honest Hungarian notice when the box has NO
|
||||
// off-drive target at all — tier-1 is then the ONLY local copy and 3-2-1 needs a 2nd drive or
|
||||
// offsite. Empty when an off-drive (tier-2) target exists. Never a fake 3-2-1 guarantee.
|
||||
SingleCopyWarning string
|
||||
}
|
||||
|
||||
// systemDriveLabel is the human label for the internal SSD / system drive (F6 — a sys_drive app's
|
||||
// backup used to render with a blank drive label). Matches the tier-2 UI wording.
|
||||
const systemDriveLabel = "Belső SSD (rendszer)"
|
||||
|
||||
// singleCopyNotice is the honest single-drive signal (F6): shown when no off-drive tier-2 target
|
||||
// exists, instead of silently implying a 3-2-1 guarantee the box cannot provide.
|
||||
const singleCopyNotice = "Csak egy másolat készül (nincs második meghajtó) — a 3-2-1 mentéshez csatlakoztasson egy második meghajtót vagy offsite tárolót."
|
||||
|
||||
// DBDumpStatus holds the last DB dump result.
|
||||
type DBDumpStatus struct {
|
||||
LastRun time.Time
|
||||
@@ -116,10 +275,30 @@ func NewManager(cfg *config.Config, sett *settings.Settings, logger *log.Logger)
|
||||
settings: sett,
|
||||
systemDataPath: cfg.Paths.SystemDataPath,
|
||||
}
|
||||
// R-166: its OWN file next to quiesce-state.json, never inside it — one file, one writer.
|
||||
m.appStop = NewAppStopGuard(filepath.Join(cfg.Paths.DataDir, "appstop-state.json"), logger)
|
||||
m.reconcileCrashedRun()
|
||||
return m
|
||||
}
|
||||
|
||||
// AppStopGuard exposes the app-stop crash marker so the exporter (a different package with the same
|
||||
// stop-work-start shape) can share the one marker file rather than opening a second one.
|
||||
func (m *Manager) AppStopGuard() *AppStopGuard { return m.appStop }
|
||||
|
||||
// SetAppStopGuard injects the guard instead of using the one NewManager built. INIT-ONLY — call once
|
||||
// during single-threaded startup, before any backup runs.
|
||||
//
|
||||
// It exists because of a startup ORDERING constraint, not for testing: the guard's Recover must
|
||||
// complete before the boot reconciler is launched (main.go:~236) and this manager is not constructed
|
||||
// until ~line 272. So main.go builds the guard early, recovers, and hands the SAME object here —
|
||||
// rather than a second guard over the same file, which would be one file with two owners, the exact
|
||||
// shape this marker was kept out of quiesce's file to avoid.
|
||||
func (m *Manager) SetAppStopGuard(g *AppStopGuard) {
|
||||
if g != nil {
|
||||
m.appStop = g
|
||||
}
|
||||
}
|
||||
|
||||
// reconcileCrashedRun makes the persisted offbox status truthful after a crash (campaign C1): a controller
|
||||
// that died mid-run left LastStatus="running" on disk (the in-memory single-flight mutex is gone with the
|
||||
// process, but the persisted status keeps lying "running" forever). Flip it to error with a Hungarian
|
||||
@@ -160,7 +339,10 @@ func (m *Manager) GetAppDrivePath(stackName string) string {
|
||||
// as-is; only the SSD-only system-data fallback gets the felhom-data subdir appended. This is what
|
||||
// keeps a drive-resident app's backups single-nested instead of .../felhom-data/felhom-data/... .
|
||||
func (m *Manager) namespaceRoot(drivePath string) string {
|
||||
return NamespaceRoot(drivePath, drivePath != m.systemDataPath)
|
||||
// R-203: delegates to the ONE expression of the rule (appbackup.NamespaceRootFor). This used to
|
||||
// hold its own copy — `drivePath != m.systemDataPath`, without Clean on either side — while
|
||||
// stacks.Manager.inGuest held a second copy WITH Clean. Two copies that already differed.
|
||||
return NamespaceRootFor(drivePath, m.systemDataPath)
|
||||
}
|
||||
|
||||
// AppNamespaceRoot returns the felhom-data namespace root for a stack's keep-side backups, resolving
|
||||
@@ -234,11 +416,48 @@ func (m *Manager) RunDBDumps(ctx context.Context) error {
|
||||
return m.runDBDumpsInternal(ctx)
|
||||
}
|
||||
|
||||
// offsiteRunStamp returns the in-flight offsite run's coherence stamp ("" when none).
|
||||
func (m *Manager) offsiteRunStamp() (runID, dumpsAt string) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
return m.offsiteRunID, m.offsiteRunDumpAt
|
||||
}
|
||||
|
||||
// beginOffsiteRunStamp marks the start of an offsite run's coherence window and returns the cleanup.
|
||||
// The stamp is what CaptureRecoveryUnit writes into each unit manifest, so it must be live across
|
||||
// BOTH the dump leg and the unit capture that follows it — those two together are the pair.
|
||||
func (m *Manager) beginOffsiteRunStamp(runID string) func() {
|
||||
m.mu.Lock()
|
||||
m.offsiteRunID = runID
|
||||
m.offsiteRunDumpAt = time.Now().UTC().Format(time.RFC3339)
|
||||
m.mu.Unlock()
|
||||
return func() {
|
||||
m.mu.Lock()
|
||||
m.offsiteRunID, m.offsiteRunDumpAt = "", ""
|
||||
m.mu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// runDBDumpsInternal is the implementation of RunDBDumps. Caller must hold the running flag.
|
||||
func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
start := time.Now()
|
||||
m.logger.Printf("[INFO] [backup] Starting database dump run")
|
||||
|
||||
// R-181: open the per-run admission scope HERE, because this function is the single orchestrator
|
||||
// of all three write legs. Each app's reserve verdict is taken at its first write of this run and
|
||||
// then reused by the other two legs, so a refused app writes nothing at all and is alerted once.
|
||||
// The scope is closed on every exit path — a set that outlived its run would answer tonight's
|
||||
// question with last night's disk.
|
||||
defer m.beginAdmissionRun()()
|
||||
|
||||
// R-182: the digest scope has the same lifetime. `emitRunSummary` runs BEFORE the closer (defers
|
||||
// unwind last-in-first-out), so the summary is still populated when it is sent, and it sends
|
||||
// nothing at all when the run was clean.
|
||||
kind := m.runKindFor()
|
||||
m.manualRun.Store(false) // tags exactly ONE run; a stale flag would mislabel every later nightly
|
||||
defer m.beginRunSummary(kind, newRunID())()
|
||||
defer m.emitRunSummary()
|
||||
|
||||
dbs, err := DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Database discovery failed: %v", err)
|
||||
@@ -274,6 +493,16 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
continue
|
||||
}
|
||||
|
||||
// R-181: the reserve, BEFORE the first byte of this app's backup is written. This is usually
|
||||
// where an app's verdict is taken, because the DB leg runs first; the volume leg and the
|
||||
// capture then read the same memo. SKIP, not FAIL — a deliberate hold is not a broken dump,
|
||||
// and the operator alert (fired once, inside admitApp) is the signal that it happened.
|
||||
m.noteAttempted(db.StackName)
|
||||
if !m.admitApp(db.StackName) {
|
||||
summary = append(summary, fmt.Sprintf("SKIP %s (reserve — app backup refused)", db.ContainerName))
|
||||
continue
|
||||
}
|
||||
|
||||
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), db.StackName)
|
||||
|
||||
result := DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
|
||||
@@ -282,6 +511,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
if result.Error != nil {
|
||||
allOK = false
|
||||
summary = append(summary, fmt.Sprintf("FAIL %s: %v", result.DB.ContainerName, result.Error))
|
||||
m.noteFailure(db.StackName, "database dump", result.Error.Error())
|
||||
m.logger.Printf("[ERROR] [backup] DB dump failed for %s: %v", result.DB.ContainerName, result.Error)
|
||||
} else {
|
||||
totalSize += result.Size
|
||||
@@ -336,6 +566,11 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest).
|
||||
m.captureAllRecoveryUnits()
|
||||
|
||||
// F5 (CAMPAIGN-3): after the units are fresh on the CURRENT drives, prune any orphaned
|
||||
// backups/primary/<app> dir an app left on an OLD drive when its HDD_PATH moved — pure disk
|
||||
// residue, invisible in the snapshot list. Guarded (deployed + different current drive only).
|
||||
m.pruneStalePrimaryDirs()
|
||||
|
||||
// No silent partials: a DB-dump or volume-dump failure fails the whole run.
|
||||
if !allOK {
|
||||
return fmt.Errorf("some backup steps failed: %s", strings.Join(failedSummaryLines(summary), "; "))
|
||||
@@ -363,6 +598,12 @@ func failedSummaryLines(summary []string) []string {
|
||||
// variant stops the stack before its own volume check — calling it unconditionally would bounce
|
||||
// every volume-less app on every nightly run. Per-stack isolation mirrors the DB loop: one app's
|
||||
// failure is recorded and does not abort the others.
|
||||
//
|
||||
// R-181 adds the reserve to that order, and for the SAME reason: it sits ahead of DumpAppVolumesSafe,
|
||||
// so a refused app is never stopped. A refusal decided inside the Safe variant would already have
|
||||
// bounced the app it was refusing to back up. It sits AFTER the volume-less check because an app with
|
||||
// no named volumes writes nothing in this leg — there is no first write here to gate, and consulting
|
||||
// the reserve for it would only decide a verdict early on a stale reading.
|
||||
func (m *Manager) runVolumeDumps() (summary []string, dumped int, allOK bool) {
|
||||
allOK = true
|
||||
if m.stackProvider == nil {
|
||||
@@ -398,9 +639,19 @@ func (m *Manager) runVolumeDumps() (summary []string, dumped int, allOK bool) {
|
||||
continue
|
||||
}
|
||||
|
||||
// R-181: the reserve, ahead of DumpAppVolumesSafe so a refused app is NOT stopped. For an app
|
||||
// that already has a DB this is a memo lookup taken before its DB dump; for a volume-only app
|
||||
// this is where its verdict is taken, still before its first byte.
|
||||
m.noteAttempted(stack.Name)
|
||||
if !m.admitApp(stack.Name) {
|
||||
summary = append(summary, fmt.Sprintf("SKIP %s volumes (reserve — app backup refused)", stack.Name))
|
||||
continue
|
||||
}
|
||||
|
||||
if err := dump(stack.Name); err != nil {
|
||||
allOK = false
|
||||
summary = append(summary, fmt.Sprintf("FAIL %s volumes: %v", stack.Name, err))
|
||||
m.noteFailure(stack.Name, "volume dump", err.Error())
|
||||
m.logger.Printf("[ERROR] [backup] Volume dump failed for %s: %v", stack.Name, err)
|
||||
continue
|
||||
}
|
||||
@@ -436,22 +687,35 @@ func (m *Manager) DumpAppVolumes(stackName string) error {
|
||||
var dumpErrors []string
|
||||
for _, volName := range volumes {
|
||||
tarPath := filepath.Join(dumpDir, volName+".tar")
|
||||
// F7 (CAMPAIGN-3, HIGH): write the tar to a `.tar.tmp` sibling and only atomically rename it
|
||||
// over the restore point on success — the same crash-safe pattern the DB-dump path uses
|
||||
// (appbackup/dbdump.go DumpOne). Before this, tar wrote the `.tar` IN PLACE, so a mid-write NFS
|
||||
// cut left a 0-byte tar REPLACING the last good dump (tier-1 restore is replace-semantics → an
|
||||
// empty volume). Now a failed/interrupted write only ever touches the `.tmp`; the last good
|
||||
// `.tar` is untouched. The `.tmp` name (ends `.tmp`, not `.tar`) is invisible to the
|
||||
// restore-point/stale scans, so it is never mistaken for a restore point.
|
||||
tmpPath := tarPath + ".tmp"
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] [backup] Dumping volume %s for %s", volName, stackName)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
||||
cmd := exec.CommandContext(ctx, "docker", "run", "--rm",
|
||||
"-v", volName+":/vol:ro",
|
||||
"-v", dumpDir+":/out",
|
||||
"alpine", "tar", "cf", "/out/"+volName+".tar", "-C", "/vol", ".")
|
||||
out, err := cmd.CombinedOutput()
|
||||
cancel()
|
||||
out, err := m.tarVolumeOrDefault(volName, dumpDir)
|
||||
|
||||
if err != nil {
|
||||
m.logger.Printf("[WARN] [backup] Volume dump failed for %s/%s: %s — %v",
|
||||
// Any tar error or context timeout (incl. a dead NFS target → EIO): remove ONLY the tmp,
|
||||
// leave the existing `.tar` restore point byte-untouched, WARN, continue.
|
||||
m.logger.Printf("[WARN] [backup] Volume dump failed for %s/%s (last good dump preserved): %s — %v",
|
||||
stackName, volName, strings.TrimSpace(string(out)), err)
|
||||
os.Remove(tarPath)
|
||||
os.Remove(tmpPath)
|
||||
dumpErrors = append(dumpErrors, volName)
|
||||
continue
|
||||
}
|
||||
|
||||
// fsync the tmp file (flush the tar to disk) then atomically rename over the restore point.
|
||||
if err := atomicPromoteTar(tmpPath, tarPath); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] Volume dump promote failed for %s/%s (last good dump preserved): %v",
|
||||
stackName, volName, err)
|
||||
os.Remove(tmpPath)
|
||||
dumpErrors = append(dumpErrors, volName)
|
||||
continue
|
||||
}
|
||||
@@ -461,17 +725,24 @@ func (m *Manager) DumpAppVolumes(stackName string) error {
|
||||
}
|
||||
}
|
||||
|
||||
// Clean up tars for volumes that no longer exist
|
||||
// Clean up tars (and any orphan `.tar.tmp` from a killed run) for volumes that no longer exist.
|
||||
entries, _ := os.ReadDir(dumpDir)
|
||||
activeVols := make(map[string]bool)
|
||||
for _, v := range volumes {
|
||||
activeVols[v+".tar"] = true
|
||||
}
|
||||
for _, e := range entries {
|
||||
if !activeVols[e.Name()] && strings.HasSuffix(e.Name(), ".tar") {
|
||||
os.Remove(filepath.Join(dumpDir, e.Name()))
|
||||
name := e.Name()
|
||||
// A leftover `.tar.tmp` is never a restore point — always safe to remove (its `.tar` sibling,
|
||||
// if any, is the real restore point and is handled by the `.tar` branch).
|
||||
if strings.HasSuffix(name, ".tar.tmp") {
|
||||
os.Remove(filepath.Join(dumpDir, name))
|
||||
continue
|
||||
}
|
||||
if !activeVols[name] && strings.HasSuffix(name, ".tar") {
|
||||
os.Remove(filepath.Join(dumpDir, name))
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] [backup] Removed stale volume dump: %s/%s", stackName, e.Name())
|
||||
m.logger.Printf("[DEBUG] [backup] Removed stale volume dump: %s/%s", stackName, name)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -482,16 +753,78 @@ func (m *Manager) DumpAppVolumes(stackName string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// tarVolumeOrDefault runs the F7 tar seam (m.tarVolume) or, when unset, the real docker tar into the
|
||||
// `<volName>.tar.tmp` sibling under dumpDir. The 10-minute bound matches the original.
|
||||
func (m *Manager) tarVolumeOrDefault(volName, dumpDir string) ([]byte, error) {
|
||||
if m.tarVolume != nil {
|
||||
return m.tarVolume(volName, dumpDir)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
||||
defer cancel()
|
||||
cmd := exec.CommandContext(ctx, "docker", "run", "--rm",
|
||||
"-v", volName+":/vol:ro",
|
||||
"-v", dumpDir+":/out",
|
||||
"alpine", "tar", "cf", "/out/"+volName+".tar.tmp", "-C", "/vol", ".")
|
||||
return cmd.CombinedOutput()
|
||||
}
|
||||
|
||||
// atomicPromoteTar fsyncs a completed `.tar.tmp` then atomically renames it over the final `.tar`
|
||||
// (same-dir rename = atomic on the target fs). It mirrors DumpOne's crash-safety (F7) and EXCEEDS it
|
||||
// by also best-effort fsync'ing the directory entry, so the rename itself survives a power loss — the
|
||||
// DB-dump path fsyncs the file but not the dir; a follow-up could add the dir fsync there too. On any
|
||||
// error the tmp is left for the caller to remove; the final `.tar` is never touched here except by a
|
||||
// successful rename.
|
||||
func atomicPromoteTar(tmpPath, finalPath string) error {
|
||||
// O_RDWR, not os.Open: fsync on a read-only handle is refused on Windows (dev-box test runs),
|
||||
// while a writable handle syncs on every platform. Content is not modified.
|
||||
f, err := os.OpenFile(tmpPath, os.O_RDWR, 0)
|
||||
if err != nil {
|
||||
return fmt.Errorf("opening tmp dump: %w", err)
|
||||
}
|
||||
if err := f.Sync(); err != nil {
|
||||
f.Close()
|
||||
return fmt.Errorf("syncing tmp dump: %w", err)
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
return fmt.Errorf("closing tmp dump: %w", err)
|
||||
}
|
||||
if err := os.Rename(tmpPath, finalPath); err != nil {
|
||||
return fmt.Errorf("renaming tmp dump: %w", err)
|
||||
}
|
||||
// Best-effort: fsync the directory so the rename is durable (ignore errors — the rename already
|
||||
// made the new content visible; this only hardens against a power loss immediately after).
|
||||
if dir, derr := os.Open(filepath.Dir(finalPath)); derr == nil {
|
||||
_ = dir.Sync()
|
||||
_ = dir.Close()
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DumpAppVolumesSafe stops the stack before dumping volumes and restarts after.
|
||||
// Prevents inconsistent tars of live database volumes (e.g. PostgreSQL).
|
||||
// Protected stacks that reject StopStack will return an error — callers handle as warning.
|
||||
//
|
||||
// R-166: the stop→dump→start window is marked. Before this, a controller killed between the stop
|
||||
// and the start left the app down with NOTHING on disk saying why or that it was owed a restart —
|
||||
// and a stopped app has zero containers, which the boot reconciler then read as a deliberate
|
||||
// customer stop and left alone. The marker is the mechanism, not the restart call below: a SIGKILL
|
||||
// runs no deferred function (Campaign 8 fault 10, on live hardware), so only something already
|
||||
// written to disk can survive it.
|
||||
func (m *Manager) DumpAppVolumesSafe(stackName string) error {
|
||||
if m.stackProvider == nil {
|
||||
return fmt.Errorf("no stack provider")
|
||||
}
|
||||
|
||||
// Intent before the act: refuse to stop an app we cannot promise to restart.
|
||||
if err := m.appStop.Begin("volume-dump:"+stackName, ReasonVolumeDump, []string{stackName}); err != nil {
|
||||
return fmt.Errorf("could not record the app-stop marker for %s (refusing to stop it unprotected): %w", stackName, err)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [backup] Stopping %s for safe volume dump", stackName)
|
||||
if err := m.stackProvider.StopStack(stackName); err != nil {
|
||||
// Nothing was stopped, so nothing is owed a restart — clear rather than strand a marker that
|
||||
// would cost a spurious (if harmless) restart at the next startup.
|
||||
m.appStop.End()
|
||||
return fmt.Errorf("could not stop %s for volume dump: %w", stackName, err)
|
||||
}
|
||||
|
||||
@@ -501,6 +834,10 @@ func (m *Manager) DumpAppVolumesSafe(stackName string) error {
|
||||
startErr := m.stackProvider.StartStack(stackName)
|
||||
if startErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Failed to restart %s after volume dump: %v", stackName, startErr)
|
||||
} else {
|
||||
// Cleared ONLY on a restart that succeeded. A failed restart keeps the marker so the next
|
||||
// startup retries — the app really is still owed one.
|
||||
m.appStop.End()
|
||||
}
|
||||
|
||||
// Surface both errors — callers must know if the app is left stopped
|
||||
@@ -527,6 +864,12 @@ func (m *Manager) IsRunning() bool {
|
||||
return m.running
|
||||
}
|
||||
|
||||
// AcquireRunningForTest / ReleaseRunningForTest occupy the single-flight from another package's
|
||||
// test, so the "a run is already in flight" branch can be exercised without racing a real run.
|
||||
// Test-only seam, in the same spirit as SetOffboxRunner; nothing in production calls them.
|
||||
func (m *Manager) AcquireRunningForTest() error { return m.acquireRunning() }
|
||||
func (m *Manager) ReleaseRunningForTest() { m.releaseRunning() }
|
||||
|
||||
// acquireRunning atomically sets the running flag. Returns error if already running.
|
||||
func (m *Manager) acquireRunning() error {
|
||||
m.mu.Lock()
|
||||
@@ -700,7 +1043,19 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
|
||||
// Phase 2: keep each app's recovery unit current with its definition. Idempotent
|
||||
// (checksum-skip), so this periodic refresh only writes when the config actually changed,
|
||||
// and ensures units exist shortly after startup without waiting for the daily DB dump.
|
||||
m.captureAllRecoveryUnits()
|
||||
//
|
||||
// R-182: this sweep gets its OWN digest scope. It has to, and the reason is the whole
|
||||
// balance of this change. The per-app event is now record-only, so without a digest here a
|
||||
// capture failure detected between runs would be recorded and NEVER notified — a new
|
||||
// silence introduced while closing one. But this path can fire on every status poll, so its
|
||||
// digest deliberately carries NO run id: the hub's ordinary 1-hour operator cooldown then
|
||||
// applies, which caps it at one mail an hour exactly as before, while the mail now lists
|
||||
// EVERY failing app instead of whichever one happened to be first.
|
||||
func() {
|
||||
defer m.beginRunSummary(runKindRefresh, "")()
|
||||
defer m.emitRunSummary()
|
||||
m.captureAllRecoveryUnits()
|
||||
}()
|
||||
}
|
||||
|
||||
// Fill in dynamic fields under lock.
|
||||
@@ -749,6 +1104,7 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
// Update dynamic fields that don't need subprocess calls
|
||||
status.Running = m.running
|
||||
status.NextDBDump = nextDBDump
|
||||
status.SingleCopyWarning = m.singleCopyWarning() // F6: honest single-drive signal
|
||||
// Deep-copy lastDBDump so callers cannot mutate shared state.
|
||||
if m.lastDBDump != nil {
|
||||
copyDump := *m.lastDBDump
|
||||
@@ -786,10 +1142,11 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
|
||||
// No cache yet — return a minimal status (first page load before cache is populated)
|
||||
status := &FullBackupStatus{
|
||||
Enabled: m.cfg.Backup.Enabled,
|
||||
Running: m.running,
|
||||
DBDumpSchedule: m.cfg.Backup.DBDumpSchedule,
|
||||
NextDBDump: nextDBDump,
|
||||
Enabled: m.cfg.Backup.Enabled,
|
||||
Running: m.running,
|
||||
DBDumpSchedule: m.cfg.Backup.DBDumpSchedule,
|
||||
NextDBDump: nextDBDump,
|
||||
SingleCopyWarning: m.singleCopyWarning(), // F6
|
||||
}
|
||||
if m.lastDBDump != nil {
|
||||
copyDump := *m.lastDBDump
|
||||
@@ -802,6 +1159,123 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
return status
|
||||
}
|
||||
|
||||
// sameDevice reports whether two paths sit on the same physical device, through the test seam when
|
||||
// one is installed. Nil seam → system.SamePhysicalDevice, i.e. byte-for-byte the previous behaviour.
|
||||
func (m *Manager) sameDevice(a, b string) bool {
|
||||
if m.samePhysicalDevice != nil {
|
||||
return m.samePhysicalDevice(a, b)
|
||||
}
|
||||
return system.SamePhysicalDevice(a, b)
|
||||
}
|
||||
|
||||
// hasOffDriveTarget reports whether any registered, schedulable storage path lives on a physical disk
|
||||
// OTHER than the system drive — i.e. whether a genuine off-drive (tier-2) copy is possible at all.
|
||||
// When false the box is single-drive: tier-1 is the ONLY local copy and 3-2-1 needs a 2nd drive or
|
||||
// offsite (F6 — surfaced honestly via SingleCopyWarning, never a faked guarantee).
|
||||
func (m *Manager) hasOffDriveTarget() bool {
|
||||
if m.settings == nil || m.systemDataPath == "" {
|
||||
return false
|
||||
}
|
||||
for _, sp := range m.settings.GetSchedulableStoragePaths() {
|
||||
if sp.Path == m.systemDataPath || m.sameDevice(m.systemDataPath, sp.Path) {
|
||||
continue
|
||||
}
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// singleCopyWarning returns the honest single-drive notice, or "" when an off-drive target exists.
|
||||
func (m *Manager) singleCopyWarning() string {
|
||||
if m.hasOffDriveTarget() {
|
||||
return ""
|
||||
}
|
||||
return singleCopyNotice
|
||||
}
|
||||
|
||||
// sysDriveLabelFor returns the drive label for a stack's tier-1 restore point — the clear
|
||||
// system-drive label for a sys_drive (volume-only) app (F6: never blank), else the enrolled drive's
|
||||
// storage label.
|
||||
func (m *Manager) sysDriveLabelFor(stackName string) string {
|
||||
drive := m.GetAppDrivePath(stackName)
|
||||
if drive == "" {
|
||||
return ""
|
||||
}
|
||||
if drive == m.systemDataPath {
|
||||
return systemDriveLabel
|
||||
}
|
||||
if m.settings != nil {
|
||||
return m.settings.GetStorageLabel(drive)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// pruneStalePrimaryDirs removes orphaned `backups/primary/<app>` dirs left on a drive after an app's
|
||||
// HDD_PATH moved to another drive (F5, CAMPAIGN-3 — pure disk residue, invisible in the snapshot
|
||||
// list). LOAD-BEARING GUARDS: a dir is removed ONLY when <app> is currently deployed AND its current
|
||||
// namespace root differs from this dir's drive. It NEVER removes the dir on the app's CURRENT drive
|
||||
// (that IS the live restore point), and NEVER removes a dir for an app NOT in the deployed set (an
|
||||
// undeployed app's last backup is still its restore point — orphaned-app cleanup is a separate,
|
||||
// user-driven concern). Only operates strictly under a `backups/primary/` prefix.
|
||||
func (m *Manager) pruneStalePrimaryDirs() {
|
||||
if m.stackProvider == nil {
|
||||
return
|
||||
}
|
||||
// Current namespace root per DEPLOYED app.
|
||||
current := map[string]string{}
|
||||
for _, s := range m.stackProvider.ListDeployedStacks() {
|
||||
if drive := m.GetAppDrivePath(s.Name); drive != "" {
|
||||
current[s.Name] = filepath.Clean(m.namespaceRoot(drive))
|
||||
}
|
||||
}
|
||||
// Candidate drives to scan: the system drive + every registered storage path.
|
||||
var nsRoots []string
|
||||
if m.systemDataPath != "" {
|
||||
nsRoots = append(nsRoots, filepath.Clean(m.namespaceRoot(m.systemDataPath)))
|
||||
}
|
||||
if m.settings != nil {
|
||||
for _, sp := range m.settings.GetStoragePaths() {
|
||||
nsRoots = append(nsRoots, filepath.Clean(NamespaceRoot(sp.Path, true)))
|
||||
}
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
for _, nsRoot := range nsRoots {
|
||||
if seen[nsRoot] {
|
||||
continue
|
||||
}
|
||||
seen[nsRoot] = true
|
||||
primaryDir := PrimaryBackupPath(nsRoot)
|
||||
entries, err := os.ReadDir(primaryDir)
|
||||
if err != nil {
|
||||
continue // absent/unreadable (e.g. a disconnected drive) — nothing to prune here
|
||||
}
|
||||
for _, e := range entries {
|
||||
if !e.IsDir() {
|
||||
continue
|
||||
}
|
||||
app := e.Name()
|
||||
cur, deployed := current[app]
|
||||
if !deployed {
|
||||
continue // GUARD: an undeployed app's last backup is still its restore point
|
||||
}
|
||||
if cur == nsRoot {
|
||||
continue // GUARD: this IS the app's current drive — the live restore point
|
||||
}
|
||||
stalePath := RecoveryUnitPath(nsRoot, app)
|
||||
// Prefix safety: only ever remove strictly inside `backups/primary/` (no surprise user data).
|
||||
if !strings.HasPrefix(filepath.Clean(stalePath)+string(filepath.Separator),
|
||||
filepath.Clean(primaryDir)+string(filepath.Separator)) {
|
||||
continue
|
||||
}
|
||||
if err := os.RemoveAll(stalePath); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] F5: could not remove stale primary dir for %s on old drive: %v", app, err)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [backup] F5: removed stale primary backup dir for %s on an old drive (app now on %s)", app, cur)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// isDebug returns true if logging level is "debug".
|
||||
func (m *Manager) isDebug() bool {
|
||||
return m.cfg != nil && m.cfg.Logging.Level == "debug"
|
||||
|
||||
@@ -0,0 +1,305 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/fillwatch"
|
||||
)
|
||||
|
||||
// R-165 / decision B2 — the capture floor that replaces the `mp1` bulkhead.
|
||||
//
|
||||
// Before the merge, the 20 G backup partition kept a runaway capture from reaching
|
||||
// `/var/lib/docker`, because it was a different filesystem. After the merge it is the same one, and a
|
||||
// full Docker data-root is a stopped box. These pin the replacement.
|
||||
|
||||
// floorProvider lists stacks and always resolves recovery info — the floor must refuse BEFORE any of
|
||||
// that is consulted, so a capture that gets as far as GetStackRecoveryInfo has already lost.
|
||||
type floorProvider struct {
|
||||
stacks []string
|
||||
dir string
|
||||
infoHits []string // records every app whose recovery info was read = a capture that was ATTEMPTED
|
||||
}
|
||||
|
||||
func (p *floorProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *floorProvider) ListDeployedStacks() []StackSummary {
|
||||
out := make([]StackSummary, 0, len(p.stacks))
|
||||
for _, s := range p.stacks {
|
||||
out = append(out, StackSummary{Name: s})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *floorProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *floorProvider) GetStackHDDPath(string) string { return "" }
|
||||
func (p *floorProvider) GetImportRoot() string { return "" }
|
||||
func (p *floorProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *floorProvider) StopStack(string) error { return nil }
|
||||
func (p *floorProvider) StartStack(string) error { return nil }
|
||||
func (p *floorProvider) RefreshAndIsRunning(string) bool { return true }
|
||||
func (p *floorProvider) GetStackRecoveryInfo(name string) (RecoveryInfo, bool) {
|
||||
p.infoHits = append(p.infoHits, name)
|
||||
return RecoveryInfo{StackDir: filepath.Join(p.dir, "stacks", name)}, true
|
||||
}
|
||||
func (p *floorProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *floorProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *floorProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *floorProvider) GetStackClassifiedBinds(string) ([]ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
type floorHarness struct {
|
||||
m *Manager
|
||||
prov *floorProvider
|
||||
events []unitEvent
|
||||
usage map[string]*UnitSpace
|
||||
dir string
|
||||
}
|
||||
|
||||
// newFloorHarness injects the usage read, so the filesystem's occupancy is a test input rather than
|
||||
// something the test has to manufacture on a real disk.
|
||||
func newFloorHarness(t *testing.T, stacks ...string) *floorHarness {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
h := &floorHarness{
|
||||
prov: &floorProvider{stacks: stacks, dir: dir},
|
||||
usage: map[string]*UnitSpace{},
|
||||
dir: dir,
|
||||
}
|
||||
h.m = &Manager{
|
||||
logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: dir,
|
||||
stackProvider: h.prov,
|
||||
unitSpaceFn: func(name string) *UnitSpace { return h.usage[name] },
|
||||
}
|
||||
h.m.SetUnitNotify(func(name string, err error, u *UnitSpace) {
|
||||
h.events = append(h.events, unitEvent{app: name, err: err.Error(), usage: u})
|
||||
})
|
||||
return h
|
||||
}
|
||||
|
||||
func (h *floorHarness) setSpace(app string, usedPct, availGB float64) {
|
||||
h.usage[app] = &UnitSpace{
|
||||
Path: h.dir, UsedPercent: usedPct, AvailGB: availGB,
|
||||
TotalGB: 100, UsedGB: usedPct,
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario D — the floor refuses, per app, and says so ----------------------------------------
|
||||
|
||||
func TestFloor_RefusesTheAppAndLeavesItsPreviousUnitByteIdentical(t *testing.T) {
|
||||
h := newFloorHarness(t, "homebox", "immich", "nextcloud")
|
||||
h.setSpace("homebox", 40, 60)
|
||||
h.setSpace("immich", 98, 0.4) // below the floor on BOTH terms
|
||||
h.setSpace("nextcloud", 40, 60)
|
||||
|
||||
// A previous unit exists for the app about to be refused. Checksum it before and after.
|
||||
unitDir := filepath.Join(h.dir, "felhom-data", "backups", "primary", "immich", "compose")
|
||||
if err := os.MkdirAll(unitDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prev := filepath.Join(unitDir, "app.yaml")
|
||||
if err := os.WriteFile(prev, []byte("deployed: true\nenv:\n A: previous-good-value\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before := checksumFile(t, prev)
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
// The refused app must NOT have been attempted at all — the floor is checked BEFORE any write.
|
||||
for _, hit := range h.prov.infoHits {
|
||||
if hit == "immich" {
|
||||
t.Fatal("the refused app's recovery info was read — the capture was ATTEMPTED rather than " +
|
||||
"refused up front, so a write could have started and failed partway")
|
||||
}
|
||||
}
|
||||
|
||||
if after := checksumFile(t, prev); after != before {
|
||||
t.Fatalf("the previous unit changed (%s → %s) — a refused capture must leave the last good "+
|
||||
"copy byte-identical", before, after)
|
||||
}
|
||||
if _, err := os.Stat(prev); err != nil {
|
||||
t.Fatalf("the previous unit is gone: %v — the floor REFUSES, it never deletes", err)
|
||||
}
|
||||
|
||||
// Exactly one alert, for the refused app, carrying the space figures.
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1: %+v", len(h.events), h.events)
|
||||
}
|
||||
e := h.events[0]
|
||||
if e.app != "immich" {
|
||||
t.Fatalf("alert names %q, want immich", e.app)
|
||||
}
|
||||
if e.usage == nil || e.usage.AvailGB != 0.4 {
|
||||
t.Fatalf("the alert carries no/incorrect space figures: %+v", e.usage)
|
||||
}
|
||||
if !strings.Contains(e.err, "reserve") {
|
||||
t.Fatalf("the alert message %q does not say it was a reserve refusal — an operator would read "+
|
||||
"it as a broken capture rather than a deliberate hold", e.err)
|
||||
}
|
||||
|
||||
// The other two must have been captured normally — one app's refusal must not silence its siblings.
|
||||
got := strings.Join(h.prov.infoHits, ",")
|
||||
if !strings.Contains(got, "homebox") || !strings.Contains(got, "nextcloud") {
|
||||
t.Fatalf("attempted=%v — the loop did not continue past the refusal", h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing may be deleted to make room, under any threshold. Nothing on this filesystem is
|
||||
// generational, so "the oldest" is always a DIFFERENT app's only local copy.
|
||||
func TestFloor_NeverDeletesAnotherAppsUnit(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich", "nextcloud")
|
||||
h.setSpace("immich", 99, 0.1)
|
||||
h.setSpace("nextcloud", 99, 0.1)
|
||||
|
||||
other := filepath.Join(h.dir, "felhom-data", "backups", "primary", "nextcloud")
|
||||
if err := os.MkdirAll(other, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
keep := filepath.Join(other, "manifest.json")
|
||||
if err := os.WriteFile(keep, []byte(`{"app_name":"nextcloud"}`), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before := checksumFile(t, keep)
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
if _, err := os.Stat(keep); err != nil {
|
||||
t.Fatalf("another app's unit was DELETED to make room: %v — nothing here is generational, so "+
|
||||
"pruning could only destroy an app's only local copy", err)
|
||||
}
|
||||
if after := checksumFile(t, keep); after != before {
|
||||
t.Fatal("another app's unit was modified while the filesystem was under the floor")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario E — the floor is not a wall by another name ----------------------------------------
|
||||
|
||||
// The floor is about the FILESYSTEM's remaining headroom, never the unit's size. A per-unit cap would
|
||||
// be R-163 rebuilt inside one volume.
|
||||
func TestFloor_LargeUnitWithAmpleSpaceIsCaptured(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// A huge app on a huge, mostly-empty filesystem: 40% used, 600 GB free.
|
||||
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 600, TotalGB: 1000, UsedGB: 400}
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("a capture was refused on a filesystem with 600 GB free (%+v) — the floor has become "+
|
||||
"a per-unit size cap, which is exactly the ceiling R-165 removed", h.events)
|
||||
}
|
||||
if len(h.prov.infoHits) != 1 || h.prov.infoHits[0] != "immich" {
|
||||
t.Fatalf("attempted=%v, want [immich] — the capture was not even tried", h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// The old 20 G ceiling must not survive anywhere: a unit far larger than the retired partition is
|
||||
// captured when the filesystem has room.
|
||||
func TestFloor_TheOld20GCeilingIsGone(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// 180 GB free, and the app's own data is 120 GB — SIX TIMES the retired 20 G area. The figure is
|
||||
// deliberately far above 20 so that a literal `UsedGB > 20` cap cannot survive this test: a
|
||||
// fixture sitting exactly on the old boundary would pass under the very shape it forbids.
|
||||
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 180, TotalGB: 300, UsedGB: 120}
|
||||
h.m.captureAllRecoveryUnits()
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("refused with 180 GB free: %+v — a fixed per-area limit survives somewhere", h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group E — the floor sits BELOW the critical warning band -------------------------------------
|
||||
|
||||
// A floor that fires before its own warning is a silent failure wearing a threshold: the customer
|
||||
// would get a refusal with no prior notice that anything was wrong. The customer's `disk_critical`
|
||||
// must always come first.
|
||||
func TestFloorSitsBelowTheCriticalWarningBand(t *testing.T) {
|
||||
if FloorUsedPercent <= fillwatch.CritUsedPercent {
|
||||
t.Fatalf("FloorUsedPercent (%.1f) must be strictly ABOVE fillwatch.CritUsedPercent (%.1f) — "+
|
||||
"otherwise a capture can be refused before the customer was ever warned that the disk was "+
|
||||
"filling, which is a silent failure wearing a threshold",
|
||||
FloorUsedPercent, fillwatch.CritUsedPercent)
|
||||
}
|
||||
if FloorFreeGiB >= fillwatch.CritFreeGiB {
|
||||
t.Fatalf("FloorFreeGiB (%.1f) must be strictly BELOW fillwatch.CritFreeGiB (%.1f) — the "+
|
||||
"free-byte term needs the same ordering as the percentage term, or the free-byte path "+
|
||||
"refuses before it warns", FloorFreeGiB, fillwatch.CritFreeGiB)
|
||||
}
|
||||
// And below the WARNING band too, transitively — stated explicitly so the chain is readable.
|
||||
if FloorUsedPercent <= fillwatch.WarnUsedPercent || FloorFreeGiB >= fillwatch.WarnFreeGiB {
|
||||
t.Fatal("the floor is not beyond the warning band — the customer must be warned, then warned " +
|
||||
"critically, and only then can a capture be refused")
|
||||
}
|
||||
|
||||
// Both terms must be able to refuse INDEPENDENTLY — that is why there are two. estGiB=0 is the
|
||||
// history-less case, which exercises the headroom term alone.
|
||||
if _, r := (&Manager{}).floorVerdict(&UnitSpace{UsedPercent: 50, AvailGB: 0.5}, 0); r != floorHeadroom {
|
||||
t.Fatal("a filesystem with 0.5 GiB free at only 50% used was NOT refused — the free-byte term " +
|
||||
"does not trip on its own, so a very large volume can run out without the floor engaging")
|
||||
}
|
||||
if _, r := (&Manager{}).floorVerdict(&UnitSpace{UsedPercent: 98, AvailGB: 40}, 0); r != floorHeadroom {
|
||||
t.Fatal("a filesystem 98% used was NOT refused — the percentage term does not trip on its own")
|
||||
}
|
||||
if _, r := (&Manager{}).floorVerdict(&UnitSpace{UsedPercent: 50, AvailGB: 50}, 0); r != floorAdmit {
|
||||
t.Fatal("a healthy filesystem was refused")
|
||||
}
|
||||
}
|
||||
|
||||
// --- §8.4 — a nil usage read neither refuses nor warns --------------------------------------------
|
||||
|
||||
func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// No entry → the injected reader returns nil, which is what system.GetDiskUsage does on error.
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("an UNREADABLE filesystem produced %d alert(s): %+v — an absent, unmounted or "+
|
||||
"unreadable filesystem is the drive gate's business and already has its own alert; "+
|
||||
"refusing here would block every capture on a box whose drive merely blipped", len(h.events), h.events)
|
||||
}
|
||||
if len(h.prov.infoHits) != 1 {
|
||||
t.Fatalf("the capture was not attempted on an unreadable read (attempted=%v) — a nil reading "+
|
||||
"must not refuse", h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// ErrCaptureFloor must be matchable, so a caller can tell a deliberate refusal from a broken capture.
|
||||
func TestErrCaptureFloor_IsMatchable(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
h.setSpace("immich", 99, 0.2)
|
||||
h.m.captureAllRecoveryUnits()
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("want 1 event, got %d", len(h.events))
|
||||
}
|
||||
// The seam hands a string, so assert on the sentinel's own text being present and distinct.
|
||||
if !errors.Is(errWrapForTest(), ErrCaptureFloor) {
|
||||
t.Fatal("ErrCaptureFloor does not survive wrapping")
|
||||
}
|
||||
if !strings.Contains(h.events[0].err, "Refused") && !strings.Contains(h.events[0].err, "refused") {
|
||||
t.Fatalf("the alert %q does not identify itself as a refusal", h.events[0].err)
|
||||
}
|
||||
}
|
||||
|
||||
func errWrapForTest() error { return errors.Join(ErrCaptureFloor, errors.New("ctx")) }
|
||||
|
||||
func checksumFile(t *testing.T, path string) string {
|
||||
t.Helper()
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
h := sha256.New()
|
||||
if _, err := io.Copy(h, f); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return hex.EncodeToString(h.Sum(nil))
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// oneDrivePerSubtree is the test stand-in for system.SamePhysicalDevice (st_dev equality).
|
||||
//
|
||||
// Why it exists: the real predicate asks "are these two paths on the same physical disk?", and
|
||||
// Tier-2's whole purpose is to refuse a target that is. On a host where every t.TempDir() lands on
|
||||
// one filesystem — DooPlex, and any CI box with a single volume — a fixture's "usb" and "flash"
|
||||
// dirs share one st_dev, so the guard correctly refuses them and the off-drive tests can never
|
||||
// exercise their subject. This models what the fixture is actually depicting: one drive per
|
||||
// directory subtree, so two paths share a device only when one contains the other (a path inside a
|
||||
// drive IS on that drive). Unrelated subtrees are distinct devices, exactly as real mountpoints are.
|
||||
//
|
||||
// It does NOT relax any assertion — the guard still runs, still refuses same-device targets (see
|
||||
// TestSharesTier2NeverTargetsItsOwnSourceDrive, which passes under this seam), and production keeps
|
||||
// using the real st_dev check because the seam is nil there.
|
||||
func oneDrivePerSubtree(a, b string) bool {
|
||||
a, b = filepath.Clean(a), filepath.Clean(b)
|
||||
if a == b {
|
||||
return true
|
||||
}
|
||||
sep := string(filepath.Separator)
|
||||
return strings.HasPrefix(a, b+sep) || strings.HasPrefix(b, a+sep)
|
||||
}
|
||||
@@ -0,0 +1,107 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// f5Manager wires a Manager with a real settings store holding the given enrolled drives, plus a
|
||||
// sys drive, and a fake provider mapping each deployed app to its CURRENT HDD path.
|
||||
func f5Manager(t *testing.T, sysDrive string, enrolled []string, deployed map[string]string) *Manager {
|
||||
t.Helper()
|
||||
sett, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, d := range enrolled {
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: d, Label: filepath.Base(d), Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
var stacks []StackSummary
|
||||
for name := range deployed {
|
||||
stacks = append(stacks, StackSummary{Name: name})
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.SystemDataPath = sysDrive
|
||||
fake := &volDumpFakeProvider{stacks: stacks, hdd: deployed}
|
||||
return &Manager{cfg: cfg, settings: sett, logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: sysDrive, stackProvider: fake}
|
||||
}
|
||||
|
||||
// seedPrimaryDir creates a backups/primary/<app> dir (with a marker file) under a drive's namespace.
|
||||
func seedPrimaryDir(t *testing.T, driveNsRoot, app string) string {
|
||||
t.Helper()
|
||||
dir := RecoveryUnitPath(driveNsRoot, app)
|
||||
if err := os.MkdirAll(filepath.Join(dir, "volume-dumps"), 0755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "manifest.json"), []byte("{}"), 0644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return dir
|
||||
}
|
||||
|
||||
// F5: an app redeployed from drive A to drive B → its stale primary dir on A is removed, its live
|
||||
// dir on B is kept. Guard red-proof lives in the two "kept" assertions.
|
||||
func TestPruneStalePrimaryDirs_RemovesRedeployedResidue(t *testing.T) {
|
||||
driveA := t.TempDir()
|
||||
driveB := t.TempDir()
|
||||
sys := filepath.Join(t.TempDir(), "sys")
|
||||
// nextcloud currently on B; a stale unit remains on A.
|
||||
m := f5Manager(t, sys, []string{driveA, driveB}, map[string]string{"nextcloud": driveB})
|
||||
|
||||
nsA := NamespaceRoot(driveA, true)
|
||||
nsB := NamespaceRoot(driveB, true)
|
||||
staleA := seedPrimaryDir(t, nsA, "nextcloud")
|
||||
liveB := seedPrimaryDir(t, nsB, "nextcloud")
|
||||
|
||||
m.pruneStalePrimaryDirs()
|
||||
|
||||
if _, err := os.Stat(staleA); !os.IsNotExist(err) {
|
||||
t.Errorf("stale primary dir on the OLD drive was not removed (F5)")
|
||||
}
|
||||
if _, err := os.Stat(liveB); err != nil {
|
||||
t.Errorf("the LIVE primary dir on the current drive was wrongly removed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// GUARD: an UNDEPLOYED app's primary dir is a restore point — it must be KEPT (companion: drop the
|
||||
// `!deployed { continue }` guard → this dir is deleted → fail).
|
||||
func TestPruneStalePrimaryDirs_KeepsUndeployedRestorePoint(t *testing.T) {
|
||||
driveA := t.TempDir()
|
||||
sys := filepath.Join(t.TempDir(), "sys")
|
||||
// Only "nextcloud" is deployed (on A); "removed-app" has a leftover unit but is NOT deployed.
|
||||
m := f5Manager(t, sys, []string{driveA}, map[string]string{"nextcloud": driveA})
|
||||
nsA := NamespaceRoot(driveA, true)
|
||||
seedPrimaryDir(t, nsA, "nextcloud")
|
||||
orphan := seedPrimaryDir(t, nsA, "removed-app")
|
||||
|
||||
m.pruneStalePrimaryDirs()
|
||||
|
||||
if _, err := os.Stat(orphan); err != nil {
|
||||
t.Errorf("an UNDEPLOYED app's restore point was wrongly deleted (guard failure): %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// GUARD: the dir on an app's CURRENT drive is never touched (already covered above, but assert it
|
||||
// directly with a single-drive app so no cross-drive move is involved).
|
||||
func TestPruneStalePrimaryDirs_KeepsCurrentDriveDir(t *testing.T) {
|
||||
driveA := t.TempDir()
|
||||
sys := filepath.Join(t.TempDir(), "sys")
|
||||
m := f5Manager(t, sys, []string{driveA}, map[string]string{"nextcloud": driveA})
|
||||
nsA := NamespaceRoot(driveA, true)
|
||||
live := seedPrimaryDir(t, nsA, "nextcloud")
|
||||
|
||||
m.pruneStalePrimaryDirs()
|
||||
|
||||
if _, err := os.Stat(live); err != nil {
|
||||
t.Errorf("the current-drive primary dir was wrongly removed: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// F6: a volume-only app (no HDD_PATH → data on sys_drive) must now flow through the tier-2 run — it
|
||||
// used to be skipped, leaving a single controller-level copy. COMPANION red-proof: restore the
|
||||
// `GetStackHDDPath == "" { continue }` skip in RunAllTier2 → "actualbudget" is absent → this fails.
|
||||
func TestRunAllTier2_IncludesVolumeOnlyApps(t *testing.T) {
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.SystemDataPath = filepath.Join(t.TempDir(), "sys")
|
||||
fake := &volDumpFakeProvider{
|
||||
stacks: []StackSummary{{Name: "nextcloud"}, {Name: "actualbudget"}},
|
||||
hdd: map[string]string{"nextcloud": filepath.Join(t.TempDir(), "usb")}, // actualbudget = volume-only
|
||||
}
|
||||
m := &Manager{cfg: cfg, logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: cfg.Paths.SystemDataPath, stackProvider: fake}
|
||||
|
||||
var processed []string
|
||||
m.perAppTier2 = func(name string) error { processed = append(processed, name); return nil }
|
||||
m.RunAllTier2()
|
||||
|
||||
var sawVolumeOnly bool
|
||||
for _, p := range processed {
|
||||
if p == "actualbudget" {
|
||||
sawVolumeOnly = true
|
||||
}
|
||||
}
|
||||
if !sawVolumeOnly {
|
||||
t.Errorf("volume-only app 'actualbudget' was NOT included in the tier-2 run (F6): processed=%v", processed)
|
||||
}
|
||||
}
|
||||
|
||||
// F6: a single-drive box (no off-drive target) surfaces the honest single-copy notice; a box with an
|
||||
// off-drive target does not (never a faked 3-2-1 guarantee).
|
||||
func TestSingleCopyWarning_HonestOnSingleDrive(t *testing.T) {
|
||||
// Single-drive: only the system drive, no schedulable off-drive paths.
|
||||
m1, _ := newTestManager(t, "/srv/sys")
|
||||
if got := m1.singleCopyWarning(); got != singleCopyNotice {
|
||||
t.Errorf("single-drive box must surface the honest notice, got %q", got)
|
||||
}
|
||||
|
||||
// Multi-drive: an enrolled off-drive path → no warning.
|
||||
m2, sett := newTestManager(t, "/srv/sys")
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: "/mnt/usb", Label: "USB", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := m2.singleCopyWarning(); got != "" {
|
||||
t.Errorf("a box with an off-drive target must NOT show the single-copy notice, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// F6: a sys_drive (volume-only) app's restore point carries a clear drive label, never blank.
|
||||
func TestRestorePoint_SysDriveLabelNotBlank(t *testing.T) {
|
||||
cfg := &config.Config{}
|
||||
sysDrive := t.TempDir()
|
||||
cfg.Paths.SystemDataPath = sysDrive
|
||||
fake := &volDumpFakeProvider{
|
||||
stacks: []StackSummary{{Name: "actualbudget"}},
|
||||
volumes: map[string][]string{"actualbudget": {"actualbudget_data"}},
|
||||
}
|
||||
m := &Manager{cfg: cfg, logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: sysDrive, stackProvider: fake}
|
||||
|
||||
label := m.sysDriveLabelFor("actualbudget")
|
||||
if label != systemDriveLabel {
|
||||
t.Errorf("sys_drive app drive label = %q, want %q (never blank — F6)", label, systemDriveLabel)
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,561 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
pathpkg "path"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// mandAbs builds the capture-set Abs the code produces: ComputeCaptureSet uses path.Join (slash) —
|
||||
// the 3-core separator rule — so on the Windows test host the mandatory path is drive + "/rel".
|
||||
func mandAbs(drive, rel string) string { return pathpkg.Join(drive, rel) }
|
||||
|
||||
// offbox3aProvider is a configurable StackDataProvider for the 3a capture-set tests: per-stack HDD
|
||||
// path + classified binds.
|
||||
type offbox3aProvider struct {
|
||||
hdd map[string]string
|
||||
binds map[string][]ClassifiedBind
|
||||
has map[string]bool
|
||||
// deployed is OPT-IN and defaults to nil, so every existing fixture keeps ListDeployedStacks()
|
||||
// returning nil and nothing about their behaviour moves. R-234's classification is the only
|
||||
// thing that needs a real deployed set.
|
||||
deployed map[string]bool
|
||||
}
|
||||
|
||||
func (p *offbox3aProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *offbox3aProvider) ListDeployedStacks() []StackSummary {
|
||||
if len(p.deployed) == 0 {
|
||||
return nil
|
||||
}
|
||||
out := make([]StackSummary, 0, len(p.deployed))
|
||||
for n := range p.deployed {
|
||||
out = append(out, StackSummary{Name: n})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *offbox3aProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *offbox3aProvider) GetStackHDDPath(n string) string { return p.hdd[n] }
|
||||
func (p *offbox3aProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
func (p *offbox3aProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *offbox3aProvider) StopStack(string) error { return nil }
|
||||
func (p *offbox3aProvider) StartStack(string) error { return nil }
|
||||
func (p *offbox3aProvider) RefreshAndIsRunning(string) bool { return false }
|
||||
func (p *offbox3aProvider) GetStackRecoveryInfo(string) (RecoveryInfo, bool) {
|
||||
return RecoveryInfo{}, false
|
||||
}
|
||||
func (p *offbox3aProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *offbox3aProvider) RecreateStackDefinitionFromUnit(_, _ string, _ map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *offbox3aProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *offbox3aProvider) GetStackClassifiedBinds(n string) ([]ClassifiedBind, bool) {
|
||||
return p.binds[n], p.has[n]
|
||||
}
|
||||
|
||||
func mandatoryHDD(rel string) ClassifiedBind {
|
||||
return ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootHDD, RelPath: rel}, Class: appbackup.ClassMandatory}
|
||||
}
|
||||
func optionalUserdata(rel string) ClassifiedBind {
|
||||
return ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootUserdata, RelPath: rel, ReadOnly: true}, Class: appbackup.ClassOptional}
|
||||
}
|
||||
func excludedHDD(rel string) ClassifiedBind {
|
||||
return ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootHDD, RelPath: rel}, Class: appbackup.ClassExcluded}
|
||||
}
|
||||
|
||||
// classifiedOffboxManager: a configured offbox manager + a classified provider + the drive registered
|
||||
// as a schedulable storage path (so discoverOffboxUnit finds units on it).
|
||||
func classifiedOffboxManager(t *testing.T, drive string) (*Manager, *settings.Settings, *offbox3aProvider) {
|
||||
t.Helper()
|
||||
m, sett := newOffboxManager(t)
|
||||
prov := &offbox3aProvider{hdd: map[string]string{}, binds: map[string][]ClassifiedBind{}, has: map[string]bool{}}
|
||||
m.SetStackProvider(prov)
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: drive, Label: "USB", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return m, sett, prov
|
||||
}
|
||||
|
||||
// mkUnit lays down a discoverable recovery unit for stack on drive.
|
||||
func mkUnit(t *testing.T, drive, stack string) string {
|
||||
t.Helper()
|
||||
u := RecoveryUnitPath(drive, stack)
|
||||
if err := os.MkdirAll(u, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return u
|
||||
}
|
||||
|
||||
// captureBackupRunner records the FULL argv of each backup call (keyed by stack tag) + forget argv, and
|
||||
// answers the probes so RunOffboxBackup completes.
|
||||
type backupCapture struct {
|
||||
mu sync.Mutex
|
||||
byStack map[string][]string
|
||||
forgets [][]string
|
||||
backups int
|
||||
}
|
||||
|
||||
func (c *backupCapture) runner() offboxRunner {
|
||||
c.byStack = map[string][]string{}
|
||||
return func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
switch {
|
||||
case contains(args, "cat") && contains(args, "config"):
|
||||
return []byte(`{"version":2}`), nil
|
||||
case contains(args, "backup"):
|
||||
c.backups++
|
||||
c.byStack[tagOf(args)] = append([]string{}, args...)
|
||||
return nil, nil
|
||||
case contains(args, "forget"):
|
||||
c.forgets = append(c.forgets, append([]string{}, args...))
|
||||
return nil, nil
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[]`), nil
|
||||
case contains(args, "stats"):
|
||||
return []byte(`{"total_size":123}`), nil
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario A: classified enlarged push (immich shape) ---
|
||||
|
||||
func TestOffbox3a_EnlargedPush_MandatoryOnly(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
unit := mkUnit(t, drive, "immich")
|
||||
if err := os.MkdirAll(filepath.Join(drive, "appdata", "immich"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The optional :ro library exists on disk — so if the tier filter ever leaked it, the stat-filter
|
||||
// would NOT hide it (this makes the RP-A tier-filter red-proof observable).
|
||||
if err := os.MkdirAll(filepath.Join(drive, "userdata", "media", "photos"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["immich"] = drive
|
||||
prov.has["immich"] = true
|
||||
prov.binds["immich"] = []ClassifiedBind{mandatoryHDD("appdata/immich"), optionalUserdata("media/photos")}
|
||||
_ = sett.SetAppOffbox("immich", true)
|
||||
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
if cap.backups != 1 {
|
||||
t.Fatalf("exactly ONE snapshot per app, got %d backup calls", cap.backups)
|
||||
}
|
||||
args := cap.byStack["immich"]
|
||||
wantMandatory := mandAbs(drive, "appdata/immich")
|
||||
if !contains(args, unit) {
|
||||
t.Errorf("backup argv missing the unit path %q: %v", unit, args)
|
||||
}
|
||||
if !contains(args, wantMandatory) {
|
||||
t.Errorf("backup argv missing the mandatory userdata path %q: %v", wantMandatory, args)
|
||||
}
|
||||
if contains(args, mandAbs(drive, "userdata/media/photos")) {
|
||||
t.Errorf("OPTIONAL :ro path must NOT ship offsite: %v", args)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario B: legacy / undeployed stay unit-only ---
|
||||
|
||||
func TestOffbox3a_LegacyUnitOnly(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
unit := mkUnit(t, drive, "sonarr")
|
||||
if err := os.MkdirAll(filepath.Join(drive, "appdata", "sonarr"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["sonarr"] = drive
|
||||
prov.has["sonarr"] = false // block REJECTED / absent → legacy (binds present but no class semantics)
|
||||
prov.binds["sonarr"] = []ClassifiedBind{mandatoryHDD("appdata/sonarr")}
|
||||
_ = sett.SetAppOffbox("sonarr", true)
|
||||
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
args := cap.byStack["sonarr"]
|
||||
// unit-only: exactly the base shape, last arg is the unit, no extra resolved paths.
|
||||
if args[len(args)-1] != unit {
|
||||
t.Errorf("legacy app argv must END at the unit (no resolved paths), got %v", args)
|
||||
}
|
||||
for _, a := range args {
|
||||
if strings.Contains(a, "appdata") || strings.Contains(a, "userdata") {
|
||||
t.Errorf("legacy app resolved a bind into offsite argv (SQ5 regression): %v", args)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffbox3a_UndeployedUnitOnlyWithWarning(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
_ = mkUnit(t, drive, "immich")
|
||||
prov.hdd["immich"] = "" // undeployed → no live HDD_PATH
|
||||
prov.has["immich"] = true
|
||||
prov.binds["immich"] = []ClassifiedBind{mandatoryHDD("appdata/immich")}
|
||||
_ = sett.SetAppOffbox("immich", true)
|
||||
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
args := cap.byStack["immich"]
|
||||
if strings.Contains(strings.Join(args, " "), "appdata") {
|
||||
t.Errorf("undeployed app must push unit-only: %v", args)
|
||||
}
|
||||
if w := sett.GetOffboxTarget().LastWarning; !strings.Contains(w, "nincs telepítve") {
|
||||
t.Errorf("undeployed warning missing from LastWarning: %q", w)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario C: pre-push enlargement gate ---
|
||||
|
||||
func TestOffbox3a_EnlargementGateBlocks(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
for _, app := range []string{"immich", "small"} {
|
||||
_ = mkUnit(t, drive, app)
|
||||
if err := os.MkdirAll(filepath.Join(drive, "appdata", app), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd[app] = drive
|
||||
prov.has[app] = true
|
||||
prov.binds[app] = []ClassifiedBind{mandatoryHDD("appdata/" + app)}
|
||||
_ = sett.SetAppOffbox(app, true)
|
||||
}
|
||||
_ = sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.QuotaGB = 50; o.RepoSizeBytes = 20 << 30 })
|
||||
// immich's mandatory set is 40 GiB (20+40 ≥ 50 → blocked); small's is 1 GiB (20+1 < 50 → fits).
|
||||
m.SetOffboxSizer(func(p string) int64 {
|
||||
if strings.Contains(p, "immich") {
|
||||
return 40 << 30
|
||||
}
|
||||
return 1 << 30
|
||||
})
|
||||
var noteMu sync.Mutex
|
||||
var notes []string
|
||||
m.SetOffboxEnlargeBlockedNotifier(func(stack string, _ int64, usedGB, quotaGB int) {
|
||||
noteMu.Lock()
|
||||
defer noteMu.Unlock()
|
||||
notes = append(notes, stack)
|
||||
if usedGB != 20 || quotaGB != 50 {
|
||||
t.Errorf("notifier numbers wrong: used=%d quota=%d", usedGB, quotaGB)
|
||||
}
|
||||
})
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run must be OK (a blocked enlargement is not a run failure): %v", err)
|
||||
}
|
||||
// immich → unit-only; small → enlarged.
|
||||
if strings.Contains(strings.Join(cap.byStack["immich"], " "), "appdata") {
|
||||
t.Errorf("blocked immich must be unit-only: %v", cap.byStack["immich"])
|
||||
}
|
||||
if !contains(cap.byStack["small"], mandAbs(drive, "appdata/small")) {
|
||||
t.Errorf("fitting 'small' must still push enlarged: %v", cap.byStack["small"])
|
||||
}
|
||||
tgt := sett.GetOffboxTarget()
|
||||
if len(tgt.EnlargedBlocked) != 1 || tgt.EnlargedBlocked[0] != "immich" {
|
||||
t.Errorf("EnlargedBlocked = %v, want [immich]", tgt.EnlargedBlocked)
|
||||
}
|
||||
if tgt.LastStatus != "ok" {
|
||||
t.Errorf("run status = %q, want ok", tgt.LastStatus)
|
||||
}
|
||||
if !strings.Contains(tgt.LastWarning, "tárhelykeret miatt") || !strings.Contains(tgt.LastWarning, "immich") {
|
||||
t.Errorf("blocked LastWarning missing: %q", tgt.LastWarning)
|
||||
}
|
||||
if len(notes) != 1 || notes[0] != "immich" {
|
||||
t.Errorf("notifier must fire ONCE for immich, got %v", notes)
|
||||
}
|
||||
|
||||
// EnlargedBlocked clears on a subsequent run where nothing is blocked.
|
||||
m.SetOffboxSizer(func(string) int64 { return 1 << 30 }) // now immich fits too
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if b := sett.GetOffboxTarget().EnlargedBlocked; len(b) != 0 {
|
||||
t.Errorf("EnlargedBlocked must clear when nothing is blocked, got %v", b)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario D: capture gaps are loud (SP-3.4) ---
|
||||
|
||||
func TestOffbox3a_CaptureGapsAreLoud(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
_ = mkUnit(t, drive, "app")
|
||||
if err := os.MkdirAll(filepath.Join(drive, "appdata", "good"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["app"] = drive
|
||||
prov.has["app"] = true
|
||||
prov.binds["app"] = []ClassifiedBind{
|
||||
mandatoryHDD("appdata/good"), // exists → captured
|
||||
mandatoryHDD("../evil"), // D1: traversal → Skipped
|
||||
mandatoryHDD("appdata/ghost"), // D2: passes guards but absent on disk → stat-filtered
|
||||
}
|
||||
_ = sett.SetAppOffbox("app", true)
|
||||
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
args := cap.byStack["app"]
|
||||
joined := strings.Join(args, " ")
|
||||
if !contains(args, mandAbs(drive, "appdata/good")) {
|
||||
t.Errorf("the valid mandatory path must still push: %v", args)
|
||||
}
|
||||
if strings.Contains(joined, "evil") {
|
||||
t.Errorf("traversal path escaped into argv: %v", args)
|
||||
}
|
||||
if strings.Contains(joined, "ghost") {
|
||||
t.Errorf("stat-missing mandatory path must NOT be in argv (SP-3.4 silent-skip): %v", args)
|
||||
}
|
||||
if w := sett.GetOffboxTarget().LastWarning; !strings.Contains(w, "nem kerültek a távoli mentésbe") {
|
||||
t.Errorf("capture-gap warning missing from LastWarning: %q", w)
|
||||
}
|
||||
}
|
||||
|
||||
// --- §8 all-excluded row (radarr shape): unit-only, NO warning ---
|
||||
|
||||
func TestOffbox3a_AllExcludedUnitOnlyNoWarning(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
unit := mkUnit(t, drive, "radarr")
|
||||
prov.hdd["radarr"] = drive
|
||||
prov.has["radarr"] = true
|
||||
prov.binds["radarr"] = []ClassifiedBind{excludedHDD("appdata/radarr"), excludedHDD("downloads")}
|
||||
_ = sett.SetAppOffbox("radarr", true)
|
||||
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
if args := cap.byStack["radarr"]; args[len(args)-1] != unit {
|
||||
t.Errorf("all-excluded app must be unit-only: %v", args)
|
||||
}
|
||||
if w := sett.GetOffboxTarget().LastWarning; strings.Contains(w, "nem kerültek") {
|
||||
t.Errorf("all-excluded is correct, NOT a gap — no warning expected, got %q", w)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario E: raw-data stats mode ---
|
||||
|
||||
func TestOffbox3a_StatsRawDataMode(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
var statsArgs []string
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
switch {
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[{"id":"a"}]`), nil
|
||||
case contains(args, "stats"):
|
||||
statsArgs = append([]string{}, args...)
|
||||
return []byte(`{"total_size":987654321}`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
base, env := m.offboxBaseArgs(sett.GetOffboxTarget())
|
||||
m.offboxRecordStats(context.Background(), base, env)
|
||||
if !contains(statsArgs, "--mode") || valAfter(statsArgs, "--mode") != "raw-data" {
|
||||
t.Fatalf("stats must run in raw-data mode, got %v", statsArgs)
|
||||
}
|
||||
if got := sett.GetOffboxTarget().RepoSizeBytes; got != 987654321 {
|
||||
t.Errorf("RepoSizeBytes = %d, want 987654321 (parsed from raw-data total_size)", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario F: both forget call sites carry --group-by host,tags ---
|
||||
|
||||
func TestOffbox3a_ForgetGrouping_MainRun(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
_ = mkUnit(t, drive, "app")
|
||||
prov.has["app"] = false
|
||||
_ = sett.SetAppOffbox("app", true)
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(cap.forgets) != 1 {
|
||||
t.Fatalf("expected one forget call, got %d", len(cap.forgets))
|
||||
}
|
||||
if valAfter(cap.forgets[0], "--group-by") != "host,tags" {
|
||||
t.Errorf("main-run forget missing --group-by host,tags: %v", cap.forgets[0])
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffbox3a_ForgetGrouping_OverQuotaPrune(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
_ = sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.QuotaGB = 50; o.RepoSizeBytes = 51 << 30 })
|
||||
var forgetArgs []string
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
switch {
|
||||
case contains(args, "cat") && contains(args, "config"):
|
||||
return []byte(`{}`), nil
|
||||
case contains(args, "forget"):
|
||||
forgetArgs = append([]string{}, args...)
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[]`), nil
|
||||
case contains(args, "stats"):
|
||||
return []byte(`{"total_size":1}`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
_ = m.RunOffboxBackup(context.Background()) // over-quota → prune-only path
|
||||
if valAfter(forgetArgs, "--group-by") != "host,tags" {
|
||||
t.Errorf("over-quota prune forget missing --group-by host,tags: %v", forgetArgs)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario E-restore: unit-only restore argv (ID-first + --include) + scratch OFF the rootfs ---
|
||||
|
||||
func TestOffbox3a_UnitOnlyRestoreArgv(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["immich"] = drive
|
||||
m.SetOffboxFreeFn(func(string) int64 { return 100 << 30 }) // plenty
|
||||
unitPath := filepath.ToSlash(filepath.Join(drive, "backups", "primary", "immich"))
|
||||
var restoreArgs []string
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
switch {
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[{"short_id":"deadbeef","time":"2026-07-14T00:00:00Z","paths":["` + filepath.ToSlash(filepath.Join(drive, "appdata", "immich")) + `","` + unitPath + `"]}]`), nil
|
||||
case contains(args, "restore"):
|
||||
restoreArgs = append([]string{}, args...)
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
if err := m.RestoreOffboxScratch(context.Background(), "immich", false); err != nil {
|
||||
t.Fatalf("unit-only restore: %v", err)
|
||||
}
|
||||
if valAfter(restoreArgs, "restore") != "deadbeef" {
|
||||
t.Errorf("restore must be ID-first (deadbeef): %v", restoreArgs)
|
||||
}
|
||||
if valAfter(restoreArgs, "--include") != unitPath {
|
||||
t.Errorf("unit-only restore must --include the absolute unit path %q: %v", unitPath, restoreArgs)
|
||||
}
|
||||
target := valAfter(restoreArgs, "--target")
|
||||
if !strings.HasPrefix(target, drive) || strings.Contains(target, m.cfg.Paths.DataDir) {
|
||||
t.Errorf("scratch target must be on the data drive, never DataDir: %q", target)
|
||||
}
|
||||
}
|
||||
|
||||
// full restore refuses fail-closed when the snapshot size is unknown (no restore call made).
|
||||
func TestOffbox3a_FullRestoreRefusesOnSizeUnknown(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["immich"] = drive
|
||||
m.SetOffboxFreeFn(func(string) int64 { return 100 << 30 })
|
||||
unitPath := filepath.Join(drive, "backups", "primary", "immich")
|
||||
restoreCalled := false
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
switch {
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[{"short_id":"a","time":"2026-07-14T00:00:00Z","paths":["` + filepath.ToSlash(unitPath) + `"]}]`), nil
|
||||
case contains(args, "stats"):
|
||||
return nil, context.DeadlineExceeded // size lookup fails → unknown
|
||||
case contains(args, "restore"):
|
||||
restoreCalled = true
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
err := m.RestoreOffboxScratch(context.Background(), "immich", true)
|
||||
if err == nil || !strings.Contains(err.Error(), "nem állapítható meg") {
|
||||
t.Fatalf("full restore must refuse fail-closed on unknown size, got err=%v", err)
|
||||
}
|
||||
if restoreCalled {
|
||||
t.Error("no restic restore call may run when the size is unknown")
|
||||
}
|
||||
}
|
||||
|
||||
// old rootfs scratch is cleaned up on a new restore.
|
||||
func TestOffbox3a_LegacyRootfsScratchCleanup(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["immich"] = drive
|
||||
m.SetOffboxFreeFn(func(string) int64 { return 100 << 30 })
|
||||
legacy := filepath.Join(m.cfg.Paths.DataDir, "offbox-restore", "immich")
|
||||
if err := os.MkdirAll(legacy, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
unitPath := filepath.Join(drive, "backups", "primary", "immich")
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[{"short_id":"a","time":"2026-07-14T00:00:00Z","paths":["` + filepath.ToSlash(unitPath) + `"]}]`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
if err := m.RestoreOffboxScratch(context.Background(), "immich", false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := os.Stat(legacy); !os.IsNotExist(err) {
|
||||
t.Errorf("legacy rootfs scratch %s must be removed, stat err=%v", legacy, err)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario G: place-to-live mapping (pure) + the wrong cases ---
|
||||
|
||||
func TestMapOffsiteRestorePaths(t *testing.T) {
|
||||
old := "/old/ns"
|
||||
newNs := "/new/ns"
|
||||
scratch := "/scratch"
|
||||
snap := []string{
|
||||
old + "/backups/primary/app",
|
||||
old + "/appdata/app",
|
||||
old + "/userdata/media/x",
|
||||
}
|
||||
got, err := mapOffsiteRestorePaths(snap, "app", scratch, newNs)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if len(got) != 3 {
|
||||
t.Fatalf("got %d placements, want 3: %+v", len(got), got)
|
||||
}
|
||||
byDst := map[string]placement{}
|
||||
for _, pl := range got {
|
||||
byDst[pl.dst] = pl
|
||||
}
|
||||
// anchor derived by trimming backups/primary/app off the unit path → oldNs; dst = newNs/<rel>,
|
||||
// src = scratch/<abs-source> (SP-3.1). Built with filepath.Join to match the code (OS separators).
|
||||
check := func(snapPath, rel string, isUnit bool) {
|
||||
dst := filepath.Join(newNs, rel)
|
||||
pl, ok := byDst[dst]
|
||||
if !ok {
|
||||
t.Errorf("missing placement for dst %q", dst)
|
||||
return
|
||||
}
|
||||
if pl.src != filepath.Join(scratch, snapPath) {
|
||||
t.Errorf("src for %q = %q, want %q", snapPath, pl.src, filepath.Join(scratch, snapPath))
|
||||
}
|
||||
if pl.isUnit != isUnit {
|
||||
t.Errorf("isUnit for %q = %v, want %v", snapPath, pl.isUnit, isUnit)
|
||||
}
|
||||
}
|
||||
check(old+"/backups/primary/app", "backups/primary/app", true)
|
||||
check(old+"/appdata/app", "appdata/app", false)
|
||||
check(old+"/userdata/media/x", "userdata/media/x", false)
|
||||
|
||||
// Wrong cases — each REFUSES the whole placement.
|
||||
if _, err := mapOffsiteRestorePaths([]string{old + "/appdata/app"}, "app", scratch, newNs); err == nil {
|
||||
t.Error("no unit path → must refuse")
|
||||
}
|
||||
if _, err := mapOffsiteRestorePaths([]string{old + "/backups/primary/app", "/elsewhere/x"}, "app", scratch, newNs); err == nil {
|
||||
t.Error("a path outside the namespace → must refuse")
|
||||
}
|
||||
if _, err := mapOffsiteRestorePaths([]string{old + "/backups/primary/app", old + "/backups/secondary/y"}, "app", scratch, newNs); err == nil {
|
||||
t.Error("a non-unit path in the reserved backups/ zone → must refuse")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,299 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// ABANDONMENT — deciding to give up the old off-site history is a finishable thing (R-241, v0.206.0).
|
||||
//
|
||||
// THE PROBLEM THIS SOLVES. `resetOrphanedRepo` renamed the remote store aside and touched neither the
|
||||
// escrow nor the key, so the hub went on holding a sealed package for a key the box no longer used.
|
||||
// Shape (c) compares those two, finds them different, and offers recovery — correctly, and for ever.
|
||||
// A customer who has already said "I do not want the old data" would be asked again at every login.
|
||||
//
|
||||
// THE OPERATOR'S RULING (2026-08-07) is that the answer is NOT a "they decided" flag. A flag would
|
||||
// leave the box in a state that is genuinely wrong (the hub holding a package for a key nobody uses)
|
||||
// and paper over it. Instead the decision starts a **14-day countdown**, at the end of which the
|
||||
// set-aside store and the sealed package that protects it are removed TOGETHER — after which there is
|
||||
// nothing left to compare and nothing left to ask about. **Fix the state, do not remember that it is
|
||||
// wrong.**
|
||||
//
|
||||
// THE GRACE IS REAL, NOT DECORATIVE. The recovery offer stays reachable for the whole window; that is
|
||||
// the change-of-mind path (Scenario G). A grace period during which recovery is impossible would be
|
||||
// theatre.
|
||||
|
||||
// abandonGraceDays is the countdown the operator set. Reminders fire at 5, 3 and 1 days (see
|
||||
// AbandonRemindAtDays) — visible, reversible, and running out in public.
|
||||
const abandonGraceDays = 14
|
||||
|
||||
// AbandonGraceDays is the exported grace, for the customer-facing copy. The confirmation screen must
|
||||
// state the SAME number the countdown uses — a literal typed into prose is how a promise drifts away
|
||||
// from the code that keeps it.
|
||||
const AbandonGraceDays = abandonGraceDays
|
||||
|
||||
// AbandonRemindAtDays are the remaining-day marks at which the abandoning box reminds the customer.
|
||||
// Descending, so the surface can pick the first one that has been reached.
|
||||
var AbandonRemindAtDays = []int{5, 3, 1}
|
||||
|
||||
// abandonNow is the countdown's clock seam. Tests inject; nil → time.Now. It exists so the terminal
|
||||
// step can be driven deterministically — §7.4 forbids shortening a live timer to watch it fire,
|
||||
// because that is how an irreversible step gets tested once and regretted once.
|
||||
func (m *Manager) abandonNow() time.Time {
|
||||
if m.offboxNow != nil {
|
||||
return m.offboxNow()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
// SetOffboxClock injects the abandonment clock (tests only).
|
||||
func (m *Manager) SetOffboxClock(fn func() time.Time) { m.offboxNow = fn }
|
||||
|
||||
// startAbandonCountdown records the decision and the date the terminal step will run. Called by
|
||||
// resetOrphanedRepo AFTER the move-aside has succeeded — a countdown started before the store has
|
||||
// actually moved would count down to deleting a path that does not exist.
|
||||
func (m *Manager) startAbandonCountdown(setAsidePath string) {
|
||||
now := m.abandonNow().UTC()
|
||||
due := now.AddDate(0, 0, abandonGraceDays)
|
||||
// R-302: pin the hub's escrow key fingerprint HERE, at the decision — the one moment it is a fact
|
||||
// rather than something inferred later from an adjacent value. From now on the banner asks exactly
|
||||
// one question, "is the hub still holding that same package?", instead of guessing which key is
|
||||
// which. Written once and never refreshed: a field re-read at render answers a different question
|
||||
// and would silently restore the defect this replaces.
|
||||
pinned := ""
|
||||
if m.settings != nil {
|
||||
pinned, _ = m.settings.GetHubEscrowKeySHA256()
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonStartedAt = now.Format(time.RFC3339)
|
||||
o.AbandonAt = due.Format(time.RFC3339)
|
||||
o.AbandonRepoPath = setAsidePath
|
||||
o.AbandonPurgeRequested = false
|
||||
o.AbandonPinnedEscrowKeySHA256 = pinned
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not record the abandonment countdown: %v", err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment countdown started: the set-aside history at %s and the hub's sealed package "+
|
||||
"are removed together on %s (%d days). The recovery screen stays reachable until then.",
|
||||
setAsidePath, due.Format("2006-01-02"), abandonGraceDays)
|
||||
}
|
||||
|
||||
// AbandonState is the surface's read model. Zero value = nothing in progress.
|
||||
type AbandonState struct {
|
||||
Active bool // a countdown is running
|
||||
StartedAt time.Time //
|
||||
DueAt time.Time // when the terminal step runs
|
||||
DaysLeft int // ceiling, so "0 days left" only ever means "today"
|
||||
RepoPath string // the set-aside store awaiting deletion
|
||||
PurgeRequested bool // the store is gone; awaiting the hub to drop the sealed package
|
||||
// RetrievalStillOffered (R-302) — may the banner still say the set-aside copies can be retrieved
|
||||
// with the recovery code? TRUE only while the hub is holding the SAME sealed package it held when
|
||||
// the customer decided. Derived here, once, so the banner and anything else asking cannot disagree.
|
||||
//
|
||||
// FALSE covers: the package was replaced after the decision (a fresh escrow ceremony — the act that
|
||||
// cost both demo boxes their history); the hub reports an empty hash (a legacy package sealing no
|
||||
// repository password); and a countdown started before R-302, which carries no pin. All three are
|
||||
// "we cannot see that this is still true", and all three must read as such rather than as a promise.
|
||||
RetrievalStillOffered bool
|
||||
}
|
||||
|
||||
// AbandonStatus reports the countdown for the UI and the report. It never mutates.
|
||||
func (m *Manager) AbandonStatus() AbandonState {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil {
|
||||
return AbandonState{}
|
||||
}
|
||||
st := AbandonState{RepoPath: t.AbandonRepoPath, PurgeRequested: t.AbandonPurgeRequested}
|
||||
if t.AbandonAt == "" {
|
||||
return st
|
||||
}
|
||||
due, err := time.Parse(time.RFC3339, t.AbandonAt)
|
||||
if err != nil {
|
||||
// A malformed stamp must not silently mean "never due" — that would strand the store for ever
|
||||
// with a countdown the customer can see and nothing behind it.
|
||||
m.logger.Printf("[WARN] [offbox] abandonment due-date is unparseable (%q) — treating the countdown as NOT running: %v", t.AbandonAt, err)
|
||||
return st
|
||||
}
|
||||
st.Active, st.DueAt = true, due
|
||||
if s, serr := time.Parse(time.RFC3339, t.AbandonStartedAt); serr == nil {
|
||||
st.StartedAt = s
|
||||
}
|
||||
// R-302: the pinned fingerprint vs what the hub reports NOW. Both must be non-empty and equal.
|
||||
// Empty on either side is "we could not see", never "they match" — the settings comment on
|
||||
// HubEscrowKeySHA256 establishes that the hub sends "" for a package sealing no repo password.
|
||||
if cur, _ := m.settings.GetHubEscrowKeySHA256(); cur != "" &&
|
||||
t.AbandonPinnedEscrowKeySHA256 != "" && cur == t.AbandonPinnedEscrowKeySHA256 {
|
||||
st.RetrievalStillOffered = true
|
||||
}
|
||||
// Ceiling: a countdown with 30 minutes left says "1 day", never "0". Zero is reserved for due.
|
||||
remaining := due.Sub(m.abandonNow())
|
||||
if remaining <= 0 {
|
||||
st.DaysLeft = 0
|
||||
} else {
|
||||
st.DaysLeft = int((remaining + 24*time.Hour - time.Nanosecond) / (24 * time.Hour))
|
||||
}
|
||||
return st
|
||||
}
|
||||
|
||||
// CancelAbandon stops a running countdown — the change-of-mind path (Scenario G). Called when a
|
||||
// recovery succeeds: the customer has their code after all, and the history they were about to give
|
||||
// up is exactly what the code opens.
|
||||
//
|
||||
// It clears the schedule but KEEPS AbandonRepoPath, so the set-aside store remains nameable on the
|
||||
// backups page. Nothing has been deleted at this point by construction — the terminal step is the
|
||||
// only thing that deletes, and it has not run.
|
||||
func (m *Manager) CancelAbandon(reason string) {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || (t.AbandonAt == "" && !t.AbandonPurgeRequested) {
|
||||
return // nothing running — silent, so a healthy recovery does not log about a countdown
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonStartedAt, o.AbandonAt = "", ""
|
||||
o.AbandonPurgeRequested = false
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not cancel the abandonment countdown: %v", err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment countdown CANCELLED (%s) — the set-aside history at %s is kept and nothing was deleted", reason, t.AbandonRepoPath)
|
||||
}
|
||||
|
||||
// AbandonSweep is the daily terminal step. It is the ONLY thing in the product that deletes a
|
||||
// customer's off-site history, and it does so on a date the customer was shown.
|
||||
//
|
||||
// ⚠ IT REMOVES BOTH HALVES OR NEITHER — Scenario F. The set-aside store and the sealed package that
|
||||
// protects it are the two halves of one thing; removing only the store leaves the hub holding a
|
||||
// package for a key that opens nothing, and removing only the package leaves ciphertext nobody can
|
||||
// ever decrypt. Either is a state that asks a question nobody can answer.
|
||||
//
|
||||
// The two halves cannot be made atomic across two machines, so this is a two-phase commit with the
|
||||
// STORE FIRST and a durable marker: delete the remote store, record AbandonPurgeRequested, and keep
|
||||
// declaring it in the report until the hub's ACK stops reporting a superseded package. A crash
|
||||
// between the two leaves the marker set and the next sweep re-declares — it never leaves the pair
|
||||
// half-removed and silent.
|
||||
//
|
||||
// Returns (deleted, err). deleted=false with err=nil is the normal "nothing due" case.
|
||||
func (m *Manager) AbandonSweep(ctx context.Context) (bool, error) {
|
||||
st := m.AbandonStatus()
|
||||
// Phase 2 outstanding: the store is gone, the hub has not confirmed. Re-declare and wait.
|
||||
if st.PurgeRequested {
|
||||
m.logger.Printf("[DEBUG] [offbox] abandonment: the set-aside store is deleted; awaiting the hub to drop the sealed package")
|
||||
return false, nil
|
||||
}
|
||||
if !st.Active || st.DueAt.After(m.abandonNow()) {
|
||||
return false, nil // not due — quiet by construction on every healthy box
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || t.AbandonRepoPath == "" {
|
||||
m.logger.Printf("[WARN] [offbox] abandonment is due but no set-aside path is recorded — nothing deleted; clearing the countdown so it does not retry for ever")
|
||||
m.CancelAbandon("no set-aside path recorded")
|
||||
return false, fmt.Errorf("abandonment due with no recorded path")
|
||||
}
|
||||
port := t.Port
|
||||
if port == 0 {
|
||||
port = 22
|
||||
}
|
||||
m.logger.Printf("[WARN] [offbox] abandonment DUE — deleting the set-aside off-site history at %s (chosen by the customer on %s; this is irreversible)",
|
||||
t.AbandonRepoPath, st.StartedAt.Format("2006-01-02"))
|
||||
out, err := m.sshRunner()(ctx, t.Host, t.User, port, m.offboxKeyPath(), m.offboxKnownHosts(),
|
||||
"rm -rf "+shellQuote(t.AbandonRepoPath))
|
||||
if err != nil {
|
||||
// NOT cleared: a transport failure must retry tomorrow, not silently abandon the abandonment.
|
||||
m.logger.Printf("[ERROR] [offbox] abandonment: deleting the set-aside history failed — the countdown stays due and retries: %v: %s", err, truncate(out))
|
||||
return false, fmt.Errorf("delete set-aside history: %w", err)
|
||||
}
|
||||
// Phase 1 done. Record it durably BEFORE anything else, so a crash here re-declares rather than
|
||||
// forgetting that the store is already gone.
|
||||
if uerr := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPurgeRequested = true
|
||||
o.AbandonAt = "" // the schedule has fired; the marker now drives the rest
|
||||
}); uerr != nil {
|
||||
m.logger.Printf("[ERROR] [offbox] abandonment: the store was deleted but the marker could not be saved — the hub's package may outlive it: %v", uerr)
|
||||
return true, uerr
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment: set-aside history deleted; requesting the hub to drop the sealed package that protected it")
|
||||
if m.offboxOrphanEvent != nil {
|
||||
m.offboxOrphanEvent("offbox_abandon_completed", t.AbandonRepoPath)
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// ClearAbandonPurgeIfConfirmed closes the two-phase commit: once the hub's ACK stops reporting a
|
||||
// superseded package, both halves are gone and the abandonment is finished. Called from the ACK path.
|
||||
//
|
||||
// This is what makes §2.1 work without a "they decided" flag: afterwards the hub holds a package for
|
||||
// the key the box is actually using (or none at all), shape (c) has nothing to compare, and the
|
||||
// recovery offer falls silent on its own — because the state is right, not because something is
|
||||
// remembering that it once was not.
|
||||
func (m *Manager) ClearAbandonPurgeIfConfirmed(supersededPresent bool) {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || !t.AbandonPurgeRequested || supersededPresent {
|
||||
return
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPurgeRequested = false
|
||||
o.AbandonRepoPath = ""
|
||||
o.AbandonStartedAt = ""
|
||||
o.OrphanedRenamedTo = ""
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not close out the abandonment: %v", err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment COMPLETE — the set-aside history and the sealed package that protected it are both gone; nothing further to ask about")
|
||||
}
|
||||
|
||||
// ── OPERATOR CONTROL (§7.5) ─────────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// The automatic 30-day abandonment is deliberately NOT built (see R-245). What IS built is the path
|
||||
// that actually happens: **the customer gets in touch.** Someone who cannot find their recovery code
|
||||
// rings support, and support needs something to press — either "give them longer" or "stop it".
|
||||
//
|
||||
// Both live on the controller CLI rather than in the customer UI, deliberately: extending a deletion
|
||||
// the customer asked for is an operator judgement, not a self-service button, and a customer who
|
||||
// wants it stopped already has the self-service route — they recover with their code, which cancels
|
||||
// it (Scenario G).
|
||||
|
||||
// ExtendAbandon pushes the terminal step out by `days` from NOW. Returns the new due date.
|
||||
//
|
||||
// It refuses when no countdown is running: extending nothing would print a reassuring date for a
|
||||
// deletion that was never scheduled, which is the kind of comfort this project keeps removing.
|
||||
func (m *Manager) ExtendAbandon(days int) (time.Time, error) {
|
||||
if days <= 0 {
|
||||
return time.Time{}, fmt.Errorf("the extension must be a positive number of days")
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
if st.PurgeRequested {
|
||||
return time.Time{}, fmt.Errorf("too late: the set-aside history has already been deleted and only the sealed package is still being removed")
|
||||
}
|
||||
return time.Time{}, fmt.Errorf("no abandonment countdown is running on this box — nothing to extend")
|
||||
}
|
||||
due := m.abandonNow().UTC().AddDate(0, 0, days)
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonAt = due.Format(time.RFC3339)
|
||||
}); err != nil {
|
||||
return time.Time{}, fmt.Errorf("record the extension: %w", err)
|
||||
}
|
||||
m.logger.Printf("[WARN] [offbox] abandonment EXTENDED by an operator: the set-aside history at %s is now deleted on %s (was %s)",
|
||||
st.RepoPath, due.Format("2006-01-02"), st.DueAt.Format("2006-01-02"))
|
||||
return due, nil
|
||||
}
|
||||
|
||||
// StopAbandon cancels the countdown outright — the operator's version of Scenario G, for the
|
||||
// customer who telephoned instead of finding their code. The set-aside history is kept and nothing
|
||||
// is deleted; it is `CancelAbandon` with an operator's reason and a refusal when nothing is running,
|
||||
// so an operator never gets a silent no-op they might read as success.
|
||||
func (m *Manager) StopAbandon() error {
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
if st.PurgeRequested {
|
||||
return fmt.Errorf("too late: the set-aside history has already been deleted")
|
||||
}
|
||||
return fmt.Errorf("no abandonment countdown is running on this box — nothing to stop")
|
||||
}
|
||||
m.CancelAbandon("stopped by an operator")
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,208 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// ── R-302 — THE BANNER PROMISES ONLY WHAT THE BOX CAN STILL SEE IS TRUE ─────────────────────────
|
||||
//
|
||||
// The abandon banner said "until then you can still retrieve them with your recovery code",
|
||||
// unconditionally, on every page. Yesterday's reading proved that false on a reachable state.
|
||||
//
|
||||
// THE CONDITION IS A PIN, NOT A COMPARISON AGAINST THE CURRENT KEY, and the difference is the whole
|
||||
// design. The obvious proxy — "does the hub hold a key different from the one I use?" — asks about the
|
||||
// wrong key: the set-aside copies were written under an OLDER key the box no longer has, which is why
|
||||
// they were set aside. On a twice-rebuilt box the proxy answers "yes, promise it" about copies no key
|
||||
// on file can open. The pin instead records the package the hub held AT THE DECISION and asks only
|
||||
// "is the hub still holding that same one?".
|
||||
//
|
||||
// ⚠ THE PIN IS A RECORDED ASSUMPTION. It presumes the package held at the decision is the one that
|
||||
// opens the set-aside copies. Nothing on the box records which key wrote them. See the field comment
|
||||
// on settings.AbandonPinnedEscrowKeySHA256.
|
||||
//
|
||||
// The countdown is never started, shortened or triggered on a real machine — the clock is injected.
|
||||
|
||||
const pinnedHubKey = "1111111111111111111111111111111111111111111111111111111111111111"
|
||||
const replacedHubKey = "2222222222222222222222222222222222222222222222222222222222222222"
|
||||
|
||||
// startedCountdown drives the PRODUCTION path (ResetOrphanedRepo → resetOrphanedRepo →
|
||||
// startAbandonCountdown) so the pin cannot be written by tests alone while the live path never sets
|
||||
// it — the inert-seam shape that has shipped here before, fully green.
|
||||
func startedCountdown(t *testing.T, hubKeyAtDecision string) (*Manager, *settings.Settings, time.Time) {
|
||||
t.Helper()
|
||||
start := time.Date(2026, 8, 12, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, _ := abandonFixture(t, start)
|
||||
if err := sett.SetHubEscrowKeySHA256(hubKeyAtDecision, start.Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatalf("the production reset path failed: %v", err)
|
||||
}
|
||||
return m, sett, start
|
||||
}
|
||||
|
||||
// PRODUCTION WIRING: the live decision path writes the pin. If this fails, every render test below is
|
||||
// testing a field nothing sets.
|
||||
func TestR302_ProductionResetPathWritesThePin(t *testing.T) {
|
||||
_, sett, _ := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
got := sett.GetOffboxTarget().AbandonPinnedEscrowKeySHA256
|
||||
if got != pinnedHubKey {
|
||||
t.Fatalf("pinned fingerprint = %q, want the hub key cached at the decision (%q). The whole "+
|
||||
"design is that this is recorded when it is a fact; if the live path does not write it, "+
|
||||
"the banner falls to the cautious branch for ever and the grace period becomes theatre",
|
||||
got, pinnedHubKey)
|
||||
}
|
||||
if sett.GetOffboxTarget().AbandonAt == "" {
|
||||
t.Error("no countdown recorded — the fixture is not exercising the path it claims to")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO A — package unchanged since the decision → the promise stands ──────────────────────
|
||||
//
|
||||
// RED-PROOF: force the condition false (drop the `cur == t.AbandonPinnedEscrowKeySHA256` arm) and this
|
||||
// fails — a customer who can genuinely still change their mind loses the clause, which the code says
|
||||
// explicitly must not happen ("a grace period during which recovery is impossible would be theatre").
|
||||
func TestR302_ScenarioA_PackageUnchanged_RetrievalStillOffered(t *testing.T) {
|
||||
m, _, _ := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("countdown not active")
|
||||
}
|
||||
if !st.RetrievalStillOffered {
|
||||
t.Error("the hub still holds the same package it held at the decision, so the customer really " +
|
||||
"can still change their mind — the promise must stand")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO B — the package was REPLACED after the decision → promise withdrawn ────────────────
|
||||
//
|
||||
// This is the act that cost both demo boxes their history on 2026-08-04: a fresh escrow ceremony
|
||||
// supersedes the package, and the old key it covered is unreachable (superseded packages grant no
|
||||
// read path — hub store.go's own comment).
|
||||
//
|
||||
// RED-PROOF: re-read the pin at render (compare `cur` against itself, i.e. use the CURRENT cached
|
||||
// value on both sides) and this fails — the promise returns, which is today's defect.
|
||||
func TestR302_ScenarioB_PackageReplaced_PromiseWithdrawn(t *testing.T) {
|
||||
m, sett, start := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
// A fresh ceremony after the decision.
|
||||
if err := sett.SetHubEscrowKeySHA256(replacedHubKey, start.Add(48*time.Hour).Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := m.AbandonStatus(); st.RetrievalStillOffered {
|
||||
t.Error("the hub's package was replaced after the customer decided, so the key that opened the " +
|
||||
"set-aside copies is no longer served — the banner must stop promising retrieval")
|
||||
}
|
||||
// The pin itself must NOT have moved: it is written once, at the decision.
|
||||
if got := sett.GetOffboxTarget().AbandonPinnedEscrowKeySHA256; got != pinnedHubKey {
|
||||
t.Errorf("the pin was refreshed to %q — a field re-read later answers a different question and "+
|
||||
"silently restores the defect this replaces", got)
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO D — a countdown started BEFORE this shipped carries no pin ─────────────────────────
|
||||
//
|
||||
// RED-PROOF: backfill the pin from the current cached value when it is empty and this fails — a legacy
|
||||
// countdown gets promised at, asserting as recorded-at-the-decision something read long afterwards.
|
||||
func TestR302_ScenarioD_LegacyCountdownWithoutAPin_TakesTheCautiousBranch(t *testing.T) {
|
||||
m, sett, _ := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
// Model the pre-R-302 on-disk shape: a live countdown, no pin.
|
||||
if err := sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPinnedEscrowKeySHA256 = ""
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("countdown should still be running")
|
||||
}
|
||||
if st.RetrievalStillOffered {
|
||||
t.Error("a countdown with no pin was promised at. There is no honest way to know whether the " +
|
||||
"hub's package is still the one from the decision, and the cautious answer is the only one " +
|
||||
"available")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO E — pinned present, hub's cached value EMPTY → cautious ────────────────────────────
|
||||
//
|
||||
// The hub sends "" for a legacy package that provably seals no repository password. Empty is a
|
||||
// measurement, not a match.
|
||||
//
|
||||
// RED-PROOF: treat empty as equal (drop the `cur != ""` arm) and this fails.
|
||||
func TestR302_ScenarioE_EmptyHubHash_IsNotAMatch(t *testing.T) {
|
||||
m, sett, start := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
if err := sett.SetHubEscrowKeySHA256("", start.Add(time.Hour).Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := m.AbandonStatus(); st.RetrievalStillOffered {
|
||||
t.Error("an EMPTY hub hash was read as a match. It means the hub holds a package that seals no " +
|
||||
"repository password — the opposite of evidence that retrieval works")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO F — no countdown → nothing about retrieval is claimed at all ───────────────────────
|
||||
func TestR302_ScenarioF_NoCountdown_NoClaim(t *testing.T) {
|
||||
start := time.Date(2026, 8, 12, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
|
||||
st := m.AbandonStatus()
|
||||
if st.Active {
|
||||
t.Fatal("no countdown was started, yet one is reported active")
|
||||
}
|
||||
if st.RetrievalStillOffered {
|
||||
t.Error("retrieval was offered with no countdown running — the flag must be meaningless " +
|
||||
"outside an abandonment, not default-true")
|
||||
}
|
||||
}
|
||||
|
||||
// The pin is a hash of a secret. It must never reach a customer-facing surface or the report; this
|
||||
// pins that it is not accidentally exported through the read model.
|
||||
func TestR302_PinIsNotExposedThroughTheReadModel(t *testing.T) {
|
||||
m, _, _ := startedCountdown(t, pinnedHubKey)
|
||||
st := m.AbandonStatus()
|
||||
if st.RepoPath == pinnedHubKey {
|
||||
t.Fatal("the pin leaked into RepoPath")
|
||||
}
|
||||
// AbandonState carries a BOOLEAN verdict, never the fingerprint itself.
|
||||
if got := st.RetrievalStillOffered; got != true && got != false {
|
||||
t.Fatal("unreachable")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO E, the case that actually bites — BOTH sides empty ─────────────────────────────────
|
||||
//
|
||||
// A legacy countdown (no pin) on a box whose hub reports an empty hash (a package sealing no repo
|
||||
// password). "" == "" is the equality that would quietly become a promise, and it is the ONLY state
|
||||
// where dropping the emptiness guards changes the answer — TestR302_ScenarioE above passes even with
|
||||
// them removed, because its pin is non-empty so the equality fails on its own. That test guards the
|
||||
// sentence; this one guards the claim.
|
||||
//
|
||||
// RED-PROOF: drop either `cur != ""` or `t.AbandonPinnedEscrowKeySHA256 != ""` and this fails.
|
||||
func TestR302_ScenarioE2_BothSidesEmpty_IsNotAMatch(t *testing.T) {
|
||||
m, sett, start := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
if err := sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPinnedEscrowKeySHA256 = "" // legacy countdown, no pin
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetHubEscrowKeySHA256("", start.Add(time.Hour).Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("countdown should still be running")
|
||||
}
|
||||
if st.RetrievalStillOffered {
|
||||
t.Error("two absences compared equal and became a promise. Empty means we could not see; two " +
|
||||
"things we could not see are not a match")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,335 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-241 — abandoning starts a countdown that ENDS THE QUESTION (Scenarios E, F, G).
|
||||
//
|
||||
// The countdown is driven by an injected clock throughout. §7.4 forbids shortening a live timer to
|
||||
// watch the terminal step fire: it is the only thing in the product that deletes a customer's
|
||||
// off-site history, and a step tested once on real data is a step regretted once.
|
||||
|
||||
// abandonFixture: an orphaned, configured box holding a key, with the hub holding a package for a
|
||||
// DIFFERENT key — i.e. shape (c) is live and the customer is being offered recovery.
|
||||
// Returns the manager and a recorder of every remote shell command issued.
|
||||
type sshRecorder struct{ cmds []string }
|
||||
|
||||
func (r *sshRecorder) run(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
|
||||
r.cmds = append(r.cmds, remoteCmd)
|
||||
return []byte(""), nil
|
||||
}
|
||||
|
||||
func abandonFixture(t *testing.T, now time.Time) (*Manager, *settings.Settings, *sshRecorder) {
|
||||
t.Helper()
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, now.Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.EscrowState = "escrowed"
|
||||
o.RepoState = "orphaned"
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rec := &sshRecorder{}
|
||||
m.SetOffboxSSH(rec.run)
|
||||
m.SetOffboxRunner(func(ctx context.Context, env []string, args ...string) ([]byte, error) { return []byte(""), nil })
|
||||
m.SetOffboxClock(func() time.Time { return now })
|
||||
return m, sett, rec
|
||||
}
|
||||
|
||||
// ── SCENARIO E — abandoning sets aside, keeps the package, starts a countdown, stays reversible ──
|
||||
func TestR241_ScenarioE_AbandonStartsAReversibleCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, rec := abandonFixture(t, start)
|
||||
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatalf("abandon: %v", err)
|
||||
}
|
||||
// The store was MOVED, not deleted — no rm anywhere in this phase.
|
||||
joined := strings.Join(rec.cmds, " | ")
|
||||
if !strings.Contains(joined, "mv ") {
|
||||
t.Errorf("the old store must be moved aside; commands were: %s", joined)
|
||||
}
|
||||
if strings.Contains(joined, "rm -rf") {
|
||||
t.Fatalf("NOTHING may be deleted when the customer abandons — only at the end of the grace. Commands: %s", joined)
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("a countdown must be running after an abandonment")
|
||||
}
|
||||
if got := st.DueAt.Sub(start); got != abandonGraceDays*24*time.Hour {
|
||||
t.Errorf("countdown length = %v, want %d days", got, abandonGraceDays)
|
||||
}
|
||||
if st.DaysLeft != abandonGraceDays {
|
||||
t.Errorf("DaysLeft = %d, want %d", st.DaysLeft, abandonGraceDays)
|
||||
}
|
||||
if st.RepoPath == "" {
|
||||
t.Error("the set-aside path must be recorded, or the terminal step has nothing to delete")
|
||||
}
|
||||
// THE GRACE IS REAL: the recovery offer stays reachable for the whole window. A grace in which
|
||||
// recovery is impossible would be decorative.
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("the recovery offer MUST stay reachable during the grace — that is the change-of-mind path")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO G — changing your mind inside the window ───────────────────────────────────────────
|
||||
//
|
||||
// RED-PROOF: make the countdown uncancellable (delete the body of CancelAbandon). The countdown then
|
||||
// survives a successful recovery and this test fails — a customer who proved they hold their code
|
||||
// would still have the history deleted under them.
|
||||
func TestR241_ScenarioG_RecoveryInsideTheWindowCancelsTheCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
day6 := start.AddDate(0, 0, 6)
|
||||
m.SetOffboxClock(func() time.Time { return day6 })
|
||||
if st := m.AbandonStatus(); !st.Active || st.DaysLeft != 8 {
|
||||
t.Fatalf("precondition: day 6 of 14 should leave 8 days, got %+v", st)
|
||||
}
|
||||
pathBefore := m.AbandonStatus().RepoPath
|
||||
|
||||
m.CancelAbandon("the customer recovered with their code")
|
||||
|
||||
st := m.AbandonStatus()
|
||||
if st.Active {
|
||||
t.Fatal("a countdown must be cancellable — the customer found their code")
|
||||
}
|
||||
if st.RepoPath != pathBefore {
|
||||
t.Errorf("the set-aside store must stay NAMEABLE after a cancel: got %q want %q", st.RepoPath, pathBefore)
|
||||
}
|
||||
// And a sweep now deletes nothing, on any later date.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 90) })
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if err != nil || deleted {
|
||||
t.Fatalf("a cancelled countdown must never delete: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO F — the countdown ends the question, and removes BOTH halves ────────────────────────
|
||||
//
|
||||
// RED-PROOF (store half): make AbandonSweep skip the rm. The first assertion fails.
|
||||
// RED-PROOF (package half): drop AbandonPurgeRequested from OffboxReportStatus. The declaration
|
||||
// assertion fails — the hub is never asked and the package outlives the store for ever.
|
||||
func TestR241_ScenarioF_TerminalStepRemovesBothHalvesTogether(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, rec := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
setAside := m.AbandonStatus().RepoPath
|
||||
|
||||
// Not due yet — nothing happens, quietly.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 13) })
|
||||
if deleted, err := m.AbandonSweep(context.Background()); deleted || err != nil {
|
||||
t.Fatalf("day 13 must not delete: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
|
||||
// Due.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 14).Add(time.Minute) })
|
||||
rec.cmds = nil
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("terminal step: %v", err)
|
||||
}
|
||||
if !deleted {
|
||||
t.Fatal("the terminal step must delete when due")
|
||||
}
|
||||
// HALF 1: the store is gone.
|
||||
joined := strings.Join(rec.cmds, " | ")
|
||||
if !strings.Contains(joined, "rm -rf") || !strings.Contains(joined, setAside) {
|
||||
t.Fatalf("the set-aside store at %s must be deleted; commands: %s", setAside, joined)
|
||||
}
|
||||
// HALF 2: the hub is ASKED for the package, and keeps being asked until it confirms.
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil || !st.AbandonPurgeRequested {
|
||||
t.Fatalf("the report must declare abandon_purge_requested until the hub drops the package, got %+v", st)
|
||||
}
|
||||
// It repeats — a lost request must retry rather than leave the pair half-removed.
|
||||
if d2, err2 := m.AbandonSweep(context.Background()); d2 || err2 != nil {
|
||||
t.Fatalf("a second sweep must be a quiet no-op while awaiting the hub: deleted=%v err=%v", d2, err2)
|
||||
}
|
||||
if st2 := m.OffboxReportStatus(); st2 == nil || !st2.AbandonPurgeRequested {
|
||||
t.Fatal("the declaration must persist across sweeps until confirmed")
|
||||
}
|
||||
|
||||
// The hub confirms by no longer reporting a superseded package → the question is over.
|
||||
m.ClearAbandonPurgeIfConfirmed(false)
|
||||
if got := sett.GetOffboxTarget(); got.AbandonPurgeRequested || got.AbandonRepoPath != "" || got.AbandonAt != "" {
|
||||
t.Errorf("the abandonment must be fully closed out, got %+v", got)
|
||||
}
|
||||
if st3 := m.OffboxReportStatus(); st3 != nil && st3.AbandonPurgeRequested {
|
||||
t.Error("the declaration must stop once the hub has confirmed")
|
||||
}
|
||||
}
|
||||
|
||||
// While the hub STILL reports a superseded package, the close-out must not fire — otherwise the box
|
||||
// stops asking and the package outlives the store silently, which is exactly half of Scenario F.
|
||||
func TestR241_PurgeIsNotClosedOutWhileThePackageRemains(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
if _, err := m.AbandonSweep(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.ClearAbandonPurgeIfConfirmed(true) // the hub STILL holds a retained package
|
||||
if !sett.GetOffboxTarget().AbandonPurgeRequested {
|
||||
t.Fatal("the request must stand while the hub still reports a superseded package")
|
||||
}
|
||||
}
|
||||
|
||||
// A transport failure during the terminal step must NOT clear the countdown — it retries tomorrow.
|
||||
// Silently abandoning the abandonment would leave the store for ever with nothing counting down.
|
||||
func TestR241_TerminalStepFailureKeepsTheCountdownDue(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetOffboxSSH(func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
|
||||
return []byte("ssh: connect to host nas.local port 22: No route to host"), context.DeadlineExceeded
|
||||
})
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if deleted || err == nil {
|
||||
t.Fatalf("a failed deletion must be reported, not swallowed: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.AbandonAt == "" || got.AbandonPurgeRequested {
|
||||
t.Fatalf("a failed terminal step must leave the countdown DUE and unrequested, got %+v", got)
|
||||
}
|
||||
if !m.AbandonStatus().Active {
|
||||
t.Error("the countdown must still be active so tomorrow's sweep retries")
|
||||
}
|
||||
}
|
||||
|
||||
// Quiet by construction: a box with no countdown does no work and says nothing (§ the daily job's
|
||||
// own contract). Asserted, because "it probably does nothing" is how a sweep with a bug hides.
|
||||
func TestR241_Sweep_QuietWhenNothingDue(t *testing.T) {
|
||||
m, _, rec := abandonFixture(t, time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC))
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if deleted || err != nil {
|
||||
t.Fatalf("a box with no countdown must be a pure no-op: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
if len(rec.cmds) != 0 {
|
||||
t.Fatalf("a no-op sweep must issue no remote commands, got %v", rec.cmds)
|
||||
}
|
||||
if m.AbandonStatus().Active {
|
||||
t.Error("no countdown should be reported")
|
||||
}
|
||||
}
|
||||
|
||||
// The UNCLAIMED auto-reset must NOT start a customer countdown — nobody decided anything there.
|
||||
// An as-delivered box tidying a stranger's leftover store must not put a 14-day deletion clock on it.
|
||||
func TestR241_UnclaimedAutoResetStartsNoCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
t2 := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t2)
|
||||
if err := m.resetOrphanedRepo(context.Background(), base, env, "auto (unclaimed)"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.AbandonStatus().Active {
|
||||
t.Fatal("the unclaimed auto-reset must not start a customer abandonment countdown")
|
||||
}
|
||||
}
|
||||
|
||||
// ── §7.5 — THE OPERATOR LEVERS ──────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// The automatic 30-day ending is deliberately NOT built (R-245). These are what IS built: the path
|
||||
// that actually happens is the customer telephoning, and support needs something to press.
|
||||
func TestR241_OperatorCanExtendARunningCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, rec := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
day10 := start.AddDate(0, 0, 10)
|
||||
m.SetOffboxClock(func() time.Time { return day10 })
|
||||
|
||||
due, err := m.ExtendAbandon(30)
|
||||
if err != nil {
|
||||
t.Fatalf("extend: %v", err)
|
||||
}
|
||||
if want := day10.AddDate(0, 0, 30); !due.Equal(want) {
|
||||
t.Errorf("new due = %v, want %v (from NOW, not from the old date)", due, want)
|
||||
}
|
||||
// The original date has passed and nothing is deleted, because the extension moved it.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
rec.cmds = nil
|
||||
if deleted, serr := m.AbandonSweep(context.Background()); deleted || serr != nil {
|
||||
t.Fatalf("an extended countdown must not fire on the old date: deleted=%v err=%v", deleted, serr)
|
||||
}
|
||||
if len(rec.cmds) != 0 {
|
||||
t.Fatalf("nothing may be deleted after an extension, got %v", rec.cmds)
|
||||
}
|
||||
}
|
||||
|
||||
func TestR241_OperatorCanStopARunningCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, rec := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.StopAbandon(); err != nil {
|
||||
t.Fatalf("stop: %v", err)
|
||||
}
|
||||
if m.AbandonStatus().Active {
|
||||
t.Fatal("the countdown must be stopped")
|
||||
}
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 90) })
|
||||
rec.cmds = nil
|
||||
if deleted, err := m.AbandonSweep(context.Background()); deleted || err != nil {
|
||||
t.Fatalf("a stopped countdown must never delete: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
if len(rec.cmds) != 0 {
|
||||
t.Fatalf("a stopped countdown must issue no remote commands, got %v", rec.cmds)
|
||||
}
|
||||
}
|
||||
|
||||
// Both levers REFUSE when nothing is running. A silent no-op is the thing an operator most easily
|
||||
// mistakes for success — they would tell the customer it was handled.
|
||||
func TestR241_OperatorLeversRefuseWhenNothingIsRunning(t *testing.T) {
|
||||
m, _, _ := abandonFixture(t, time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC))
|
||||
if _, err := m.ExtendAbandon(30); err == nil {
|
||||
t.Error("extending a countdown that is not running must be an error, never a quiet success")
|
||||
}
|
||||
if err := m.StopAbandon(); err == nil {
|
||||
t.Error("stopping a countdown that is not running must be an error, never a quiet success")
|
||||
}
|
||||
if _, err := m.ExtendAbandon(0); err == nil {
|
||||
t.Error("a non-positive extension must be refused")
|
||||
}
|
||||
}
|
||||
|
||||
// Once the store is deleted there is nothing left to extend or stop, and saying otherwise would be
|
||||
// the worst kind of reassurance: an operator telling a customer their data is safe when it is gone.
|
||||
func TestR241_OperatorLeversRefuseAfterTheDeletion(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
if _, err := m.AbandonSweep(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := m.ExtendAbandon(30); err == nil {
|
||||
t.Error("extending after the deletion must be refused — there is nothing left to save")
|
||||
}
|
||||
if err := m.StopAbandon(); err == nil {
|
||||
t.Error("stopping after the deletion must be refused — there is nothing left to save")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// Offsite capture-set resolution (Task 3a, architecture doc §2/§6). Turns an app's Task-3-core
|
||||
// TierOffsite capture set (recovery unit + MANDATORY userdata only) into the extra absolute paths
|
||||
// appended to the app's restic snapshot, plus the Hungarian customer warnings for LOUD capture gaps.
|
||||
//
|
||||
// SP-3.4 is law here: restic 0.14.0 does NOT error on a missing source path — it skips with a warning,
|
||||
// exits 0, and silently writes a partial snapshot. So a skipped/missing MANDATORY path is detected in
|
||||
// THIS function (the structural-guard Skipped list + an os.Stat filter) and surfaced in BOTH the
|
||||
// English log and the Hungarian LastWarning. A restic exit code proves nothing about a missing path.
|
||||
|
||||
// offboxBlocked records an app whose enlarged (userdata-carrying) push was refused by the pre-push
|
||||
// quota gate. The unit-only push still proceeds (never a protection regression). estBytes is the
|
||||
// mandatory-set size estimate that would have been added.
|
||||
type offboxBlocked struct {
|
||||
stack string
|
||||
estBytes int64
|
||||
}
|
||||
|
||||
// offboxCaptureSet computes an app's OFFSITE mandatory capture paths to add to its recovery-unit
|
||||
// snapshot, plus any Hungarian warnings for capture gaps. It never returns optional/excluded paths
|
||||
// (the TierOffsite filter drops them — §2). Returns (nil, nil) for the legacy / no-provider / no-block
|
||||
// world: offsite stays UNIT-ONLY, byte-identical to pre-v0.134.0 (the SQ5 cost-regression guard).
|
||||
func (m *Manager) offboxCaptureSet(stack string) (extra []string, warns []string, gaps []string) {
|
||||
if m.stackProvider == nil {
|
||||
return nil, nil, nil // no provider wired → legacy world → unit only
|
||||
}
|
||||
binds, has := m.stackProvider.GetStackClassifiedBinds(stack)
|
||||
if !has {
|
||||
return nil, nil, nil // no backup block → legacy → unit only
|
||||
}
|
||||
// Resolve against the app's LIVE HDD_PATH (raw — NOT GetAppDrivePath, whose systemDataPath fallback
|
||||
// would resolve userdata onto the wrong drive). Empty ⇒ undeployed / no HDD (decision §2.4):
|
||||
// mandatory-path resolution needs the live HDD_PATH, so push unit-only + a loud WARN.
|
||||
hdd := strings.TrimSpace(m.stackProvider.GetStackHDDPath(stack))
|
||||
if hdd == "" {
|
||||
m.logger.Printf("[WARN] [offbox] %s: not deployed — offsite push is unit-only (mandatory userdata not resolvable)", stack)
|
||||
return nil, []string{fmt.Sprintf("Figyelmeztetés: a(z) %s nincs telepítve — csak a mentési egység került a távoli mentésbe.", stack)}, nil
|
||||
}
|
||||
nsRoot := m.namespaceRoot(hdd)
|
||||
cs := appbackup.ComputeCaptureSet(binds, has, appbackup.TierOffsite, nsRoot, m.stackProvider.GetImportRoot())
|
||||
|
||||
// Structurally-refused MANDATORY paths (traversal / bare drive-root / reserved backups/ zone) are
|
||||
// loud ERROR gaps — the path the customer thinks is protected is not in the snapshot.
|
||||
for _, sk := range cs.Skipped {
|
||||
if sk.Class == appbackup.ClassMandatory {
|
||||
m.logger.Printf("[ERROR] [offbox] %s: mandatory path refused by a structural guard (%s): %s/%s — NOT in the offsite snapshot",
|
||||
stack, sk.Reason, sk.Root, sk.RelPath)
|
||||
gaps = append(gaps, sk.RelPath)
|
||||
}
|
||||
}
|
||||
// Stat-filter (§2.5): a declared mandatory path absent on disk. restic would skip it SILENTLY
|
||||
// (SP-3.4), so drop it from argv AND warn — never a silent "looks backed up but isn't".
|
||||
//
|
||||
// R-203: the class check mirrors tier2_capture.go's ("optional-missing is silent"). It is a NO-OP
|
||||
// today — TierOffsite's tierKeeps() already admits ClassMandatory only, so cs.Paths cannot contain
|
||||
// an optional path here — and it is written anyway so the two tiers read the same and so the
|
||||
// verdict below can never be flipped by an unused optional folder if that filter ever widens.
|
||||
for _, p := range cs.Paths {
|
||||
if _, err := os.Stat(p.Abs); err != nil {
|
||||
if p.Class == appbackup.ClassMandatory {
|
||||
m.logger.Printf("[WARN] [offbox] %s: mandatory data path missing on disk, skipped from offsite: %s", stack, p.Abs)
|
||||
gaps = append(gaps, p.RelPath)
|
||||
}
|
||||
continue // optional-missing is silent (not a gap) — parity with Tier 2
|
||||
}
|
||||
extra = append(extra, p.Abs)
|
||||
}
|
||||
if len(gaps) > 0 {
|
||||
// R-234 §7.4: this sits beside the whole-app gap message on the same card, and both now drive
|
||||
// the same `incomplete` verdict — so it says what to do, not only what happened.
|
||||
warns = append(warns, fmt.Sprintf("Figyelmeztetés: a(z) %s alkalmazás egyes adatmappái nem kerültek a távoli mentésbe: %s. Ellenőrizd, hogy a mappák megvannak-e a meghajtón; ha igen és ez a következő mentés után is látszik, szólj az üzemeltetőnek.",
|
||||
stack, strings.Join(gaps, ", ")))
|
||||
}
|
||||
return extra, warns, gaps
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-204 item 4 / R-193 — a REBUILT box declares that it needs an off-site credential, instead of
|
||||
// reporting an absence the hub cannot interpret.
|
||||
//
|
||||
// THE POINT OF THESE TESTS is the conjunction. An absent off-site object has FOUR meanings (never
|
||||
// configured / mid-restart / a transient read failure / rebuilt-and-stranded). The declaration has
|
||||
// one, and it is only sound because BOTH halves are required: a fresh data area AND a hub-held
|
||||
// recovery package. Scenario B is the one that matters most — drop the escrow half and every
|
||||
// un-configured box in the fleet starts asking for a credential.
|
||||
|
||||
// bareManager builds a Manager with NO off-site target and NO repository password — the shape of a
|
||||
// freshly rebuilt box before anything is configured.
|
||||
func bareManager(t *testing.T) (*Manager, *settings.Settings) {
|
||||
t.Helper()
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
sett, err := settings.Load(filepath.Join(dataDir, "settings.json"), lg)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
cfg.Paths.SystemDataPath = filepath.Join(dataDir, "sys")
|
||||
return NewManager(cfg, sett, lg), sett
|
||||
}
|
||||
|
||||
// SCENARIO A — a rebuilt box (fresh data area + a hub-held escrow) DECLARES the state.
|
||||
//
|
||||
// RED-PROOF: remove the `GetHubEscrowIdentityPresent()` condition from needsOffsiteCredential —
|
||||
// Scenario A still passes (it has an escrow), and SCENARIO B FAILS, which is the point: the plausible
|
||||
// wrong fix is to declare on freshness alone, and that would make every un-configured box in the
|
||||
// fleet ask for a credential.
|
||||
func TestOffsiteDeclare_RebuiltBoxDeclaresNeedsCredential(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("a rebuilt box reported NO off-site object — the hub cannot distinguish it from a box that never had off-site backups (this is the defect)")
|
||||
}
|
||||
if st.State != OffsiteStateNeedsCredential {
|
||||
t.Fatalf("declared state = %q, want %q", st.State, OffsiteStateNeedsCredential)
|
||||
}
|
||||
// Enabled MUST be false and the sizes zero — that is what makes the declaration inert to the
|
||||
// hub's existing fill and staleness checkers (and to a pre-upgrade hub).
|
||||
if st.Enabled {
|
||||
t.Error("a declaration must not claim the tier is enabled — the hub's staleness check keys on it")
|
||||
}
|
||||
if st.QuotaGB != 0 || st.RepoSizeBytes != 0 || st.SnapshotCount != 0 {
|
||||
t.Errorf("a declaration must carry zero sizes (fill band keys on them): %+v", st)
|
||||
}
|
||||
// And it must be on the off-site object, not a new top-level field.
|
||||
b, err := json.Marshal(st)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(b), `"state":"needs_credential"`) {
|
||||
t.Fatalf("declared state absent from the marshalled off-site object: %s", b)
|
||||
}
|
||||
if !strings.Contains(string(b), `"enabled":false`) {
|
||||
t.Fatalf("marshalled object must carry enabled:false: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO B — a box that never had off-site backups says NOTHING. This is the guard on the
|
||||
// conjunction; without it the feature churns credentials fleet-wide.
|
||||
func TestOffsiteDeclare_NeverHadOffsiteSaysNothing(t *testing.T) {
|
||||
m, _ := bareManager(t) // fresh data area, but NO hub-held escrow
|
||||
|
||||
if st := m.OffboxReportStatus(); st != nil {
|
||||
t.Fatalf("a box that never had off-site backups DECLARED a need: %+v — every un-configured box in the fleet would now ask for a credential", st)
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO D (R-218) — THE DECLARATION STOPS WHEN THE TIER WORKS, NOT WHEN A KEY EXISTS.
|
||||
//
|
||||
// ⚠ THIS TEST ASSERTED THE OPPOSITE until v0.201.0, and it was green the whole time. It required a
|
||||
// box holding a repository password to stay SILENT — which reads as a sound freshness test and is the
|
||||
// exact opposite on the one path that matters, because installing a repository password is the
|
||||
// RECOVERY SCREEN'S WHOLE JOB. Measured live 2026-08-05 (CAMPAIGN-11 Phase 1): 32 seconds after the
|
||||
// hub re-staged the credential, the customer's successful unlock switched off the mechanism that
|
||||
// would have delivered the coordinates for the key they had just recovered. Deadlock, both halves.
|
||||
//
|
||||
// RED-PROOF: restore the `if _, ok := m.OffboxRepoPasswordHash(); ok { return false }` short-circuit
|
||||
// in needsOffsiteCredential and this test FAILS — the box goes silent again with no target, which is
|
||||
// the deadlock. Demonstrated failing before this test was kept.
|
||||
func TestOffsiteDeclare_StillDeclaresAfterARecoveredKeyIsPlaced(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The post-unlock shape: the recovered repository password is on disk, and there is STILL no
|
||||
// off-site target — so the box cannot use what it just recovered.
|
||||
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(m.offboxPwPath(), []byte("a-recovered-repository-password"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("R-218: the box went SILENT after recovering its key while still having no off-site target — the hub's staged credential is never collected and nothing ever asks again")
|
||||
}
|
||||
if st.State != OffsiteStateNeedsCredential {
|
||||
t.Fatalf("declared state = %q, want %q", st.State, OffsiteStateNeedsCredential)
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO E — and once the tier ACTUALLY WORKS the box goes quiet. This is the condition that
|
||||
// replaces the deleted one, and the pair above/below is what makes the deletion safe.
|
||||
func TestOffsiteDeclare_ConfiguredTierIsSilent(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
Schedule: "daily", EscrowState: "escrowed",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("a configured tier must still report its ordinary off-site object")
|
||||
}
|
||||
if st.State == OffsiteStateNeedsCredential {
|
||||
t.Fatal("a box whose tier is configured must not keep asking for a credential")
|
||||
}
|
||||
}
|
||||
|
||||
// A DISABLED target is the customer's own choice, not a rebuild — it must not declare either.
|
||||
func TestOffsiteDeclare_DisabledTargetIsNotStranded(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: false, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if st := m.OffboxReportStatus(); st != nil {
|
||||
t.Fatalf("a deliberately DISABLED target declared a need: %+v", st)
|
||||
}
|
||||
}
|
||||
|
||||
// A CONFIGURED box's report object must be byte-identical to v0.198.0's — no `state` key at all.
|
||||
// This is what lets a pre-upgrade hub and every existing checker read the fleet unchanged.
|
||||
func TestOffsiteDeclare_ConfiguredBoxJSONIsUnchanged(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
Schedule: "daily", EscrowState: "escrowed", LastStatus: "ok",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("a configured box must still report an off-site object")
|
||||
}
|
||||
if st.State != "" {
|
||||
t.Errorf("a configured box must declare NO state, got %q", st.State)
|
||||
}
|
||||
b, err := json.Marshal(st)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Contains(string(b), `"state"`) {
|
||||
t.Fatalf("a healthy report's JSON gained a `state` key — it must stay byte-compatible: %s", b)
|
||||
}
|
||||
if !strings.Contains(string(b), `"enabled":true`) {
|
||||
t.Fatalf("a configured box must report enabled:true: %s", b)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-193 Part 3 — WHAT IS IN THERE. After a successful unlock the customer is shown the contents of the
|
||||
// repository they just opened: which apps, from when, how big.
|
||||
//
|
||||
// READ-ONLY, AND THAT IS THE POINT. This restores nothing, puts nothing back, and compares nothing
|
||||
// against live data. Unlocking and restoring are separate (operator ruling, 2026-08-05): restore is
|
||||
// already per-app and already lives in the backups area, and a screen that unlocks and then offers to
|
||||
// overwrite is two decisions wearing one button.
|
||||
//
|
||||
// WHY A LISTING AT ALL, rather than a success message: "unlocked" with nothing shown is
|
||||
// indistinguishable from having unlocked an EMPTY store, and the customer has no way to tell whether
|
||||
// what came back is the right thing. Seeing their own app names and dates is how they know.
|
||||
|
||||
// errNoOffsiteTarget is returned when the repository cannot even be addressed — no off-site target is
|
||||
// configured on this box yet. Distinguished from a read failure because the remedy differs: this one
|
||||
// resolves by itself once the tier is re-applied.
|
||||
var errNoOffsiteTarget = errors.New("no off-site target is configured on this box yet")
|
||||
|
||||
// ErrNoOffsiteTarget reports whether err is the not-yet-configured case, so a caller can say the right
|
||||
// thing rather than showing a generic failure.
|
||||
func ErrNoOffsiteTarget(err error) bool { return errors.Is(err, errNoOffsiteTarget) }
|
||||
|
||||
// ErrNoOffsiteTargetSentinel exposes the sentinel itself so other packages — and their tests — can
|
||||
// construct the not-yet-configured case. Added for R-237, whose restore list must distinguish
|
||||
// "no target yet" (resolves by itself) from "could not read" (does not), and must be able to pin
|
||||
// both in a table test.
|
||||
func ErrNoOffsiteTargetSentinel() error { return errNoOffsiteTarget }
|
||||
|
||||
// OffsiteInventoryApp is one app's presence in the opened repository. Non-secret throughout.
|
||||
type OffsiteInventoryApp struct {
|
||||
App string // the restic tag == the stack name
|
||||
LatestAt time.Time // the newest snapshot's time for this app
|
||||
SizeBytes int64 // restore size of that newest snapshot (0 = could not be determined)
|
||||
}
|
||||
|
||||
// OffsiteInventory is the whole answer, including the EMPTY case stated explicitly.
|
||||
type OffsiteInventory struct {
|
||||
Apps []OffsiteInventoryApp
|
||||
// Empty is true when the repository opened cleanly and holds no snapshots. It is a real and
|
||||
// confusing outcome — a bare list there reads as a broken page — so it is named rather than
|
||||
// inferred from len(Apps)==0, which is also what a failed read looks like.
|
||||
Empty bool
|
||||
}
|
||||
|
||||
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
|
||||
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
|
||||
//
|
||||
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
|
||||
// app is in there matters more than knowing how big it is, and dropping it would under-report the
|
||||
// customer's own data.
|
||||
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
|
||||
var inv OffsiteInventory
|
||||
// A box can hold a recovered key and still have no off-site COORDINATES — the pristine rebuilt
|
||||
// shape, before its target is re-applied. Reading the repository is impossible then, and saying so
|
||||
// is the honest answer; without this guard offboxBaseArgs nil-derefs on the missing target.
|
||||
if !m.OffboxConfigured() {
|
||||
return inv, errNoOffsiteTarget
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||||
defer cancel()
|
||||
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
||||
if err != nil {
|
||||
return inv, err
|
||||
}
|
||||
var snaps []struct {
|
||||
ShortID string `json:"short_id"`
|
||||
ID string `json:"id"`
|
||||
Time time.Time `json:"time"`
|
||||
Tags []string `json:"tags"`
|
||||
}
|
||||
if uerr := json.Unmarshal(out, &snaps); uerr != nil {
|
||||
return inv, uerr
|
||||
}
|
||||
if len(snaps) == 0 {
|
||||
inv.Empty = true
|
||||
return inv, nil
|
||||
}
|
||||
// Newest snapshot per tag. A snapshot may carry several tags; each names an app it belongs to.
|
||||
newest := map[string]struct {
|
||||
id string
|
||||
at time.Time
|
||||
}{}
|
||||
for _, s := range snaps {
|
||||
id := s.ShortID
|
||||
if id == "" {
|
||||
id = s.ID
|
||||
}
|
||||
for _, tag := range s.Tags {
|
||||
if tag == "" {
|
||||
continue
|
||||
}
|
||||
if cur, ok := newest[tag]; !ok || s.Time.After(cur.at) {
|
||||
newest[tag] = struct {
|
||||
id string
|
||||
at time.Time
|
||||
}{id: id, at: s.Time}
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(newest) == 0 {
|
||||
// Snapshots exist but carry no tags — not "empty", and saying so would be a lie. Report an
|
||||
// empty app list without the Empty flag; the page renders the honest in-between wording.
|
||||
return inv, nil
|
||||
}
|
||||
for tag, n := range newest {
|
||||
app := OffsiteInventoryApp{App: tag, LatestAt: n.at}
|
||||
if size, serr := m.offboxSnapshotSize(ctx, n.id); serr == nil {
|
||||
app.SizeBytes = size
|
||||
} else {
|
||||
m.logger.Printf("[WARN] [offbox] inventory: size of %s's newest snapshot unknown: %v (listing it anyway)", tag, serr)
|
||||
}
|
||||
inv.Apps = append(inv.Apps, app)
|
||||
}
|
||||
sort.Slice(inv.Apps, func(i, j int) bool { return inv.Apps[i].App < inv.Apps[j].App })
|
||||
return inv, nil
|
||||
}
|
||||
|
||||
// HumanizeBytes exposes the shared byte formatter to the web layer so the recovery page renders sizes
|
||||
// the same way every other surface does.
|
||||
func HumanizeBytes(n int64) string { return humanizeBytes(n) }
|
||||
@@ -0,0 +1,132 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
func newTestSettings(t *testing.T) *settings.Settings {
|
||||
t.Helper()
|
||||
sett, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatalf("settings.Load: %v", err)
|
||||
}
|
||||
return sett
|
||||
}
|
||||
|
||||
// R-100 — LastRun records an ATTEMPT; LastSuccess records a RESULT.
|
||||
//
|
||||
// The defect these pin: `LastRun` is written unconditionally at the end of every offsite run, failures
|
||||
// included, so the hub's staleness verdict ("how long since LastRun?") was really asking "how long
|
||||
// since we last TRIED?" — and a tier failing on every single run read as perfectly fresh forever.
|
||||
//
|
||||
// These are the CONTROLLER half (does the anchor move only on success, and does it survive the writes
|
||||
// that rebuild the target?). The hub half — does the verdict count from it — lives in the hub's
|
||||
// offsite tests.
|
||||
|
||||
// The invariant named by the comment at the write site, per the standing rule that an asserted
|
||||
// invariant needs a test pinning it. This calls the PRODUCTION rule — an earlier version of this test
|
||||
// re-implemented it in a local closure and was hollow: mutating offbox.go left it green.
|
||||
//
|
||||
// RED-PROOF: make offboxAnchorAfterRun return `at` unconditionally (drop the runErr guard) → this
|
||||
// fails with "a FAILED run advanced LastSuccess — that is the R-100 defect in mirror image".
|
||||
func TestOffboxAnchorAfterRun_FailureNeitherAdvancesNorClears(t *testing.T) {
|
||||
const monday = "2026-07-20T02:15:00Z"
|
||||
boom := errors.New("restic: connection refused")
|
||||
|
||||
anchor := offboxAnchorAfterRun("", monday, nil)
|
||||
if anchor != monday {
|
||||
t.Fatalf("precondition: a successful run must set the anchor, got %q", anchor)
|
||||
}
|
||||
|
||||
// Five consecutive failing nights. The attempt clock moves; the anchor must not.
|
||||
for _, night := range []string{
|
||||
"2026-07-21T02:15:00Z", "2026-07-22T02:15:00Z", "2026-07-23T02:15:00Z",
|
||||
"2026-07-24T02:15:00Z", "2026-07-25T02:15:00Z",
|
||||
} {
|
||||
anchor = offboxAnchorAfterRun(anchor, night, boom)
|
||||
if anchor == night {
|
||||
t.Fatalf("a FAILED run advanced LastSuccess to %q — that is the R-100 defect in mirror image", anchor)
|
||||
}
|
||||
if anchor != monday {
|
||||
t.Fatalf("a FAILED run CLEARED or moved the anchor (got %q, want %q) — one bad night must not make an established tier read as never-succeeded", anchor, monday)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Recovery: a later success moves it forward, or a tier would stay permanently stale after one good
|
||||
// night.
|
||||
//
|
||||
// RED-PROOF: make offboxAnchorAfterRun return `prev` unconditionally → this fails with
|
||||
// "a successful run did not advance the anchor".
|
||||
func TestOffboxAnchorAfterRun_SuccessAdvances(t *testing.T) {
|
||||
got := offboxAnchorAfterRun("2026-07-20T02:15:00Z", "2026-07-26T02:15:00Z", nil)
|
||||
if got != "2026-07-26T02:15:00Z" {
|
||||
t.Errorf("a successful run did not advance the anchor: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A never-run tier stays empty on failure — it must not acquire a fabricated anchor, because "" is the
|
||||
// signal the hub's newborn-box path keys on.
|
||||
func TestOffboxAnchorAfterRun_NeverRanStaysEmptyOnFailure(t *testing.T) {
|
||||
if got := offboxAnchorAfterRun("", "2026-07-21T02:15:00Z", errors.New("boom")); got != "" {
|
||||
t.Errorf("a failed first run fabricated an anchor (%q) — the newborn-box path keys on empty", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The wire carries it. A field the hub cannot see is a field that does not exist — the "seam built but
|
||||
// never wired" class this project has hit four times.
|
||||
//
|
||||
// RED-PROOF: drop `LastSuccess: t.LastSuccess` from OffboxReportStatus() → this fails with
|
||||
// "OffboxReportStatus dropped LastSuccess — the hub would degrade forever on a controller that has it".
|
||||
func TestOffboxReportStatus_CarriesLastSuccess(t *testing.T) {
|
||||
m := &Manager{settings: newTestSettings(t)}
|
||||
if err := m.settings.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true,
|
||||
Host: "nas.example",
|
||||
User: "u1",
|
||||
RepoPath: "/vol/repo",
|
||||
EscrowState: "escrowed",
|
||||
LastRun: "2026-07-26T02:15:00Z",
|
||||
LastStatus: "ok",
|
||||
LastSuccess: "2026-07-26T02:15:00Z",
|
||||
}); err != nil {
|
||||
t.Fatalf("seed: %v", err)
|
||||
}
|
||||
got := m.OffboxReportStatus()
|
||||
if got == nil {
|
||||
t.Fatal("OffboxReportStatus returned nil for an enabled target")
|
||||
}
|
||||
if got.LastSuccess != "2026-07-26T02:15:00Z" {
|
||||
t.Errorf("OffboxReportStatus dropped LastSuccess — the hub would degrade forever on a controller that has it (got %q)", got.LastSuccess)
|
||||
}
|
||||
}
|
||||
|
||||
// A re-apply from the hub is not a new tier. Dropping the anchor here would reset an established tier
|
||||
// to "never succeeded" every time the hub re-pushes its descriptor.
|
||||
//
|
||||
// RED-PROOF: remove `tgt.LastSuccess = cur.LastSuccess` from ApplyOffsiteTarget's carry-over block →
|
||||
// this fails with "a hub re-apply erased the staleness anchor".
|
||||
func TestApplyOffsiteTarget_PreservesLastSuccess(t *testing.T) {
|
||||
m := &Manager{settings: newTestSettings(t)}
|
||||
if err := m.settings.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.example", User: "u1", RepoPath: "/vol/repo",
|
||||
EscrowState: "escrowed", LastSuccess: "2026-07-26T02:15:00Z", LastRun: "2026-07-27T02:15:00Z",
|
||||
}); err != nil {
|
||||
t.Fatalf("seed: %v", err)
|
||||
}
|
||||
cur := m.settings.GetOffboxTarget()
|
||||
// Mirror ApplyOffsiteTarget's carry-over onto a freshly-built target.
|
||||
tgt := &settings.OffboxTarget{Enabled: true, Host: "nas.example", User: "u1", RepoPath: "/vol/repo", Schedule: "daily"}
|
||||
tgt.EscrowState = cur.EscrowState
|
||||
tgt.LastRun, tgt.LastStatus, tgt.LastError = cur.LastRun, cur.LastStatus, cur.LastError
|
||||
tgt.LastSuccess = cur.LastSuccess
|
||||
if tgt.LastSuccess != "2026-07-26T02:15:00Z" {
|
||||
t.Errorf("a hub re-apply erased the staleness anchor (got %q)", tgt.LastSuccess)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-241 — THE MINT GUARD. This file is the session's headline test.
|
||||
//
|
||||
// The defect, measured on the final walk (SPIKE-r241-recovery-offer-2026-08-07): a rebuilt box's
|
||||
// credential self-heal reached WriteOffboxSecrets at 03:18:06Z and minted a fresh repository password
|
||||
// over a hub package sealing a DIFFERENT key. The recovery screen then correctly reported that there
|
||||
// was nothing recoverable under the key the box held. The screen was honest; the minting was not.
|
||||
//
|
||||
// Scenario A asserts the key is NOT written. Scenario B asserts the guard is narrow enough that a
|
||||
// first-time box still starts — the guard's own failure mode, and the one an over-broad fix produces.
|
||||
|
||||
// mintGuardManager builds a Manager with NO offbox secrets written, so the mint branch is live.
|
||||
// hubHoldsPackage sets the ACK-cached fact the guard consults.
|
||||
func mintGuardManager(t *testing.T, hubHoldsPackage bool) (*Manager, *settings.Settings, string) {
|
||||
t.Helper()
|
||||
logger := log.New(os.Stderr, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
sett, err := settings.Load(filepath.Join(dataDir, "settings.json"), logger)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
cfg.Paths.SystemDataPath = filepath.Join(dataDir, "sys")
|
||||
m := NewManager(cfg, sett, logger)
|
||||
if err := sett.SetHubEscrowIdentityPresent(hubHoldsPackage); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return m, sett, filepath.Join(dataDir, "offbox", "repo_password")
|
||||
}
|
||||
|
||||
// ── SCENARIO A — the box does not mint over a sealed package ────────────────────────────────────
|
||||
//
|
||||
// RED-PROOF: delete the `if m.sealedPackageHeld()` block in WriteOffboxSecrets. The password file
|
||||
// then exists and this test fails on the first assertion — which is exactly the 03:18:06Z event.
|
||||
func TestR241_ScenarioA_NoMintWhenHubHoldsSealedPackage(t *testing.T) {
|
||||
m, _, pwPath := mintGuardManager(t, true)
|
||||
|
||||
err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey")
|
||||
|
||||
if !IsOffboxSealedPackageHeld(err) {
|
||||
t.Fatalf("want the sealed-package refusal sentinel, got %v", err)
|
||||
}
|
||||
// THE ASSERTION THAT IS THE WHOLE SESSION: no key on disk.
|
||||
if _, serr := os.Stat(pwPath); !os.IsNotExist(serr) {
|
||||
t.Fatalf("R-241 REGRESSION: a repository password was minted over the hub's sealed package (stat err=%v)", serr)
|
||||
}
|
||||
// The transport IS still written — the refusal is a holding state, not a failure. Without this the
|
||||
// recovery screen could not bring the tier up when the key arrives (R-219).
|
||||
for _, f := range []string{"ssh_key", "known_hosts"} {
|
||||
if _, serr := os.Stat(filepath.Join(filepath.Dir(pwPath), f)); serr != nil {
|
||||
t.Errorf("transport file %s should still be written on the refusal path: %v", f, serr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario A at the APPLY level — the path the self-heal actually takes. ApplyOffsiteTarget must
|
||||
// swallow the sentinel, record the target, and NOT stage an escrow.
|
||||
func TestR241_ScenarioA_ApplyOffsiteTargetHoldsInsteadOfMinting(t *testing.T) {
|
||||
m, sett, pwPath := mintGuardManager(t, true)
|
||||
|
||||
staged := 0
|
||||
stage := func(ctx context.Context, pw string) error { staged++; return nil }
|
||||
|
||||
tgt := &settings.OffboxTarget{Enabled: true, Host: "box.example", Port: 23, User: "u1", RepoPath: "/home/felhom-repo"}
|
||||
if err := m.ApplyOffsiteTarget(context.Background(), tgt, "KEYMATERIAL", "box.example ssh-ed25519 HOSTKEY", stage); err != nil {
|
||||
t.Fatalf("apply should SUCCEED into the holding state, not fail: %v", err)
|
||||
}
|
||||
if _, serr := os.Stat(pwPath); !os.IsNotExist(serr) {
|
||||
t.Fatalf("R-241 REGRESSION: apply minted a repository password over the sealed package")
|
||||
}
|
||||
if staged != 0 {
|
||||
t.Errorf("nothing may be staged for escrow — there is no key to escrow; staged=%d", staged)
|
||||
}
|
||||
// The target is recorded, so the box stops declaring needs_credential and the hub stops re-staging.
|
||||
if got := sett.GetOffboxTarget(); got == nil {
|
||||
t.Fatal("the transport target must be recorded, or the hub re-stages a consumed credential forever")
|
||||
}
|
||||
// Runs stay gated: no password file ⇒ not configured.
|
||||
if m.OffboxConfigured() {
|
||||
t.Error("OffboxConfigured must be false while the key is awaited — runs must not proceed")
|
||||
}
|
||||
// And the box says so, in the state the hub reads.
|
||||
if !m.OffboxAwaitingRecoveryKey() {
|
||||
t.Error("OffboxAwaitingRecoveryKey should be true in the holding state")
|
||||
}
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil || st.State != OffsiteStateAwaitingRecoveryKey {
|
||||
t.Fatalf("want declared state %q, got %+v", OffsiteStateAwaitingRecoveryKey, st)
|
||||
}
|
||||
if st.Enabled {
|
||||
t.Error("the declared holding object must carry Enabled=false so existing hub readers stay inert")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO B — a box the hub holds nothing for still mints, exactly as today ───────────────────
|
||||
//
|
||||
// RED-PROOF: widen the guard to `if true` (or drop the GetHubEscrowIdentityPresent() conjunct in
|
||||
// sealedPackageHeld). A first-time box then cannot start, and this test fails — the failure mode an
|
||||
// over-broad fix produces, which is why the guard is written as a conjunction.
|
||||
func TestR241_ScenarioB_FirstTimeBoxStillMints(t *testing.T) {
|
||||
m, _, pwPath := mintGuardManager(t, false) // the hub holds nothing for us
|
||||
|
||||
if err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey"); err != nil {
|
||||
t.Fatalf("a first-time box must mint exactly as before, got %v", err)
|
||||
}
|
||||
pw, rerr := os.ReadFile(pwPath)
|
||||
if rerr != nil {
|
||||
t.Fatalf("a first-time box must get a repository password: %v", rerr)
|
||||
}
|
||||
if !offboxRepoPwPattern.Match(pw) {
|
||||
t.Errorf("minted password is not the expected 64-hex shape")
|
||||
}
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
t.Error("a box with no sealed package is not awaiting anything")
|
||||
}
|
||||
}
|
||||
|
||||
// The guard must not fire once a key EXISTS — a healthy box re-applying its target (a quota bump,
|
||||
// a hub re-push) must be untouched, package or no package. This is the idempotency half.
|
||||
func TestR241_ExistingKeyIsNeverDisturbed(t *testing.T) {
|
||||
m, _, pwPath := mintGuardManager(t, false)
|
||||
if err := m.WriteOffboxSecrets("K", "kh"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before, err := os.ReadFile(pwPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Now the hub starts holding a package (the ceremony ran) and the target is re-applied.
|
||||
if err := m.settings.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.WriteOffboxSecrets("K2", "kh2"); err != nil {
|
||||
t.Fatalf("a re-apply on a box that already has a key must not be refused: %v", err)
|
||||
}
|
||||
after, err := os.ReadFile(pwPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if string(before) != string(after) {
|
||||
t.Error("the existing repository password must never be rotated by an apply")
|
||||
}
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
t.Error("a box holding its key is not awaiting one")
|
||||
}
|
||||
}
|
||||
|
||||
// Fail-safe: an unreadable settings store must not block a tier. A transient read failure turning
|
||||
// into a permanently-held tier is a worse defect than the one being fixed.
|
||||
func TestR241_NilSettingsDoesNotBlockTheMint(t *testing.T) {
|
||||
logger := log.New(os.Stderr, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
m := NewManager(cfg, nil, logger)
|
||||
if m.sealedPackageHeld() {
|
||||
t.Fatal("a nil settings store must read as 'no package held' — fail toward letting the box work")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario E's carve-out, pinned for the HOLDING state too. A customer who switched off-site off is
|
||||
// not awaiting a recovery key, and must not declare one. The first draft of
|
||||
// OffboxAwaitingRecoveryKey omitted `t.Enabled` and TestOffsiteDeclare_DisabledTargetIsNotStranded
|
||||
// caught it; this test pins the same invariant from the new predicate's own side, so a future edit
|
||||
// to THIS function fails here rather than in a neighbouring file.
|
||||
func TestR241_DisabledTargetIsNotAwaitingAnything(t *testing.T) {
|
||||
m, sett, _ := mintGuardManager(t, true) // the hub holds a package, and there is no key
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: false, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
t.Fatal("a deliberately DISABLED target must never declare the holding state (Scenario E)")
|
||||
}
|
||||
if st := m.OffboxReportStatus(); st != nil {
|
||||
t.Fatalf("a disabled target must stay silent in the report, got %+v", st)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-241 shape (c) — the recovery offer is driven by the comparison the box already makes.
|
||||
//
|
||||
// Scenarios C and D from the task, plus §7.2's two staleness cases. The point of shape (c) is that
|
||||
// it asks the real question — *does the hub hold a package for a key other than the one I am
|
||||
// using?* — rather than the two proxies that have each now been wrong in opposite directions.
|
||||
|
||||
// offerFixture builds a manager holding a repository password, with the hub's cached facts settable.
|
||||
// Returns the local key's hash so a test can make the hub's hash match or differ deliberately.
|
||||
func offerFixture(t *testing.T, hubHoldsPackage bool) (*Manager, *settings.Settings, string) {
|
||||
t.Helper()
|
||||
m, sett, pwPath := mintGuardManager(t, false) // mint freely first
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo", Schedule: "daily",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.WriteOffboxSecrets("KEYMATERIAL", "nas.local ssh-ed25519 HOSTKEY"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := os.Stat(pwPath); err != nil {
|
||||
t.Fatalf("fixture should hold a repository password: %v", err)
|
||||
}
|
||||
local, ok := m.OffboxRepoPasswordHash()
|
||||
if !ok {
|
||||
t.Fatal("fixture should be able to hash its own key")
|
||||
}
|
||||
if err := sett.SetHubEscrowIdentityPresent(hubHoldsPackage); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return m, sett, local
|
||||
}
|
||||
|
||||
const otherKeyHash = "9b4a9a9dcec7898e7544f35b18470aac77c3d9064e5d3a302897617fa62edd65"
|
||||
|
||||
// ── SCENARIO C — a differing key offers recovery, whatever the reason for the difference ────────
|
||||
//
|
||||
// This is the venue's exact state on 2026-08-07: a key present, no orphan recorded, escrow stuck
|
||||
// pending — and before shape (c), silence.
|
||||
func TestR241_ScenarioC_DifferingKeyOffersRecovery(t *testing.T) {
|
||||
m, sett, local := offerFixture(t, true)
|
||||
if local == otherKeyHash {
|
||||
t.Fatal("fixture precondition: the local key must differ from the hub's")
|
||||
}
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2026-08-07T03:28:03Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Neither proxy fires: a key EXISTS (so not shape (a)) and nothing is orphaned (so not shape (b)).
|
||||
if _, ok := m.OffboxRepoPasswordHash(); !ok {
|
||||
t.Fatal("precondition: shape (a) must be false")
|
||||
}
|
||||
if m.OffboxOrphaned() {
|
||||
t.Fatal("precondition: shape (b) must be false")
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("R-241: the hub holds a package for a DIFFERENT key and the screen was not offered — this is the defect")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO D — a healthy box is never offered recovery ────────────────────────────────────────
|
||||
//
|
||||
// RED-PROOF: drop the `hubHash != localHash` conjunct in shape (c) (make it `hubHash != ""`). A
|
||||
// healthy box is then offered recovery forever, and this test fails — which is how a screen stops
|
||||
// being read.
|
||||
func TestR241_ScenarioD_MatchingKeyOffersNothing(t *testing.T) {
|
||||
m, sett, local := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256(local, "2026-08-07T09:00:00Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a box whose key the hub's package covers must never be offered recovery")
|
||||
}
|
||||
}
|
||||
|
||||
// A box the hub holds nothing for is never offered, even if a stale hash lingers in settings. Fact 1
|
||||
// stays required — the spike's comment block calls dropping it "the plausible wrong fix".
|
||||
func TestR241_ShapeC_NeverHadOffsiteIsStillSilent(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, false) // the hub holds NOTHING
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2026-08-07T09:00:00Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a box that never had off-site backups must never be greeted by a recovery screen")
|
||||
}
|
||||
}
|
||||
|
||||
// ── §7.2 — the staleness decision, both halves ──────────────────────────────────────────────────
|
||||
|
||||
// A KNOWN DIFFERENCE OFFERS, however old the reading. Age is deliberately not gated on: gating would
|
||||
// make a box offline from the hub silently stop offering, which is the failure this session exists
|
||||
// to remove.
|
||||
func TestR241_StaleComparison_KnownDifferenceStillOffers(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2020-01-01T00:00:00Z"); err != nil { // ancient
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a known difference must offer regardless of how old the reading is (§7.2)")
|
||||
}
|
||||
}
|
||||
|
||||
// AN ABSENT HASH FALLS BACK TO (a)/(b) — it does not offer. "" is the hub positively saying its
|
||||
// package seals no repository password (legacy hash-less escrow); there is nothing to compare, and
|
||||
// offering would put a permanent screen in front of every legacy box.
|
||||
func TestR241_StaleComparison_AbsentHashFallsBackAndDoesNotOffer(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256("", ""); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a hash never learned must fall back to (a)/(b), not offer (§7.2)")
|
||||
}
|
||||
// ...and the fallback still works: mark the repo orphaned and shape (b) fires as before.
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoState = "orphaned" }); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("shape (b) must still work when the hub's hash was never learned")
|
||||
}
|
||||
}
|
||||
|
||||
// Shape (a) is untouched: a box with no key at all is still offered, which is the pristine rebuild.
|
||||
func TestR241_ShapeAStillWorks(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := os.Remove(filepath.Join(m.cfg.Paths.DataDir, "offbox", "repo_password")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2026-08-07T09:00:00Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("shape (a) — no repository password at all — must still offer")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// classifyResticProbe maps the exact restic stderr to a repo class (the 2026-07-17 diagnosis
|
||||
// signatures). ORPHANED only on the definitive wrong-password line; ambiguous errors are NOT orphaned.
|
||||
func TestClassifyResticProbe(t *testing.T) {
|
||||
cases := []struct {
|
||||
out string
|
||||
err error
|
||||
want string
|
||||
}{
|
||||
{"", nil, ""}, // success
|
||||
{"Fatal: wrong password or no key found", fmt.Errorf("exit status 1"), "orphaned"},
|
||||
{"Fatal: unable to open config file: <sftp:...> does not exist\nIs there a repository at the following location?", fmt.Errorf("exit status 1"), "norepo"},
|
||||
{"ssh: connect to host nas.local port 22: Connection timed out", fmt.Errorf("exit status 255"), "other"},
|
||||
{"Load(<lock/...>): permission denied", fmt.Errorf("exit status 1"), "other"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := classifyResticProbe([]byte(c.out), c.err); got != c.want {
|
||||
t.Errorf("classify(%q) = %q, want %q", c.out, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// wrongPwRunner: `cat config` returns the wrong-password signature; other restic steps succeed (so a
|
||||
// post-reset run can proceed). Records the subcommands seen.
|
||||
func wrongPwRunner(seen *[]string) offboxRunner {
|
||||
return func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
sub := ""
|
||||
for i, a := range args {
|
||||
if a == "cat" && i+1 < len(args) && args[i+1] == "config" {
|
||||
sub = "cat-config"
|
||||
} else if a == "init" {
|
||||
sub = "init"
|
||||
}
|
||||
}
|
||||
if sub == "" && len(args) > 0 {
|
||||
sub = args[len(args)-1]
|
||||
}
|
||||
if seen != nil {
|
||||
*seen = append(*seen, sub)
|
||||
}
|
||||
if sub == "cat-config" {
|
||||
return []byte("Fatal: wrong password or no key found"), fmt.Errorf("exit status 1")
|
||||
}
|
||||
return nil, nil // init / unlock / backup / stats succeed
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario A (RED-PROOF = the incident): a CLAIMED box whose repo is wrong-keyed enters the explicit
|
||||
// ORPHANED state — the run skips cleanly (no raw restic banner, ONE event, no nightly re-fire) instead
|
||||
// of erroring nightly with "exit status 1". Pre-fix (no classification) surfaced the raw error and set
|
||||
// no state → these assertions FAIL.
|
||||
func TestOffbox_OrphanDetection_Claimed(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
if err := sett.SetClaimed(); err != nil { // claimed → orphan card, NEVER auto-reset
|
||||
t.Fatal(err)
|
||||
}
|
||||
var events []string
|
||||
m.SetOffboxOrphanEvent(func(evt, _ string) { events = append(events, evt) })
|
||||
m.SetOffboxRunner(wrongPwRunner(nil))
|
||||
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run should skip cleanly on an orphaned repo, got %v", err)
|
||||
}
|
||||
if !m.OffboxOrphaned() {
|
||||
t.Fatal("repo was not classified/persisted as ORPHANED")
|
||||
}
|
||||
if len(events) != 1 || events[0] != "offbox_repo_orphaned" {
|
||||
t.Fatalf("expected exactly one offbox_repo_orphaned event, got %v", events)
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.RepoState != "orphaned" || got.OrphanedAt == "" {
|
||||
t.Fatalf("RepoState=%q OrphanedAt=%q, want orphaned + a stamp", got.RepoState, got.OrphanedAt)
|
||||
}
|
||||
// The raw restic error must NOT be surfaced as the last-error banner (the card explains instead).
|
||||
if strings.Contains(got.LastError, "wrong password") || strings.Contains(got.LastError, "exit status") {
|
||||
t.Fatalf("raw restic error leaked into LastError: %q", got.LastError)
|
||||
}
|
||||
// A second scheduled run SKIPS (no nightly spam) — no new event.
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("second run: %v", err)
|
||||
}
|
||||
if len(events) != 1 {
|
||||
t.Fatalf("nightly re-fire — events=%v, want the single transition event only", events)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B: an UNCLAIMED box auto-resets on detection — move-aside (never delete) + re-init; both
|
||||
// events fire and the box ends un-orphaned (next run green).
|
||||
func TestOffbox_OrphanDetection_UnclaimedAutoReset(t *testing.T) {
|
||||
m, sett := newOffboxManager(t) // unclaimed by default
|
||||
var events []string
|
||||
m.SetOffboxOrphanEvent(func(evt, _ string) { events = append(events, evt) })
|
||||
var sshCmds []string
|
||||
m.SetOffboxSSH(func(_ context.Context, _, _ string, _ int, _, _, remoteCmd string) ([]byte, error) {
|
||||
sshCmds = append(sshCmds, remoteCmd)
|
||||
if strings.HasPrefix(remoteCmd, "test -e") {
|
||||
return nil, fmt.Errorf("exit status 1") // absent → free name
|
||||
}
|
||||
return nil, nil // mv OK
|
||||
})
|
||||
m.SetOffboxRunner(wrongPwRunner(nil))
|
||||
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("unclaimed run should auto-reset + succeed, got %v", err)
|
||||
}
|
||||
if m.OffboxOrphaned() {
|
||||
t.Fatal("unclaimed box stayed orphaned — auto-reset did not clear the state")
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.OrphanedRenamedTo == "" || !strings.Contains(got.OrphanedRenamedTo, ".orphaned-") {
|
||||
t.Fatalf("move-aside path not recorded: %q", got.OrphanedRenamedTo)
|
||||
}
|
||||
var mvSeen bool
|
||||
for _, c := range sshCmds {
|
||||
if strings.HasPrefix(c, "mv ") {
|
||||
mvSeen = true
|
||||
}
|
||||
}
|
||||
if !mvSeen {
|
||||
t.Fatalf("no move-aside mv issued: %v", sshCmds)
|
||||
}
|
||||
// Both transition events fired (orphaned → reset). No delete anywhere.
|
||||
if len(events) != 2 || events[0] != "offbox_repo_orphaned" || events[1] != "offbox_repo_reset" {
|
||||
t.Fatalf("events = %v, want [orphaned reset]", events)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C: the claimed confirmed reset (ResetOrphanedRepo) refuses unless orphaned, then move-aside +
|
||||
// re-init + clear state.
|
||||
func TestOffbox_ConfirmedReset(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
if err := sett.SetClaimed(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// refuse when not orphaned
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err == nil {
|
||||
t.Fatal("reset must refuse when the repo is not orphaned")
|
||||
}
|
||||
// mark orphaned, then confirm reset
|
||||
m.SetOffboxRunner(wrongPwRunner(nil))
|
||||
_ = m.RunOffboxBackup(context.Background())
|
||||
if !m.OffboxOrphaned() {
|
||||
t.Fatal("precondition: not orphaned")
|
||||
}
|
||||
var mv bool
|
||||
m.SetOffboxSSH(func(_ context.Context, _, _ string, _ int, _, _, cmd string) ([]byte, error) {
|
||||
if strings.HasPrefix(cmd, "test -e") {
|
||||
return nil, fmt.Errorf("exit 1")
|
||||
}
|
||||
if strings.HasPrefix(cmd, "mv ") {
|
||||
mv = true
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatalf("confirmed reset: %v", err)
|
||||
}
|
||||
if !mv {
|
||||
t.Fatal("confirmed reset did not move the old repo aside")
|
||||
}
|
||||
if m.OffboxOrphaned() {
|
||||
t.Fatal("state not cleared after confirmed reset")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// placeFixture builds a manager + provider with a scratch dir for stack on drive, a snapshot whose
|
||||
// paths anchor on `oldNs`, and (per `full`) the reconstructed scratch srcs on disk. Returns the copier
|
||||
// invocation counter pointer and the scratch dir. Free/size seams default to "plenty of room".
|
||||
func placeFixture(t *testing.T, full bool) (*Manager, *offbox3aProvider, string, *int) {
|
||||
t.Helper()
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["immich"] = drive
|
||||
|
||||
scratch, liveNs, err := m.offboxRestoreScratchDir("immich")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.MkdirAll(scratch, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Snapshot paths anchored on a synthetic POSIX namespace (drive-churn realistic; also avoids the
|
||||
// Windows volume-letter that filepath.Join can't nest — prod paths are Linux, no volume).
|
||||
oldNs := "/felhomdata/ns"
|
||||
unitP := oldNs + "/backups/primary/immich"
|
||||
dataP := oldNs + "/appdata/immich"
|
||||
snapPaths := []string{unitP, dataP}
|
||||
|
||||
// Create the reconstructed scratch srcs the code will stat — computed via the pure mapper so the
|
||||
// fixture matches the code's own path arithmetic (no hand-predicting OS separators).
|
||||
placements, err := mapOffsiteRestorePaths(snapPaths, "immich", scratch, liveNs)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, pl := range placements {
|
||||
if !full && !pl.isUnit {
|
||||
continue // unit-only scratch: userdata src deliberately absent (Scenario C)
|
||||
}
|
||||
if err := os.MkdirAll(pl.src, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
m.SetOffboxFreeFn(func(string) int64 { return 100 << 30 })
|
||||
m.SetOffboxSizer(func(string) int64 { return 1 << 20 })
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[{"short_id":"a","time":"2026-07-15T00:00:00Z","paths":["` + unitP + `","` + dataP + `"]}]`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
var copies int
|
||||
m.SetOffboxPlaceCopier(func(_, _ string) (int, error) { copies++; return 1, nil })
|
||||
return m, prov, scratch, &copies
|
||||
}
|
||||
|
||||
// A (F-3a-1a): undeployed placement refused with ZERO copies (never merges onto the SSD namespace).
|
||||
func TestPlace_UndeployedRefused(t *testing.T) {
|
||||
m, prov, scratch, copies := placeFixture(t, true)
|
||||
prov.hdd["immich"] = "" // undeployed
|
||||
err := m.PlaceOffsiteRestore(context.Background(), "immich")
|
||||
if err == nil || !strings.Contains(err.Error(), "nincs telepítve") {
|
||||
t.Fatalf("undeployed must refuse with 'nincs telepítve', got %v", err)
|
||||
}
|
||||
if *copies != 0 {
|
||||
t.Errorf("copier must NOT run for an undeployed app, got %d", *copies)
|
||||
}
|
||||
if _, sErr := os.Stat(scratch); sErr != nil {
|
||||
t.Error("scratch must be untouched on refusal")
|
||||
}
|
||||
}
|
||||
|
||||
// B (F-3a-1b): placement headroom gate refuses BEFORE any copy.
|
||||
func TestPlace_HeadroomRefused(t *testing.T) {
|
||||
m, _, _, copies := placeFixture(t, true)
|
||||
m.SetOffboxFreeFn(func(string) int64 { return 1 }) // 1 byte free
|
||||
m.SetOffboxSizer(func(string) int64 { return 1 << 30 })
|
||||
err := m.PlaceOffsiteRestore(context.Background(), "immich")
|
||||
if err == nil || !strings.Contains(err.Error(), "Nincs elég szabad hely") {
|
||||
t.Fatalf("headroom gate must refuse, got %v", err)
|
||||
}
|
||||
if *copies != 0 {
|
||||
t.Errorf("copier must NOT run when headroom fails, got %d", *copies)
|
||||
}
|
||||
}
|
||||
|
||||
// C (F-3a-4): a unit-only scratch (userdata src absent) refuses with ZERO copies (stat pre-pass).
|
||||
func TestPlace_IncompleteScratchRefusedNoCopies(t *testing.T) {
|
||||
m, _, _, copies := placeFixture(t, false) // full=false → userdata src missing
|
||||
err := m.PlaceOffsiteRestore(context.Background(), "immich")
|
||||
if err == nil || !strings.Contains(err.Error(), "hiányos") {
|
||||
t.Fatalf("incomplete scratch must refuse with 'hiányos', got %v", err)
|
||||
}
|
||||
if *copies != 0 {
|
||||
t.Errorf("stat pre-pass must refuse BEFORE any copy, got %d copies", *copies)
|
||||
}
|
||||
}
|
||||
|
||||
// E (F-3a-2): success removes the scratch (ready-gate flips false); failure keeps it.
|
||||
func TestPlace_ScratchLifecycle(t *testing.T) {
|
||||
// success
|
||||
m, _, scratch, copies := placeFixture(t, true)
|
||||
if err := m.PlaceOffsiteRestore(context.Background(), "immich"); err != nil {
|
||||
t.Fatalf("placement: %v", err)
|
||||
}
|
||||
if *copies == 0 {
|
||||
t.Error("expected at least one copy on success")
|
||||
}
|
||||
if _, sErr := os.Stat(scratch); !os.IsNotExist(sErr) {
|
||||
t.Errorf("scratch must be removed after success, stat err=%v", sErr)
|
||||
}
|
||||
if m.OffboxFullScratchReady("immich") {
|
||||
t.Error("OffboxFullScratchReady must be false after cleanup")
|
||||
}
|
||||
|
||||
// failure keeps the scratch
|
||||
m2, _, scratch2, _ := placeFixture(t, true)
|
||||
m2.SetOffboxPlaceCopier(func(_, _ string) (int, error) { return 0, os.ErrPermission })
|
||||
if err := m2.PlaceOffsiteRestore(context.Background(), "immich"); err == nil {
|
||||
t.Fatal("a copier failure must surface as an error")
|
||||
}
|
||||
if _, sErr := os.Stat(scratch2); sErr != nil {
|
||||
t.Errorf("scratch must be KEPT after a failed placement (retry), stat err=%v", sErr)
|
||||
}
|
||||
}
|
||||
|
||||
// D (F-3a-3): mapping refuses the namespace root itself among the snapshot paths.
|
||||
func TestMapOffsiteRestorePaths_RefusesNamespaceRoot(t *testing.T) {
|
||||
oldNs := "/old/ns"
|
||||
snap := []string{oldNs + "/backups/primary/app", oldNs} // oldNs itself must be refused
|
||||
if _, err := mapOffsiteRestorePaths(snap, "app", "/scratch", "/new/ns"); err == nil {
|
||||
t.Error("the namespace root itself among snapshot paths must be refused (F-3a-3)")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,345 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Offsite backup progress (v0.147.0, feedback slice 4c).
|
||||
//
|
||||
// THE PROBLEM: „Távoli mentés most" started a background restic run and redirected with „A távoli
|
||||
// mentés elindult". After that the page polled a status field whose only values were running / ok /
|
||||
// error. For a first offsite push of tens of gigabytes over SFTP that is 20+ minutes of a spinner
|
||||
// with no total, no percentage and no indication of WHICH app is being pushed — indistinguishable
|
||||
// from a hang.
|
||||
//
|
||||
// restic already reports all of it: `backup --json` writes newline-delimited status objects to
|
||||
// stdout. We only had to stop throwing them away — the existing runner seam uses CombinedOutput(),
|
||||
// which buffers everything until exit.
|
||||
//
|
||||
// SCOPE: the MANUAL trigger only. The nightly scheduled run stays silent (nobody is watching a
|
||||
// progress bar at 03:00, and a sink left installed would keep publishing stale percentages into a
|
||||
// page that never asked). The sink is installed for the duration of a manual run and cleared after.
|
||||
|
||||
// OffboxProgress is a snapshot of an in-flight manual offsite backup.
|
||||
type OffboxProgress struct {
|
||||
Active bool `json:"active"`
|
||||
CurrentApp string `json:"current_app"`
|
||||
Percent float64 `json:"percent"` // 0..100, restic's byte-based percent_done
|
||||
BytesDone int64 `json:"bytes_done"`
|
||||
TotalBytes int64 `json:"total_bytes"`
|
||||
DoneHuman string `json:"done_human"`
|
||||
TotalHuman string `json:"total_human"`
|
||||
// FilesDone/TotalFiles matter more than they look. On an INCREMENTAL run where nothing changed,
|
||||
// restic transfers no new bytes: bytes_done stays 0 (it is `omitempty`, so it is not even in the
|
||||
// JSON) and percent_done stays 0 for the whole run, while restic still walks every file. Measured
|
||||
// on the demo box: a 430MB immich push sat at 0% for 40+ seconds and then completed. A byte-only
|
||||
// bar is therefore indistinguishable from a hang precisely in the COMMON case. File counts move
|
||||
// in that case, so the page falls back to them.
|
||||
FilesDone int64 `json:"files_done"`
|
||||
TotalFiles int64 `json:"total_files"`
|
||||
// CurrentFile/ElapsedSec are the last resort, and on real data the most important fields here.
|
||||
// restic 0.14 only counts a file into files_done/bytes_done when it COMPLETES, so a single
|
||||
// dominant file freezes both counters: measured on the demo box, immich sat at files_done 1 of 46
|
||||
// and bytes_done 0 for 42 seconds while restic worked on one ~430MB volume tar. No percentage can
|
||||
// move during that window. What CAN be shown truthfully is which file is being processed and how
|
||||
// long it has been going — "working on X, 42s" is a completely different message from "0%".
|
||||
CurrentFile string `json:"current_file"`
|
||||
ElapsedSec int64 `json:"elapsed_sec"`
|
||||
// Phase names the part of the run in progress. A run is NOT just the per-app loop: after the last
|
||||
// app come the shares leg and `forget --prune`, which on the demo box took 40 of a 57-second run.
|
||||
// Without this the card froze on the last app's finished counters for that whole tail — the same
|
||||
// silence the slice exists to remove, just relocated. "" = per-app backup.
|
||||
Phase string `json:"phase"`
|
||||
}
|
||||
|
||||
// resticStatusLine is the subset of restic's `--json` status object we consume. restic emits several
|
||||
// message_types (status, summary, error, verbose_status); anything that is not "status" is ignored
|
||||
// here rather than treated as garbage, because restic adds new types between versions and an unknown
|
||||
// type must never break the run. Schema captured from
|
||||
// restic 0.14.0 (the version in the controller image) via a live `backup --dry-run --json`:
|
||||
//
|
||||
// {"message_type":"status","percent_done":0,"total_files":1,"total_bytes":112}
|
||||
// {"message_type":"status","percent_done":0.558,"total_files":173,"files_done":87,
|
||||
// "total_bytes":166878,"bytes_done":93161,"current_files":[...]}
|
||||
//
|
||||
// Note every numeric field except percent_done is `omitempty` on restic's side: a zero simply is not
|
||||
// in the JSON. That is why an incremental run reports no bytes_done at all rather than an explicit 0.
|
||||
type resticStatusLine struct {
|
||||
MessageType string `json:"message_type"`
|
||||
PercentDone float64 `json:"percent_done"` // 0..1
|
||||
TotalBytes int64 `json:"total_bytes"`
|
||||
BytesDone int64 `json:"bytes_done"`
|
||||
TotalFiles int64 `json:"total_files"`
|
||||
FilesDone int64 `json:"files_done"`
|
||||
CurrentFiles []string `json:"current_files"`
|
||||
SecondsElapsed int64 `json:"seconds_elapsed"`
|
||||
}
|
||||
|
||||
// resticProgress is one parsed status line.
|
||||
type resticProgress struct {
|
||||
Percent float64 // 0..100
|
||||
BytesDone int64
|
||||
TotalBytes int64
|
||||
FilesDone int64
|
||||
TotalFiles int64
|
||||
CurrentFile string
|
||||
ElapsedSec int64
|
||||
}
|
||||
|
||||
// parseResticStatus parses ONE line of restic --json output.
|
||||
//
|
||||
// Kept as a pure function precisely so it can be tested without restic, a network, or a repo — the
|
||||
// parser is the part that silently rots when restic changes its output, and a progress bar that
|
||||
// quietly stops moving is worse than no progress bar at all.
|
||||
func parseResticStatus(line string) (resticProgress, bool) {
|
||||
var s resticStatusLine
|
||||
if err := json.Unmarshal([]byte(line), &s); err != nil {
|
||||
return resticProgress{}, false
|
||||
}
|
||||
if s.MessageType != "status" {
|
||||
return resticProgress{}, false
|
||||
}
|
||||
pct := s.PercentDone * 100
|
||||
// restic revises its total as the scan proceeds, so percent_done legitimately moves backwards
|
||||
// mid-run and has been seen slightly above 1 near completion. Clamp — a bar wider than its track
|
||||
// is a visible bug.
|
||||
if pct < 0 {
|
||||
pct = 0
|
||||
}
|
||||
if pct > 100 {
|
||||
pct = 100
|
||||
}
|
||||
cur := ""
|
||||
if len(s.CurrentFiles) > 0 {
|
||||
cur = s.CurrentFiles[0]
|
||||
}
|
||||
return resticProgress{
|
||||
Percent: pct, BytesDone: s.BytesDone, TotalBytes: s.TotalBytes,
|
||||
FilesDone: s.FilesDone, TotalFiles: s.TotalFiles,
|
||||
CurrentFile: cur, ElapsedSec: s.SecondsElapsed,
|
||||
}, true
|
||||
}
|
||||
|
||||
// offboxProgressState is the published snapshot, guarded independently of the Manager mutex so a
|
||||
// poll never blocks behind the running backup.
|
||||
type offboxProgressState struct {
|
||||
mu sync.Mutex
|
||||
cur OffboxProgress
|
||||
live bool
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) begin() {
|
||||
p.mu.Lock()
|
||||
p.cur = OffboxProgress{Active: true}
|
||||
p.live = true
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) end() {
|
||||
p.mu.Lock()
|
||||
p.cur = OffboxProgress{}
|
||||
p.live = false
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) setApp(app string) {
|
||||
p.mu.Lock()
|
||||
if p.live {
|
||||
// A new app resets the byte counters: restic's percentages are per-invocation, and carrying
|
||||
// the previous app's 100% into the next app's start would show a bar that jumps backwards.
|
||||
p.cur.CurrentApp = app
|
||||
p.cur.Phase = ""
|
||||
p.cur.Percent, p.cur.BytesDone, p.cur.TotalBytes = 0, 0, 0
|
||||
p.cur.FilesDone, p.cur.TotalFiles = 0, 0
|
||||
p.cur.CurrentFile, p.cur.ElapsedSec = "", 0
|
||||
p.cur.DoneHuman, p.cur.TotalHuman = "", ""
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// setPhase marks a non-per-app stage of the run and clears the app-scoped counters, so the card
|
||||
// stops showing the last app's finished numbers against work that is no longer about that app.
|
||||
func (p *offboxProgressState) setPhase(phase string) {
|
||||
p.mu.Lock()
|
||||
if p.live {
|
||||
p.cur.Phase = phase
|
||||
p.cur.CurrentApp = ""
|
||||
p.cur.Percent, p.cur.BytesDone, p.cur.TotalBytes = 0, 0, 0
|
||||
p.cur.FilesDone, p.cur.TotalFiles = 0, 0
|
||||
p.cur.CurrentFile, p.cur.ElapsedSec = "", 0
|
||||
p.cur.DoneHuman, p.cur.TotalHuman = "", ""
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// OffboxPhaseDump is the PRE-app-loop stage (R-44, v0.148.0); OffboxPhaseShares /
|
||||
// OffboxPhaseRetention are the post-app-loop stages.
|
||||
const (
|
||||
OffboxPhaseDump = "dump"
|
||||
OffboxPhaseShares = "shares"
|
||||
OffboxPhaseRetention = "retention"
|
||||
)
|
||||
|
||||
func (p *offboxProgressState) update(r resticProgress) {
|
||||
p.mu.Lock()
|
||||
if p.live {
|
||||
p.cur.Percent, p.cur.BytesDone, p.cur.TotalBytes = r.Percent, r.BytesDone, r.TotalBytes
|
||||
p.cur.FilesDone, p.cur.TotalFiles = r.FilesDone, r.TotalFiles
|
||||
p.cur.ElapsedSec = r.ElapsedSec
|
||||
// Keep the last KNOWN current file: restic omits current_files on some status ticks, and
|
||||
// blanking the label every other second is its own kind of flicker.
|
||||
if r.CurrentFile != "" {
|
||||
p.cur.CurrentFile = r.CurrentFile
|
||||
}
|
||||
p.cur.DoneHuman, p.cur.TotalHuman = humanizeBytes(r.BytesDone), humanizeBytes(r.TotalBytes)
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) snapshot() OffboxProgress {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
return p.cur
|
||||
}
|
||||
|
||||
// OffboxProgressSnapshot is the poll surface for the „Távoli mentés" page.
|
||||
func (m *Manager) OffboxProgressSnapshot() OffboxProgress { return m.offboxProgress.snapshot() }
|
||||
|
||||
// offboxStreamRunner is the streaming restic-exec seam: like offboxRunner, but calls onLine for each
|
||||
// stdout line AS IT ARRIVES instead of only returning the buffered output at exit. Tests inject a
|
||||
// fake that emits canned `--json` status lines, so the whole progress path is exercised without
|
||||
// restic, a network or a repo.
|
||||
type offboxStreamRunner func(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error)
|
||||
|
||||
// SetOffboxStreamRunner installs the streaming seam (nil → the real streaming exec).
|
||||
func (m *Manager) SetOffboxStreamRunner(r offboxStreamRunner) { m.offboxStreamRunner = r }
|
||||
|
||||
func (m *Manager) streamRunner() offboxStreamRunner {
|
||||
if m.offboxStreamRunner != nil {
|
||||
return m.offboxStreamRunner
|
||||
}
|
||||
return defaultOffboxStreamRunner
|
||||
}
|
||||
|
||||
// defaultOffboxStreamRunner runs restic with stdout scanned line-by-line. stderr is captured whole
|
||||
// (restic's --json progress goes to stdout; errors go to stderr) and appended to the returned output
|
||||
// so callers keep the same error-diagnosis material CombinedOutput gave them.
|
||||
func defaultOffboxStreamRunner(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error) {
|
||||
cmd := exec.CommandContext(ctx, "restic", args...)
|
||||
cmd.Env = append(os.Environ(), env...)
|
||||
|
||||
stdout, err := cmd.StdoutPipe()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var stderr syncBuf
|
||||
cmd.Stderr = &stderr
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
var tail lineTail
|
||||
scanner := bufio.NewScanner(stdout)
|
||||
// restic status lines are small, but a --json summary listing many paths can exceed the 64KB
|
||||
// default; a scanner that dies mid-run would silently freeze the progress bar.
|
||||
scanner.Buffer(make([]byte, 0, 64*1024), 4*1024*1024)
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
tail.add(line)
|
||||
if onLine != nil {
|
||||
onLine(line)
|
||||
}
|
||||
}
|
||||
_, _ = io.Copy(io.Discard, stdout)
|
||||
|
||||
werr := cmd.Wait()
|
||||
// Keep only the tail of stdout: the full --json stream of a large backup is megabytes of status
|
||||
// spam, and every caller uses this output for error diagnosis (and lock-pattern matching) only.
|
||||
out := append(tail.bytes(), stderr.bytes()...)
|
||||
return out, werr
|
||||
}
|
||||
|
||||
// lineTail keeps the last N lines seen, so error diagnosis has context without buffering the whole
|
||||
// --json stream.
|
||||
type lineTail struct {
|
||||
lines []string
|
||||
}
|
||||
|
||||
func (t *lineTail) add(s string) {
|
||||
const keep = 40
|
||||
t.lines = append(t.lines, s)
|
||||
if len(t.lines) > keep {
|
||||
t.lines = t.lines[len(t.lines)-keep:]
|
||||
}
|
||||
}
|
||||
|
||||
func (t *lineTail) bytes() []byte {
|
||||
var b []byte
|
||||
for _, l := range t.lines {
|
||||
b = append(b, l...)
|
||||
b = append(b, '\n')
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
type syncBuf struct {
|
||||
mu sync.Mutex
|
||||
b []byte
|
||||
}
|
||||
|
||||
func (s *syncBuf) Write(p []byte) (int, error) {
|
||||
s.mu.Lock()
|
||||
s.b = append(s.b, p...)
|
||||
s.mu.Unlock()
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
func (s *syncBuf) bytes() []byte {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return append([]byte{}, s.b...)
|
||||
}
|
||||
|
||||
// resticBackupStep is resticStep's streaming twin, used ONLY by the app-backup leg when a manual run
|
||||
// has a progress sink installed. It keeps resticStep's crash-lock self-heal semantics by delegating
|
||||
// the retry path to resticStep (a retry after an unlock is rare and does not need progress).
|
||||
func (m *Manager) resticBackupStep(ctx context.Context, env, base []string, label, app string, args ...string) ([]byte, error) {
|
||||
if !m.offboxProgress.snapshot().Active {
|
||||
return m.resticStep(ctx, env, base, label, args...) // nightly / no watcher: unchanged path
|
||||
}
|
||||
m.offboxProgress.setApp(app)
|
||||
full := append(append([]string{}, base...), args...)
|
||||
// --json turns on the machine-readable progress stream. It is added ONLY here, so the nightly
|
||||
// run's output format (and everything that greps it) is untouched.
|
||||
full = append(full, "--json")
|
||||
out, err := m.streamRunner()(ctx, env, func(line string) {
|
||||
if r, ok := parseResticStatus(line); ok {
|
||||
m.offboxProgress.update(r)
|
||||
}
|
||||
}, full...)
|
||||
if err == nil || !offboxLockRe.Match(out) {
|
||||
return out, err
|
||||
}
|
||||
// Lock collision: fall back to the non-streaming step, which owns the unlock --remove-all
|
||||
// self-heal. Progress stalls for that one retry; correctness beats a moving bar.
|
||||
m.logger.Printf("[WARN] [offbox] %s hit a lock during a manual run — retrying via the self-healing step", label)
|
||||
return m.resticStep(ctx, env, base, label, args...)
|
||||
}
|
||||
|
||||
// beginManualProgress installs the progress sink for a manual run and returns the cleanup func.
|
||||
func (m *Manager) beginManualProgress() func() {
|
||||
m.offboxProgress.begin()
|
||||
started := time.Now()
|
||||
return func() {
|
||||
m.logger.Printf("[INFO] [offbox] manual run progress reporting ended after %s", time.Since(started).Round(time.Second))
|
||||
m.offboxProgress.end()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// v0.147.0 slice 4c — live progress for a MANUAL offsite run.
|
||||
//
|
||||
// These tests exercise the WHOLE path a real run takes: a fake restic emits canned `--json` status
|
||||
// lines through the streaming seam, and the assertion is on what the poll surface
|
||||
// (OffboxProgressSnapshot) reports — not on the parser in isolation. A parser that works but is
|
||||
// never wired to the snapshot would leave the customer looking at the same silent spinner, which is
|
||||
// the bug being fixed.
|
||||
//
|
||||
// RED-PROOF (run manually, both confirmed to fail):
|
||||
// 1. break the parser — flip `s.MessageType != "status"` to `== "status"` in parseResticStatus:
|
||||
// TestManualRunReportsParsedProgress fails ("percent = 0, want 42").
|
||||
// 2. drop the wiring — make resticBackupStep always delegate to resticStep:
|
||||
// the same test fails (no --json, no stream, no snapshot).
|
||||
|
||||
// jsonStatus is one restic --json status line.
|
||||
func jsonStatus(pct float64, done, total int64) string {
|
||||
return `{"message_type":"status","percent_done":` + ftoa(pct) + `,"total_bytes":` + itoa(total) + `,"bytes_done":` + itoa(done) + `}`
|
||||
}
|
||||
|
||||
func ftoa(f float64) string {
|
||||
// small helper — avoids strconv import noise in the canned lines
|
||||
switch f {
|
||||
case 0:
|
||||
return "0"
|
||||
case 0.42:
|
||||
return "0.42"
|
||||
case 1:
|
||||
return "1"
|
||||
}
|
||||
return "0.5"
|
||||
}
|
||||
|
||||
func itoa(i int64) string {
|
||||
if i == 0 {
|
||||
return "0"
|
||||
}
|
||||
var b []byte
|
||||
neg := i < 0
|
||||
if neg {
|
||||
i = -i
|
||||
}
|
||||
for i > 0 {
|
||||
b = append([]byte{byte('0' + i%10)}, b...)
|
||||
i /= 10
|
||||
}
|
||||
if neg {
|
||||
return "-" + string(b)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// TestManualRunReportsParsedProgress is the headline: a manual run driven by a fake restic that emits
|
||||
// --json status lines must make OffboxProgressSnapshot report the parsed percentage, byte counts and
|
||||
// the app currently being pushed.
|
||||
func TestManualRunReportsParsedProgress(t *testing.T) {
|
||||
e := newSharesOffboxEnv(t, "immich")
|
||||
if err := e.sett.SetSMBEnabled(false); err != nil { // keep this test to the app leg only
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var seen []OffboxProgress
|
||||
var sawJSONFlag bool
|
||||
e.m.SetOffboxStreamRunner(func(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error) {
|
||||
for _, a := range args {
|
||||
if a == "--json" {
|
||||
sawJSONFlag = true
|
||||
}
|
||||
}
|
||||
// Emit a scan phase (total not yet known), then real progress, sampling the published
|
||||
// snapshot after each line exactly as the polling page would.
|
||||
onLine(jsonStatus(0, 0, 0))
|
||||
seen = append(seen, e.m.OffboxProgressSnapshot())
|
||||
onLine(`{"message_type":"verbose_status","action":"unchanged"}`) // must be ignored, not fatal
|
||||
onLine(jsonStatus(0.42, 4200, 10000))
|
||||
seen = append(seen, e.m.OffboxProgressSnapshot())
|
||||
return []byte(`{"message_type":"summary","snapshot_id":"abc"}`), nil
|
||||
})
|
||||
// The non-streaming seam still serves every other restic call (cat config, forget, snapshots…).
|
||||
e.m.SetOffboxRunner(func(ctx context.Context, env []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[]`), nil
|
||||
}
|
||||
return []byte(""), nil
|
||||
})
|
||||
|
||||
if err := e.m.RunOffboxBackupWithProgress(context.Background()); err != nil {
|
||||
t.Fatalf("manual run: %v", err)
|
||||
}
|
||||
|
||||
if !sawJSONFlag {
|
||||
t.Fatal("the manual backup leg did not pass --json to restic — nothing could ever be parsed")
|
||||
}
|
||||
if len(seen) != 2 {
|
||||
t.Fatalf("expected 2 sampled snapshots, got %d", len(seen))
|
||||
}
|
||||
|
||||
// While restic is still scanning, total is unknown: report 0 rather than inventing a percentage.
|
||||
if seen[0].TotalBytes != 0 || seen[0].Percent != 0 {
|
||||
t.Errorf("scan phase: got %+v, want zeroed counters", seen[0])
|
||||
}
|
||||
if seen[0].CurrentApp != "immich" {
|
||||
t.Errorf("scan phase: current_app = %q, want %q", seen[0].CurrentApp, "immich")
|
||||
}
|
||||
|
||||
got := seen[1]
|
||||
if got.Percent != 42 {
|
||||
t.Errorf("percent = %v, want 42", got.Percent)
|
||||
}
|
||||
if got.BytesDone != 4200 || got.TotalBytes != 10000 {
|
||||
t.Errorf("bytes = %d/%d, want 4200/10000", got.BytesDone, got.TotalBytes)
|
||||
}
|
||||
if got.CurrentApp != "immich" {
|
||||
t.Errorf("current_app = %q, want %q", got.CurrentApp, "immich")
|
||||
}
|
||||
if got.TotalHuman == "" || got.DoneHuman == "" {
|
||||
t.Errorf("humanized byte strings not populated: %+v", got)
|
||||
}
|
||||
if !got.Active {
|
||||
t.Error("progress reported inactive during a manual run")
|
||||
}
|
||||
|
||||
// After the run the sink must be torn down, or the page would keep rendering a stale bar.
|
||||
if after := e.m.OffboxProgressSnapshot(); after.Active || after.Percent != 0 {
|
||||
t.Errorf("progress still published after the run: %+v", after)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNightlyRunStaysSilent pins the scope decision: the scheduled run must neither install the sink
|
||||
// nor pass --json, so its output format (and everything that greps it) is untouched.
|
||||
func TestNightlyRunStaysSilent(t *testing.T) {
|
||||
e := newSharesOffboxEnv(t, "immich")
|
||||
if err := e.sett.SetSMBEnabled(false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
e.m.SetOffboxStreamRunner(func(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error) {
|
||||
t.Fatal("the nightly run used the streaming runner — progress must be manual-only")
|
||||
return nil, nil
|
||||
})
|
||||
var backupArgs []string
|
||||
e.m.SetOffboxRunner(func(ctx context.Context, env []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "backup") {
|
||||
backupArgs = append([]string{}, args...)
|
||||
}
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[]`), nil
|
||||
}
|
||||
return []byte(""), nil
|
||||
})
|
||||
|
||||
if err := e.m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("nightly run: %v", err)
|
||||
}
|
||||
if backupArgs == nil {
|
||||
t.Fatal("no backup call was made")
|
||||
}
|
||||
if contains(backupArgs, "--json") {
|
||||
t.Errorf("the nightly run passed --json: %v", backupArgs)
|
||||
}
|
||||
if p := e.m.OffboxProgressSnapshot(); p.Active {
|
||||
t.Errorf("the nightly run published progress: %+v", p)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusIgnoresNonStatus covers the lines restic actually interleaves with status
|
||||
// output. An unknown message_type must be ignored, never mistaken for progress — restic adds new
|
||||
// types between versions.
|
||||
func TestParseResticStatusIgnoresNonStatus(t *testing.T) {
|
||||
for _, line := range []string{
|
||||
`{"message_type":"summary","total_bytes_processed":123}`,
|
||||
`{"message_type":"verbose_status","action":"new"}`,
|
||||
`{"message_type":"error","error":{"message":"boom"}}`,
|
||||
`not json at all`,
|
||||
``,
|
||||
`{}`,
|
||||
} {
|
||||
if _, ok := parseResticStatus(line); ok {
|
||||
t.Errorf("parsed a non-status line as progress: %q", line)
|
||||
}
|
||||
}
|
||||
r, ok := parseResticStatus(jsonStatus(0.42, 4200, 10000))
|
||||
if !ok {
|
||||
t.Fatal("a real status line was not parsed")
|
||||
}
|
||||
if r.Percent != 42 || r.BytesDone != 4200 || r.TotalBytes != 10000 {
|
||||
t.Errorf("got %v %d %d, want 42 4200 10000", r.Percent, r.BytesDone, r.TotalBytes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusIncrementalRun is the case that MATTERS and the one a synthetic test suite
|
||||
// would never think to write. It is a real status line shape from restic 0.14 on the demo box:
|
||||
// an incremental push where nothing changed transfers no new bytes, so restic omits bytes_done
|
||||
// entirely (`omitempty`) and percent_done stays 0 — while files_done climbs steadily.
|
||||
//
|
||||
// Measured live: a 430MB immich push reported 0% for 40+ seconds and then completed. A byte-only
|
||||
// progress bar is therefore indistinguishable from a hang in the COMMON case, which is the exact
|
||||
// failure 4c exists to remove. The file counters must survive parsing so the page can fall back to
|
||||
// them.
|
||||
func TestParseResticStatusIncrementalRun(t *testing.T) {
|
||||
// bytes_done and files_done absent — restic's scan-start state.
|
||||
r, ok := parseResticStatus(`{"message_type":"status","percent_done":0,"total_files":1,"total_bytes":112}`)
|
||||
if !ok {
|
||||
t.Fatal("scan-start status line was not parsed")
|
||||
}
|
||||
if r.BytesDone != 0 || r.TotalBytes != 112 || r.TotalFiles != 1 {
|
||||
t.Errorf("scan-start: got %+v", r)
|
||||
}
|
||||
|
||||
// The incremental steady state: no bytes moving, files moving.
|
||||
r, ok = parseResticStatus(`{"message_type":"status","percent_done":0,"total_files":8123,"files_done":4110,"total_bytes":451130451}`)
|
||||
if !ok {
|
||||
t.Fatal("incremental status line was not parsed")
|
||||
}
|
||||
if r.BytesDone != 0 {
|
||||
t.Errorf("bytes_done = %d, want 0 (absent in the JSON)", r.BytesDone)
|
||||
}
|
||||
if r.FilesDone != 4110 || r.TotalFiles != 8123 {
|
||||
t.Errorf("file counters lost: got %d/%d, want 4110/8123 — the page has nothing left to move",
|
||||
r.FilesDone, r.TotalFiles)
|
||||
}
|
||||
if r.TotalBytes != 451130451 {
|
||||
t.Errorf("total_bytes = %d, want 451130451", r.TotalBytes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusKeepsCurrentFileAndElapsed — the case where NO counter can move: restic 0.14
|
||||
// only counts a file when it completes, so an app dominated by one big archive freezes bytes_done
|
||||
// AND files_done. Measured on the demo box: immich at 1 of 46 files, 0 bytes, for 42 seconds while
|
||||
// restic worked through a single ~430MB volume tar. current_files + seconds_elapsed are then the only
|
||||
// honest signals of liveness left, so losing them in parsing would put the bar back to looking hung.
|
||||
func TestParseResticStatusKeepsCurrentFileAndElapsed(t *testing.T) {
|
||||
line := `{"message_type":"status","seconds_elapsed":42,"percent_done":0,"total_files":46,` +
|
||||
`"files_done":1,"total_bytes":451130451,"current_files":["/mnt/hdd/felhom-data/backups/primary/immich/volumes/immich_upload.tar"]}`
|
||||
r, ok := parseResticStatus(line)
|
||||
if !ok {
|
||||
t.Fatal("status line was not parsed")
|
||||
}
|
||||
if r.ElapsedSec != 42 {
|
||||
t.Errorf("elapsed = %d, want 42", r.ElapsedSec)
|
||||
}
|
||||
if r.CurrentFile == "" {
|
||||
t.Fatal("current_file lost — with no counter moving this is the only liveness signal left")
|
||||
}
|
||||
if want := "immich_upload.tar"; !strings.HasSuffix(r.CurrentFile, want) {
|
||||
t.Errorf("current_file = %q, want it to end in %q", r.CurrentFile, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProgressKeepsLastKnownCurrentFile — restic omits current_files on some status ticks. Blanking
|
||||
// the label every other second is its own kind of flicker, so the last known value must persist.
|
||||
func TestProgressKeepsLastKnownCurrentFile(t *testing.T) {
|
||||
var st offboxProgressState
|
||||
st.begin()
|
||||
st.setApp("immich")
|
||||
st.update(resticProgress{CurrentFile: "/data/big.tar", ElapsedSec: 5, TotalFiles: 46, FilesDone: 1})
|
||||
st.update(resticProgress{CurrentFile: "", ElapsedSec: 7, TotalFiles: 46, FilesDone: 1}) // tick without current_files
|
||||
if got := st.snapshot().CurrentFile; got != "/data/big.tar" {
|
||||
t.Errorf("current_file = %q after a tick that omitted it, want the last known value", got)
|
||||
}
|
||||
if got := st.snapshot().ElapsedSec; got != 7 {
|
||||
t.Errorf("elapsed = %d, want it to keep advancing (7)", got)
|
||||
}
|
||||
// A new app must clear it — otherwise the previous app's file is shown against the next one.
|
||||
st.setApp("nextcloud")
|
||||
if got := st.snapshot().CurrentFile; got != "" {
|
||||
t.Errorf("current_file = %q after switching app, want cleared", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSetPhaseClearsAppScopedCounters — a run is not only the per-app loop. The shares leg and
|
||||
// forget --prune follow it and took 40 of a 57-second run on the demo box. Without a phase the card
|
||||
// kept showing the last app's finished counters ("calibre-web 8 / 8") for that whole tail, which is
|
||||
// the same frozen-looking silence 4c exists to remove, just relocated to the end of the run.
|
||||
func TestSetPhaseClearsAppScopedCounters(t *testing.T) {
|
||||
var st offboxProgressState
|
||||
st.begin()
|
||||
st.setApp("calibre-web")
|
||||
st.update(resticProgress{Percent: 100, BytesDone: 850000, TotalBytes: 850000, FilesDone: 8, TotalFiles: 8, CurrentFile: "/data/x"})
|
||||
|
||||
st.setPhase(OffboxPhaseRetention)
|
||||
got := st.snapshot()
|
||||
if got.Phase != OffboxPhaseRetention {
|
||||
t.Errorf("phase = %q, want %q", got.Phase, OffboxPhaseRetention)
|
||||
}
|
||||
if got.CurrentApp != "" || got.FilesDone != 0 || got.TotalFiles != 0 || got.BytesDone != 0 || got.CurrentFile != "" {
|
||||
t.Errorf("app-scoped counters survived the phase switch: %+v — the card would show the last "+
|
||||
"app's finished numbers against retention work", got)
|
||||
}
|
||||
// Starting another app must clear the phase again, or the card would stay on „Karbantartás".
|
||||
st.setApp("immich")
|
||||
if p := st.snapshot(); p.Phase != "" || p.CurrentApp != "immich" {
|
||||
t.Errorf("after setApp: phase=%q app=%q, want phase cleared and app set", p.Phase, p.CurrentApp)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusClampsPercent — restic has been seen to report percent_done slightly above 1
|
||||
// near completion. A bar wider than its track is a visible bug.
|
||||
func TestParseResticStatusClampsPercent(t *testing.T) {
|
||||
r, ok := parseResticStatus(`{"message_type":"status","percent_done":1.04}`)
|
||||
if !ok || r.Percent != 100 {
|
||||
t.Errorf("percent = %v (ok=%v), want clamped to 100", r.Percent, ok)
|
||||
}
|
||||
r, ok = parseResticStatus(`{"message_type":"status","percent_done":-0.2}`)
|
||||
if !ok || r.Percent != 0 {
|
||||
t.Errorf("percent = %v (ok=%v), want clamped to 0", r.Percent, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLineTailKeepsOnlyTheTail — the --json stream of a large backup is megabytes of status spam, and
|
||||
// every caller uses this output for error diagnosis and lock-pattern matching. Buffering all of it
|
||||
// would be a memory leak proportional to backup size.
|
||||
func TestLineTailKeepsOnlyTheTail(t *testing.T) {
|
||||
var tl lineTail
|
||||
for i := 0; i < 500; i++ {
|
||||
tl.add("line-" + itoa(int64(i)))
|
||||
}
|
||||
out := string(tl.bytes())
|
||||
if strings.Contains(out, "line-0\n") {
|
||||
t.Error("the oldest line survived — the tail is unbounded")
|
||||
}
|
||||
if !strings.Contains(out, "line-499") {
|
||||
t.Error("the newest line was dropped")
|
||||
}
|
||||
if n := strings.Count(out, "\n"); n > 64 {
|
||||
t.Errorf("tail kept %d lines, want a small bounded number", n)
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user