Compare commits
212 Commits
3dfc49e578
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 2fa1efc5e5 | |||
| 330e4a051e | |||
| 90f2545679 | |||
| 8144a70a72 | |||
| 34d83f5a02 | |||
| c24f1920d9 | |||
| bb50e1293c | |||
| 3e3ee94b7b | |||
| ae10f64806 | |||
| 3ed5e3e770 | |||
| 3168a78935 | |||
| 89712563a0 | |||
| 1b66010298 | |||
| 68f3e12398 | |||
| f87be3575f | |||
| 38f4535bfa | |||
| 397d62136f | |||
| 86a78c6767 | |||
| b762a37097 | |||
| c732fe1283 | |||
| fcffaf573a | |||
| 37b5ba08a7 | |||
| 27d1165962 | |||
| 62998aab4f | |||
| 8dbbc98ff2 | |||
| 3d3b4496f3 | |||
| 0a9158d53e | |||
| 72368654e4 | |||
| de39e47f53 | |||
| a5d90ff801 | |||
| a491abef6c | |||
| 763de3a025 | |||
| c6b69d888e | |||
| 53e9bf0224 | |||
| 4d349d1106 | |||
| 9dc26459ea | |||
| 66d80efb9f | |||
| 7db42c5fec | |||
| a62bb3874b | |||
| 7534ea203d | |||
| c7446f2d6a | |||
| 1e759a16ec | |||
| 05cf352a2f | |||
| a3499d1807 | |||
| a315d623b8 | |||
| 62b85ecf13 | |||
| 636c51e542 | |||
| be3c5fa7f6 | |||
| 992803c10b | |||
| a91f055960 | |||
| 1214bae0a2 | |||
| 68f195676b | |||
| 33fcc502e4 | |||
| 2e936f43bf | |||
| 73b6dbc27d | |||
| f4796e0d00 | |||
| 58c703bd44 | |||
| a96c3d9473 | |||
| 73efb091d9 | |||
| 532f5712a8 | |||
| 1b1366bb6e | |||
| bdab80c933 | |||
| 9640e51321 | |||
| 0887fd676d | |||
| 88897a224e | |||
| db0d4b129d | |||
| 6c43bf6156 | |||
| fef07c3923 | |||
| 4be6467b50 | |||
| d5be67b913 | |||
| 9a3c4855d7 | |||
| 5adae4dad9 | |||
| cf48214f6c | |||
| 95eb5c2c1a | |||
| e6311f9fbc | |||
| 4bad6e06c9 | |||
| dcc3363d2f | |||
| 582135f861 | |||
| 3446609420 | |||
| a8f7c61d41 | |||
| dbcb306fcf | |||
| e7c44c0e0f | |||
| dcc400e175 | |||
| eaded79b18 | |||
| 7c32c74140 | |||
| 8cb3d7af91 | |||
| c432f701dd | |||
| 4115e88f68 | |||
| 4ed938cce4 | |||
| 2f27a363d5 | |||
| b331f18424 | |||
| cdaeb36972 | |||
| 3f7cf2a965 | |||
| 4d6c8a6056 | |||
| c1a63de1c7 | |||
| ff058a4f10 | |||
| fd50a73e65 | |||
| d8b3279731 | |||
| 3f048e042b | |||
| 3db8bfb953 | |||
| e000e201af | |||
| 4056feccee | |||
| fb91c8d766 | |||
| a63409c843 | |||
| 079265ad8e | |||
| 8f46495426 | |||
| ca013c8d27 | |||
| 86ea482fc1 | |||
| ba8bf9cd75 | |||
| e9c99566b0 | |||
| ccefff4f39 | |||
| b8598361b8 | |||
| 32200c7b5f | |||
| 3f0420ff9c | |||
| f5e106440d | |||
| 9e5ea56853 | |||
| de96efc0c5 | |||
| 47fda06ba1 | |||
| 9056f01fae | |||
| c7a3a90782 | |||
| 8fadbd9891 | |||
| 4773809334 | |||
| 2958946517 | |||
| 3b672ba74c | |||
| 7d5b0163ef | |||
| f6a8249593 | |||
| de14eedb9f | |||
| 9cc8424954 | |||
| 2487681396 | |||
| 0a582ea07b | |||
| dbf631312e | |||
| c97975c1df | |||
| e164fef70c | |||
| 82c67e32e1 | |||
| e33c1aeabc | |||
| d37bb1eb6a | |||
| 59d182a6b9 | |||
| e9a365e59c | |||
| dbd1ff0c17 | |||
| bf44216e79 | |||
| 1fd070615d | |||
| a04afc367b | |||
| 570fb30147 | |||
| 15206314ab | |||
| 8e5edb2865 | |||
| c23a0f6d2d | |||
| 77956d8df2 | |||
| 2c80868c63 | |||
| 7a53cb43ef | |||
| ea432ca74a | |||
| 4aa7d41cac | |||
| 987e915bf2 | |||
| a71cc58327 | |||
| cb8bf14599 | |||
| ce8531426c | |||
| 0eba37d5cd | |||
| 59cd260e57 | |||
| 9610906916 | |||
| 7013a5fd2e | |||
| 4aa2ce4b61 | |||
| 0fbd272ad2 | |||
| 5fdd2039fd | |||
| ea0d3f1764 | |||
| a96226a8ae | |||
| 4ab793ef31 | |||
| 94fa4d6fb9 | |||
| ac3790a11b | |||
| 83f20c8293 | |||
| 1dad3c97fd | |||
| 984ea8c8bd | |||
| 285dd1032f | |||
| 0f9b29a19a | |||
| f134eab609 | |||
| 9d1b4983f5 | |||
| 70cb21b058 | |||
| 3a9d744360 | |||
| b30e2e5a28 | |||
| 8d33a90888 | |||
| 332b2b84bc | |||
| 4978d355b8 | |||
| 78ff991f1c | |||
| fd40b29119 | |||
| 37e12c82a7 | |||
| 5c105fb49b | |||
| badf17bebd | |||
| 8db9232dea | |||
| 9f436c8a3b | |||
| 4646be1515 | |||
| c059fe4c28 | |||
| 9d001771ea | |||
| 29eda5d86e | |||
| 062357f778 | |||
| 2fcae041ae | |||
| ac7323dc9a | |||
| da56c3e994 | |||
| 86de16f6dd | |||
| 10ca8ac884 | |||
| 63a22e5911 | |||
| 111369dd10 | |||
| 77e8d5590b | |||
| b5d78d1e0f | |||
| 7f0b41c3e7 | |||
| fd93020f39 | |||
| 24d23b80bd | |||
| 57be785bad | |||
| cc50918244 | |||
| a5d870ea0e | |||
| d0c1d77491 | |||
| 3b70a9e9ab | |||
| 900c870212 | |||
| 85b76e0fc3 | |||
| c81df55dcb |
@@ -0,0 +1,43 @@
|
||||
---
|
||||
paths: ["controller/internal/agentapi/**"]
|
||||
---
|
||||
|
||||
# Coupling to the host agent — felhom-controller
|
||||
|
||||
`internal/agentapi` is **the disk seam**: the pinned-TLS client to the host agent's per-guest local
|
||||
API. The controller holds no Proxmox credentials; everything disk/host/Proxmox goes through here.
|
||||
|
||||
## Declaring a coupled feature
|
||||
|
||||
Controller behaviour that depends on a specific agent version needs **all three**, or it ships broken
|
||||
on an older box:
|
||||
|
||||
1. a `featureProbes` table row in `internal/agentapi/features.go`
|
||||
2. a `Supports` gate call **at the feature's entry point** — not somewhere on the path to it
|
||||
3. `MinAgent: X.Y.Z` in the CHANGELOG entry header
|
||||
|
||||
Rules: `felhom.eu/documentation/runbooks/publish-train-rules.md`.
|
||||
|
||||
## Never push a controller past the agent it depends on
|
||||
|
||||
The R-216 guard compared the box's agent against the **golden's** MinAgent while serving a **floor**
|
||||
that could point elsewhere. Raise a floor above the vouched golden — which the day-0 runbook
|
||||
recommends and a per-customer override makes trivial — and the guard checks a version it is not
|
||||
serving. A box then landed on a controller needing a newer agent, and its customer was told a correct
|
||||
recovery code was wrong.
|
||||
|
||||
**A floor above the vouched golden is HELD, with its own reason** (hub v0.97.0).
|
||||
|
||||
## Distinguish "could not reach" from "wrong answer"
|
||||
|
||||
A failed bundle FETCH must not be reported to a customer as a bad recovery code. Classify by **value**
|
||||
(`ErrBundleFetch` → HTTP 502), never by error string — a string is not something a caller can branch
|
||||
on. Unknown class → neutral message, never the typing message.
|
||||
|
||||
<!--
|
||||
R-224, measured live 2026-08-05 (CAMPAIGN-11 F3/F4) with a correct current code: 0.0556 s with the
|
||||
hub firewalled off and 0.0299 s with the agent stopped, against ~1.0 s for a genuine unseal — the
|
||||
machine accused the customer of something it had not attempted. A green test named this exact
|
||||
consequence since v0.125.0 and did not prevent it, because it asserted this package's error STRING
|
||||
one layer below where the merge happened. Fixed agent v0.126.0 + controller v0.202.0.
|
||||
-->
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
paths: ["controller/internal/backup/**", "controller/internal/appbackup/**", "controller/internal/recovery/**", "controller/internal/appexport/**", "controller/internal/quiesce/**"]
|
||||
---
|
||||
|
||||
# Backup, recovery units and export — felhom-controller
|
||||
|
||||
## Assert the consequence across the whole run, not the mechanism inside one function
|
||||
|
||||
The R-181 recovery-unit refusal claimed *"the previous unit is untouched and NOTHING was deleted"*.
|
||||
*Nothing deleted* held; **untouched was measured false** — the floor was checked ONLY in
|
||||
`captureAllRecoveryUnits`, while the two dump legs wrote the bulk into the same tree first and
|
||||
unguarded, so a 182,272 B tar became 2,147,666,432 B under a manifest that had not moved. A full
|
||||
green suite plus three of its own red-proofs missed it, because every one asserted the mechanism
|
||||
inside `captureAllRecoveryUnits`.
|
||||
|
||||
**The test that catches this class: fingerprint the tree before and after the whole backup run, and
|
||||
compare.** Full doctrine and the other eight instances: the `felhom-testing` skill.
|
||||
|
||||
## Presence is not success
|
||||
|
||||
A timestamp recording an **attempt** must never be read as evidence of a **result**. Where a status
|
||||
field travels alongside a timestamp, the verdict consults both — or the timestamp records only
|
||||
successes. Ask of any timestamp: *what exactly must have happened for this to be set?* If the answer
|
||||
is "we tried", it cannot answer "did it work".
|
||||
|
||||
**Corollary:** when a verdict changes which field it counts from, the alarm text has to change with
|
||||
it. `last run 8h ago` while alarming on a six-day-old success turns a true alarm into one the
|
||||
operator dismisses.
|
||||
|
||||
<!--
|
||||
Two instances. F-CRIT-2: a phantom snapshot's ctime set tier freshness — an aborted 1-byte upload
|
||||
made the tier look backed up. R-100: LastRun is written on failure, so a nightly-failing offsite
|
||||
tier kept the staleness clock fresh forever.
|
||||
-->
|
||||
|
||||
## Storage keys and paths
|
||||
|
||||
- Never guess a persisted key — it is `offbox`, not `offbox_target` (R-7b).
|
||||
- `.fab` export/import uses strict segment validation; bundles from controller ≤0.124.0 are hollow.
|
||||
- Recovery-unit restore and tier-2 copies share `appbackup`'s path primitives — change them there,
|
||||
once, not per caller.
|
||||
@@ -0,0 +1,54 @@
|
||||
---
|
||||
paths: ["controller/**/*.go", "controller/**/*.html", "controller/**/*.css", "controller/scripts/**"]
|
||||
---
|
||||
|
||||
# Gates and logging — felhom-controller
|
||||
|
||||
## The ONE entry point
|
||||
|
||||
**Run `python3 controller/scripts/controller_gates.py` (from `controller/`) after ANY change in this
|
||||
repo.** It runs all seven local gates — `template_id_gate`, `emoji_gate`, `native_confirm_gate`,
|
||||
`offbox_rename_gate`, `app_row_dedup_gate`, `mojibake_gate`, `docker_run_volume_path_gate` — plus
|
||||
`reuse_refs_check` and `instructions_gate` on the repo root, streaming each gate's own output and
|
||||
exiting non-zero if any fails.
|
||||
|
||||
- `--fast` selects the gates that touch no network and no container runtime; today that is all of them.
|
||||
- **A missing gate script is a FAILURE, never a skip.**
|
||||
- **The shared `reuse_refs_check.py` and `instructions_gate.py` live in `felhom.eu/scripts/` and are
|
||||
never copied here** — a copy would recreate the drift they detect; an absent sibling clone FAILS.
|
||||
- **The pre-push hook** (`.githooks/pre-push`) runs it with `--fast` and refuses a failing push. It is
|
||||
per-clone — switch it on once with `git config core.hooksPath .githooks`, and a manual run WARNS
|
||||
when this clone is unarmed. `git push --no-verify` bypasses it deliberately; **say so in the session
|
||||
report when you use it** — CI re-runs the same entry point on every push and **emails the operator
|
||||
on failure**, so a bypass is noticed even though it is not blocked (R-168, CLOSED 2026-08-02).
|
||||
|
||||
<!--
|
||||
WHY A RUNNER AND NOT SEVEN INVOCATIONS (2026-08-02, R-29) — rationale, not a directive.
|
||||
A census of all thirteen gates across the four repos found that every check a CLAUDE.md named was
|
||||
passing, and two of the four nobody is told to run were failing. This repo's CLAUDE.md used to name
|
||||
two of the seven; the other five were reachable only through a line in REUSE.md, and
|
||||
docker_run_volume_path_gate.py was RED. The single-entry-point shape is the only one that
|
||||
demonstrably gets run. app-catalog-felhom.eu/scripts/catalog_gates.py is the canonical version of
|
||||
the runner (R-161); repo_gates.py copies it. site_gates.py is a *gate*, not a runner — do not model
|
||||
new work on it.
|
||||
-->
|
||||
|
||||
## Logging
|
||||
|
||||
New leveled lines use `internal/logx` — DEBUG always reaches the debug ring; stdout respects
|
||||
`logging.level`. English, keys-never-values, durations on outcomes. Full rules:
|
||||
`felhom.eu/documentation/runbooks/logging-conventions.md`.
|
||||
|
||||
## Health checks issue no block I/O
|
||||
|
||||
A probe that touches a wedged device enters uninterruptible sleep, survives `SIGKILL`, and cannot be
|
||||
recovered until the device returns or the host reboots — so `systemctl restart` hangs too. A timeout
|
||||
protects the caller's control flow and nothing else: the blocked thread remains. Liveness is decided
|
||||
from `/proc` and kernel state, never by reading or writing the filesystem.
|
||||
|
||||
<!--
|
||||
Measured, R-117 spike §6.3 (felhom.eu/documentation/audits/SPIKE-r117-bind-liveness-2026-07-30.md):
|
||||
a probe stayed in D state 3m50s after kill -9; a buffered write with no fsync blocked too (O_CREAT
|
||||
needs journal access); and statfs/getdents returned HEALTHY on a namespace that EIOs every byte —
|
||||
fast, and wrong.
|
||||
-->
|
||||
@@ -0,0 +1,27 @@
|
||||
---
|
||||
paths: ["controller/internal/web/templates/**", "controller/internal/web/**/*.go", "**/*.css", "**/*.html"]
|
||||
---
|
||||
|
||||
# UI and Hungarian copy — felhom-controller
|
||||
|
||||
- **All UI text is Hungarian**, Budapest timezone.
|
||||
- Design tokens, badge/colour rules, the 2px/no-shadow/no-emoji/BOM hard rules and the mechanical
|
||||
gate to run after each surface: **use the `felhom-ui-design` skill.**
|
||||
- Template methods need **value receivers** — pointer receivers compile, pass `go vet`, pass the
|
||||
suite, and then 500 at render time.
|
||||
|
||||
## Grep fetched pages with ASCII-only substrings
|
||||
|
||||
Accented Hungarian patterns get mangled through the `ssh → pct exec → bash -c` chain and return a
|
||||
false `0` — which reads exactly like the banner or string being gone. Use `kezel`, `Utols`,
|
||||
`Biztons`. **Never let an accented pattern gate a conclusion.**
|
||||
|
||||
<!--
|
||||
From the 2026-07-20 remediation: an accented grep nearly produced a wrong "banner cleared" claim.
|
||||
This is the "an absent line is not evidence" rule aimed at a UTF-8 transport, not at a log.
|
||||
-->
|
||||
|
||||
## Credentials containing `!` or `'` break in heredoc-built helper scripts
|
||||
|
||||
History expansion eats `!!`. Use the proven inline `-d "password=$PW"` form for authed curl, and
|
||||
delete any credential-bearing helper from `/tmp` (host AND guest) when done.
|
||||
@@ -0,0 +1,104 @@
|
||||
# gates — re-run this repo's gate entry point on every push, on a machine that does not care who
|
||||
# pushed or what they typed.
|
||||
#
|
||||
# *** THIS REPORTS. IT CANNOT REFUSE. ***
|
||||
#
|
||||
# felhom repos push straight to `main` with no pull request, so there is no merge for a status
|
||||
# check to stand at. The refusing half is `.githooks/pre-push`, which is local to a clone and which
|
||||
# `git push --no-verify` skips; this half is what notices when that happened. Neither half is the
|
||||
# whole thing, and both are named in felhom.eu documentation/backlog/OPEN-ITEMS.md R-168.
|
||||
#
|
||||
# NO `uses:` STEP ANYWHERE, deliberately: JavaScript actions need a node runtime in the runner, and
|
||||
# the runner is a host-mode container with python3 and git and nothing else (see
|
||||
# homelab-manifests/gitea-system/act-runner.yaml for why it is not privileged). Probe P3 measured
|
||||
# that a plain `git fetch` of the pushed SHA from the in-cluster Gitea service is enough.
|
||||
#
|
||||
# A failing run must reach a person — a detector nobody hears is the defect R-29 filed, rebuilt one
|
||||
# layer up. That is the last step, and it runs ONLY on failure.
|
||||
name: gates
|
||||
on: [push]
|
||||
|
||||
jobs:
|
||||
gates:
|
||||
runs-on: felhom-gates
|
||||
steps:
|
||||
- name: Fetch the pushed commit and the sibling clone it needs
|
||||
# This repo's entry point invokes a SHARED checker that lives in the felhom.eu clone next
|
||||
# door and is deliberately never copied here — so CI has to reproduce the workspace's
|
||||
# sibling layout or the gate fails closed with "gate is MISSING". The sibling is also
|
||||
# needed for CONTENT: this repo's REUSE.md cites a path that lives in the hub.
|
||||
run: |
|
||||
# Shallow, and pinned to the exact SHA that was pushed — not to the branch tip,
|
||||
# which can move under us if two pushes race.
|
||||
mkdir -p ws/felhom-controller
|
||||
cd ws/felhom-controller
|
||||
git init -q .
|
||||
git remote add origin http://gitea.gitea-system.svc.cluster.local:3000/admin/felhom-controller.git
|
||||
git fetch -q --depth 1 origin "$GITHUB_SHA"
|
||||
git checkout -q FETCH_HEAD
|
||||
echo "checked out $(git rev-parse HEAD)"
|
||||
cd .. && git clone -q --depth 1 http://gitea.gitea-system.svc.cluster.local:3000/admin/felhom.eu.git felhom.eu
|
||||
echo "sibling felhom.eu present at $(cd felhom.eu && git rev-parse --short HEAD)"
|
||||
|
||||
- name: Run the gate entry point
|
||||
# The ONLY thing CI runs. No go build, no go test, no linting, no deploy. The
|
||||
# exit code IS the result: no `|| true`, no pipe that could swallow it.
|
||||
run: cd ws/felhom-controller/controller && python3 scripts/controller_gates.py --fast
|
||||
|
||||
- name: Alarm on failure
|
||||
# THE POINT OF THE WHOLE THING. Probe P5 measured that a failed run produces NO mail, NO
|
||||
# notification row and NO log line from Gitea itself — a red tick in a web UI nobody watches
|
||||
# is exactly the shape R-29 filed against. So the run sends its own alarm, on the project's
|
||||
# existing transactional path (Resend, the same one the hub uses), and prints the provider's
|
||||
# accepted id so "a message left the machine" is an observable, not an assumption.
|
||||
#
|
||||
# Pure python3 and urllib, NOT curl: the runner image carries python3 and git and nothing
|
||||
# else on purpose, and the first version of this step died on `curl: command not found`.
|
||||
# Reaching for a bigger image to send one HTTP request would have been the wrong trade.
|
||||
if: failure()
|
||||
env:
|
||||
RESEND_API_KEY: ${{ secrets.RESEND_API_KEY }}
|
||||
run: |
|
||||
python3 - <<'PY'
|
||||
import json, os, sys, urllib.request, urllib.error
|
||||
|
||||
key = os.environ.get("RESEND_API_KEY", "")
|
||||
if not key:
|
||||
sys.exit("ALARM FAILED: RESEND_API_KEY is empty — the alarm cannot be sent, and a "
|
||||
"silent alarm is worse than none. Set the user-level Actions secret.")
|
||||
|
||||
repo = os.environ.get("GITHUB_REPOSITORY", "?")
|
||||
sha = os.environ.get("GITHUB_SHA", "?")
|
||||
run = os.environ.get("GITHUB_RUN_NUMBER", "?")
|
||||
srv = os.environ.get("GITHUB_SERVER_URL", "https://gitea.dooplex.hu")
|
||||
|
||||
body = json.dumps({
|
||||
"from": "Felhom CI <monitoring@felhom.eu>",
|
||||
"to": ["admin@felhom.eu"],
|
||||
"subject": "[felhom CI] gates FAILED in %s" % repo,
|
||||
"text": (
|
||||
"The gate entry point exited non-zero.\n\n"
|
||||
"Repository : %s\n"
|
||||
"Commit : %s\n"
|
||||
"Run : %s/%s/actions/runs/%s\n\n"
|
||||
"The failing gate names itself in the run log.\n\n"
|
||||
"If the local pre-push hook was GREEN for this commit, then CI and the hook\n"
|
||||
"disagree - that is a finding about the gates themselves, not about CI, and it\n"
|
||||
"outranks whatever the push was for.\n"
|
||||
) % (repo, sha, srv, repo, run),
|
||||
}).encode()
|
||||
|
||||
req = urllib.request.Request(
|
||||
"https://api.resend.com/emails", data=body, method="POST",
|
||||
headers={"Authorization": "Bearer %s" % key,
|
||||
"Content-Type": "application/json",
|
||||
# Cloudflare fronts api.resend.com and BLOCKS the default
|
||||
# "Python-urllib/3.x" agent with its own 403 (error 1010) — which looks
|
||||
# exactly like an auth failure and is not one. Measured 2026-08-02.
|
||||
"User-Agent": "felhom-ci/1.0"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
print("RESEND-ACCEPTED id=%s" % json.load(r)["id"])
|
||||
except urllib.error.HTTPError as e:
|
||||
sys.exit("ALARM FAILED: Resend returned HTTP %s: %s" % (e.code, e.read().decode()[:300]))
|
||||
PY
|
||||
Executable
+82
@@ -0,0 +1,82 @@
|
||||
#!/bin/sh
|
||||
# pre-push — refuse a push that carries a broken gate. (2026-08-02, R-29 leg (b) first half.)
|
||||
#
|
||||
# Runs this repo's ONE gate entry point in --fast mode: only checks that touch no network and no
|
||||
# container runtime, so a push stays a push and never pulls images or starts containers. The slow
|
||||
# gates stay deliberate periodic runs; a hook that takes minutes gets bypassed within a week and
|
||||
# the bypass becomes the habit.
|
||||
#
|
||||
# BOTH LINES BELOW ARE DELIBERATE. An absent log line is not evidence a hook ran — a silent pass is
|
||||
# equally consistent with "gates green" and "hook never fired", so a passing push says so out loud.
|
||||
#
|
||||
# HONEST LIMITS, stated so this is not mistaken for enforcement it cannot provide:
|
||||
# * per-clone — core.hooksPath is local config and a clone does not carry it. Arm a clone once:
|
||||
# git config core.hooksPath .githooks
|
||||
# Any manual entry-point run WARNS when the clone is unarmed.
|
||||
# * skippable — `git push --no-verify` bypasses this entirely. That is on purpose: an escape
|
||||
# hatch that cannot be reached is one that gets removed the first time it is
|
||||
# inconvenient. USING IT MUST BE STATED IN THE SESSION REPORT.
|
||||
# The half that is neither per-clone nor skippable is CI — felhom.eu OPEN-ITEMS.md R-168.
|
||||
#
|
||||
# Measured 2026-08-02 (git 2.47.3): a relative core.hooksPath resolves correctly and the hook's cwd
|
||||
# is the repo root whether `git push` is issued from the root or from any subdirectory. The
|
||||
# explicit rev-parse below does not depend on that.
|
||||
set -u
|
||||
|
||||
root=$(git rev-parse --show-toplevel 2>/dev/null) || {
|
||||
echo "pre-push: FAIL - cannot resolve the repo root (git rev-parse --show-toplevel)." >&2
|
||||
exit 1
|
||||
}
|
||||
cd "$root" || exit 1
|
||||
|
||||
# ── WORKSPACE-ROOT ASSERTION (2026-08-05, R-204 rider) ───────────────────────────────────────────
|
||||
# Refuse a push from a clone outside the felhom workspace.
|
||||
#
|
||||
# WHY THIS IS A HOOK AND NOT A LINE IN A DOCUMENT: the workspace root is ALREADY written down, in
|
||||
# documentation/runbooks/workspace-CLAUDE.md and in the workspace-root CLAUDE.md ("stay inside it"),
|
||||
# and work drifted into a home directory anyway. A rule that has failed once as a reminder is not
|
||||
# fixed by writing it down again — it has to be asserted where it can bite.
|
||||
#
|
||||
# A PUSH IS THE RIGHT TRIGGER, deliberately: throwaway clones under /tmp for probes and red-proofs
|
||||
# never push, so nothing legitimate breaks. Reads and builds elsewhere stay unaffected.
|
||||
#
|
||||
# Symlinks are resolved on BOTH sides before comparison, so a symlinked path neither falsely passes
|
||||
# nor falsely fails. If the workspace root does not exist on this machine the check is SKIPPED, not
|
||||
# failed — this hook must not brick a legitimate clone on a different host.
|
||||
#
|
||||
# The only bypass is the documented `git push --no-verify`, whose use is already reportable.
|
||||
FELHOM_WORKSPACE_ROOT=/mnt/5_hdd/felhom.eu
|
||||
if [ -d "$FELHOM_WORKSPACE_ROOT" ]; then
|
||||
ws_real=$(cd "$FELHOM_WORKSPACE_ROOT" 2>/dev/null && pwd -P) || ws_real=""
|
||||
root_real=$(pwd -P) || root_real=""
|
||||
if [ -n "$ws_real" ] && [ -n "$root_real" ]; then
|
||||
case "$root_real/" in
|
||||
"$ws_real"/*) : ;; # inside the workspace — proceed
|
||||
*)
|
||||
echo "pre-push: PUSH REFUSED - this clone is OUTSIDE the felhom workspace." >&2
|
||||
echo " clone: $root_real" >&2
|
||||
echo " expected: under $ws_real (repos live in $ws_real/git/<repo>)" >&2
|
||||
echo " Work in the workspace clone, or bypass with 'git push --no-verify'" >&2
|
||||
echo " and state that you did in the session report." >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
fi
|
||||
|
||||
if ! command -v python3 >/dev/null 2>&1; then
|
||||
echo "pre-push: FAIL - python3 not found, so the gates CANNOT run. This is a failure, never a" >&2
|
||||
echo " pass by default. Install python3, or push with --no-verify and say so." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "pre-push [felhom-controller]: running controller/scripts/controller_gates.py --fast ..."
|
||||
python3 "controller/scripts/controller_gates.py" --fast
|
||||
rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo "pre-push [felhom-controller]: PUSH REFUSED - gates exited $rc. Fix the finding above, or bypass with" >&2
|
||||
echo " 'git push --no-verify' and state that you did in the session report." >&2
|
||||
else
|
||||
echo "pre-push [felhom-controller]: gates OK - push proceeding."
|
||||
fi
|
||||
exit $rc
|
||||
+4081
File diff suppressed because it is too large
Load Diff
@@ -1,171 +1,110 @@
|
||||
# CLAUDE.md — Project Instructions for Claude Code (`felhom-controller`)
|
||||
# CLAUDE.md — `felhom-controller`
|
||||
|
||||
> Read automatically at session start. Stable orientation only — **current state lives in
|
||||
> `CONTEXT.md` and the top of `CHANGELOG.md`**, never here. Cross-repo orientation: workspace-root
|
||||
> `e:\git\CLAUDE.md`.
|
||||
> Stable orientation only — **current state lives in `CONTEXT.md` and the top of `CHANGELOG.md`**,
|
||||
> never here. Cross-repo conventions (clean-tree gate, secrets, trunk-based, artifact taxonomy):
|
||||
> workspace-root `/mnt/5_hdd/felhom.eu/git/CLAUDE.md`. Path-scoped detail: `.claude/rules/`.
|
||||
|
||||
!!! IMPORTANT !!!
|
||||
- Always update CHANGELOG.md whenever you modified the code, and pushed to git!!
|
||||
- IF controller feature changed (new/modify/remove) always update the relevant part of controller/README.md with the architectural change!!
|
||||
## What this repo is
|
||||
|
||||
## Project overview
|
||||
|
||||
Felhom is a managed home-server business for Hungarian customers. This repo contains the
|
||||
**felhom-controller** — the Go application that manages Docker Compose stacks inside each customer
|
||||
LXC guest via a Hungarian-language web dashboard.
|
||||
|
||||
Read in this order:
|
||||
- **`REUSE.md`** — before writing new code (canonical helpers, patterns, traps, seams).
|
||||
- `CONTEXT.md` — current project state, decisions, roadmap (update after each session).
|
||||
- `controller/README.md` — full feature/architecture reference (update when features change).
|
||||
- `TASK.md` — the current task to implement (if it exists).
|
||||
|
||||
## System context — the three-component model
|
||||
|
||||
The project runs **on Proxmox**, with a locked three-component model:
|
||||
- **Hub** (`felhom.eu/hub/`) — operator backend on k3s.
|
||||
- **Host agent** (`felhom-agent/`) — one per Proxmox host; operator-tier; owns ALL Proxmox interaction.
|
||||
- **In-guest controller** (THIS repo) — one per customer LXC; **Docker-only; holds NO Proxmox
|
||||
credentials**. De-privileged: disk/host/Proxmox concerns are delegated to the host agent via the
|
||||
pinned local-API client (`internal/agentapi`); the controller keeps the app domain — stack/deploy
|
||||
management, the Hungarian web UI, app-data backup, metrics/telemetry, integrations, git-sync,
|
||||
notifications. Whole-guest backup (PBS vzdump) is the agent's.
|
||||
|
||||
> **Authoritative maps:** `felhom.eu/documentation/architecture/01/02/03-*.md` (topology/trust,
|
||||
> controller module map, host agent) + the code-verified feature docs in
|
||||
> `felhom.eu/documentation/controller/`. Match the current code, not summaries, if they drift.
|
||||
The **in-guest controller** — one per customer LXC, Docker-only, **holds NO Proxmox credentials**. It
|
||||
owns the app domain: stack/deploy management, the Hungarian web UI, app-data backup, metrics,
|
||||
integrations, git-sync, notifications. Disk/host/Proxmox concerns are delegated to the host agent via
|
||||
`internal/agentapi`. Whole-guest backup (PBS vzdump) is the agent's, not ours.
|
||||
|
||||
**Don't confuse the two ex-"controllers":** `felhom-agent` (host, operator-tier, was
|
||||
`proxmox-controller`) vs this `felhom-controller` (in-guest, was `deploy-felhom-compose`).
|
||||
`proxmox-controller`) vs this repo (in-guest, was `deploy-felhom-compose`).
|
||||
|
||||
## Layout (verified against the tree)
|
||||
## Doing X → read Y
|
||||
|
||||
```
|
||||
controller/cmd/controller/ entry point + startup wiring (scheduler block, init-only setters)
|
||||
controller/internal/
|
||||
agentapi/ pinned-TLS client to the host agent's per-guest local API (THE disk seam)
|
||||
api/ REST /api/* router (writeJSON envelope, limitBody, config writes)
|
||||
appbackup/ felhom-data paths/namespaces, DB dumps, userdata skeleton (shared primitives)
|
||||
appexport/ .fab export/import bundles (password crypto, strict segment validation)
|
||||
assets/ app logo/screenshot sync from the hub
|
||||
backup/ app-data backup manager, recovery units, tier-2 copies, offbox restic
|
||||
bootstrap/ bootstrap.json ingest → controller.yaml (Day-0 + refresh)
|
||||
channelhealth/ agent-channel health checker (debounce + born-down alerting)
|
||||
cloudflare/ geo-enforcement remnant (agent-delegated)
|
||||
config/ controller.yaml load/validate (LoadPermissive = setup-mode only)
|
||||
crypto/ AES-256-GCM app.yaml secret encryption (ENC: prefix)
|
||||
infra/ traefik/cloudflared/filebrowser base-stack templates
|
||||
integrations/ app-to-app integrations (e.g. OnlyOffice)
|
||||
mailrelay/ app-email SMTP shim → hub relay
|
||||
metrics/ telemetry collection
|
||||
monitor/ health checks, protected containers
|
||||
notify/ hub event push (typed Notify* wrappers)
|
||||
quiesce/ quiesce loop for whole-guest backup (marker + recover)
|
||||
recovery/ recovery-unit restore
|
||||
report/ hub report builder/pusher + pull-based config refresh
|
||||
scheduler/ background jobs (Every/Daily, Budapest DST-safe)
|
||||
selftest/ startup self-checks
|
||||
selfupdate/ controller image self-update via the agent swap
|
||||
settings/ settings.json persistence (registry, flags, corruption recovery)
|
||||
setup/ first-boot setup wizard (own CSRF)
|
||||
stacks/ compose ops: deploy/delete/migrate/state (THE app domain core)
|
||||
sync/ git-sync of the app catalog
|
||||
system/ mounts/probes (linux + permissive _other stubs)
|
||||
util/ small shared helpers
|
||||
web/ dashboard UI: server, auth/CSRF, handlers, funcmap, templates (Hungarian)
|
||||
```
|
||||
| Doing | Read |
|
||||
|---|---|
|
||||
| writing any new code | `REUSE.md` — canonical helpers, patterns, traps, seams |
|
||||
| needing current state / roadmap | `CONTEXT.md` |
|
||||
| needing a feature or architecture reference | `controller/README.md` |
|
||||
| build, deploy, publish, verify a version | the **`felhom-build-deploy`** skill |
|
||||
| writing or reviewing a test, fixing a bug | the **`felhom-testing`** skill |
|
||||
| UI, tokens, badges, Hungarian copy | the **`felhom-ui-design`** skill |
|
||||
| which box may I break | `felhom.eu/documentation/runbooks/target-selection.md` |
|
||||
| host addresses, break-glass, node facts | `felhom.eu/documentation/operations/nodes.md` |
|
||||
| what version is live anywhere | ask the hub (`/hosts`, `/configs`) or the box — **never a doc** |
|
||||
| the authoritative design | `felhom.eu/documentation/architecture/01/02/03-*.md` |
|
||||
|
||||
Per-package helpers/seams/traps: **`REUSE.md`** (maintained same-commit as helper changes).
|
||||
## Session-critical invariants
|
||||
|
||||
## Conventions & cardinal rules
|
||||
|
||||
- **Trunk-based — no branches.** All shippable work commits directly to `main`; `main` equals what is
|
||||
deployed. Report-only artifacts → `felhom.eu/documentation/` (`audits/`, `backlog/`). Risky fixes
|
||||
are implemented during the supervised session itself, on `main`; if a fix can't be verified/shipped,
|
||||
revert + report — never park on a branch.
|
||||
- Code quality: double-check for bugs/edge cases; add debug logging; **ask rather than guess**.
|
||||
- All UI text is Hungarian (Budapest timezone). Design tokens/gates: use the `felhom-ui-design`
|
||||
skill; templates must pass `controller/scripts/template_id_gate.py` + `emoji_gate.py`.
|
||||
- Testing doctrine (non-hollow tests, red-proofs, seams): use the `felhom-testing` skill.
|
||||
- **Logging**: new leveled lines use `internal/logx` (DEBUG always reaches the debug ring; stdout
|
||||
respects `logging.level`); English, keys-never-values, durations on outcomes — full rules in
|
||||
`felhom.eu/documentation/runbooks/logging-conventions.md`.
|
||||
- Update `REUSE.md` if you added/changed/deprecated a shared helper or pattern (same commit).
|
||||
- **Coupled features** (controller behavior that depends on a specific agent version): add a
|
||||
`featureProbes` table row in `internal/agentapi/features.go` + a `Supports` gate call at the
|
||||
feature's entry point; declare `MinAgent: X.Y.Z` in the CHANGELOG entry header. Rules:
|
||||
`felhom.eu/documentation/runbooks/publish-train-rules.md`.
|
||||
|
||||
> **In every repository where you make a change, update both files in that repo:**
|
||||
> - **`CHANGELOG.md`** — cumulative log, newest on top.
|
||||
> - **`REPORT.md`** — **overwrite** with the most recent implementation/validation summary only.
|
||||
>
|
||||
> **Never write secrets** into any committed file — reference them as "stored out-of-band".
|
||||
|
||||
## Live validation
|
||||
|
||||
Exercise the SERVER-SIDE PIPELINE a real user triggers, end-to-end (connect → enroll → deploy). The
|
||||
forbidden shortcut is BYPASSING that pipeline (the F9 episode: raw agent guest-attach + hand-set
|
||||
state). Invoking the exact endpoint the UI invokes is an acceptable proxy when a browser tool isn't
|
||||
available — no server logic is skipped, only rendering; say which method was used. For strict
|
||||
end-to-end UI coverage use claude-in-chrome (attaches only to sessions started AFTER the bridge
|
||||
connected) or a manual click-through.
|
||||
|
||||
## Environment & access
|
||||
|
||||
Claude Code runs on Windows 11; repos in `E:\git\` (`/e/git/` in Git Bash). All repos hosted at
|
||||
`gitea.dooplex.hu/admin/`. **SSH binary MUST be** `SSH=/c/Windows/System32/OpenSSH/ssh.exe`
|
||||
(Git Bash's ssh lacks the Windows agent — fails silently).
|
||||
|
||||
| Host | Access | Role |
|
||||
|------|--------|------|
|
||||
| Build server (k3s) | `$SSH kisfenyo@192.168.0.180` | build + push images (`/mnt/5_hdd/felhom.eu/build/felhom-controller` — all felhom dirs moved off the SSD to `/mnt/5_hdd/felhom.eu/` 2026-07-18) |
|
||||
| Demo Proxmox host `demo-felhom` | `$SSH felhom-pve` (root@192.168.0.162) | `pct` into guests; live validation |
|
||||
| Demo guest 9201 | `pct exec 9201 -- ...` on felhom-pve | the live demo controller (golden/bootstrap-managed) |
|
||||
| felhotest (legacy) | `$SSH -p 33022 kisfenyo@router.abonet.hu` | OLD /opt/docker compose mechanism |
|
||||
|
||||
External access via Cloudflare Tunnel → Traefik; Pi-hole forwards `*.demo-felhom.eu` → .162 locally.
|
||||
|
||||
## Build & deploy — MANDATORY after code changes
|
||||
|
||||
**Full runbook: use the `felhom-build-deploy` skill.** Summary (guest 9201 is bootstrap-managed —
|
||||
**no compose file**; `felhom-controller-bootstrap.service` runs the tag in `/etc/felhom-controller-image`):
|
||||
|
||||
| Step | Command |
|
||||
|------|---------|
|
||||
| 1. Commit + push | `git add -A && git commit -m "..." && git push` |
|
||||
| 2. Build + push image | `$SSH kisfenyo@192.168.0.180 "cd /mnt/5_hdd/felhom.eu/build/felhom-controller && git -C /mnt/5_hdd/felhom.eu/git/felhom-controller pull && ./build.sh <VER> --push"` (build.sh does NOT pull — the explicit pull is load-bearing) |
|
||||
| 3. Deploy (9201) | `$SSH felhom-pve "pct exec 9201 -- bash -c 'docker pull gitea.dooplex.hu/admin/felhom-controller:<VER> && echo gitea.dooplex.hu/admin/felhom-controller:<VER> > /etc/felhom-controller-image && systemctl restart felhom-controller-bootstrap.service'"` |
|
||||
| 4. Verify | `$SSH felhom-pve "pct exec 9201 -- docker ps --filter name=felhom-controller --format '{{.Image}} {{.Status}}'"` + container logs |
|
||||
|
||||
Hub build/deploy lives in `felhom.eu` (GitOps) — see that repo's CLAUDE.md / the skill. Catalog
|
||||
changes (`app-catalog-felhom.eu`): commit+push; controller sync picks them up ≤15 min or via the
|
||||
"Sablonok frissítése" button.
|
||||
|
||||
## Session-critical invariants (the rest live in REUSE.md)
|
||||
The rest live in `REUSE.md`. These cost incidents to learn:
|
||||
|
||||
- `docker compose restart` does NOT pick up new images/env — always `up -d` (`RedeployFromEnv`).
|
||||
- Docker's `.State` says "running" even for unhealthy containers — `.Status` parse is the truth.
|
||||
- In-memory `Deployed` flag is set BEFORE `compose up -d` (slow-pull race); reverted on failure.
|
||||
- `compose up -d` exits 0 on crash-loops — post-start status check is the detection.
|
||||
- Docker's `.State` says "running" even for unhealthy containers — the `.Status` parse is the truth.
|
||||
- In-memory `Deployed` is set BEFORE `compose up -d` (slow-pull race); reverted on failure.
|
||||
- `compose up -d` exits 0 on crash-loops — the post-start status check is the detection.
|
||||
- Env var KEYS are logged, never values. Protected stacks (traefik, cloudflared, felhom-controller)
|
||||
can't be stopped from the UI.
|
||||
cannot be stopped from the UI.
|
||||
- Verify a container image HAS the healthcheck tool before using it (BusyBox wget / python3 / curl —
|
||||
catalog REUSE.md maps the families).
|
||||
the catalog `REUSE.md` maps the families).
|
||||
- `IsRunning()` is CONCURRENCY, false during a verification restore — display MUST use
|
||||
`RestoreStatus()`.
|
||||
|
||||
## Live validation — the fence
|
||||
|
||||
Exercise the SERVER-SIDE PIPELINE a real user triggers, end-to-end (connect → enroll → deploy). **The
|
||||
forbidden shortcut is BYPASSING that pipeline** — the F9 episode was a raw agent guest-attach with
|
||||
hand-set state, and it proved nothing.
|
||||
|
||||
`claude-in-chrome` is NOT available on DooPlex. The standard method is endpoint-level: invoke the
|
||||
exact endpoint the UI invokes (no server logic is skipped, only rendering) and **say which method was
|
||||
used**. Strict end-to-end UI coverage is a manual click-through by the operator.
|
||||
|
||||
Two traps in that method live in `.claude/rules/ui-hungarian.md` (ASCII-only greps; `!` in
|
||||
credentials) — they load when you touch a template or stylesheet.
|
||||
|
||||
## Commands — one per surface
|
||||
|
||||
| Surface | Command |
|
||||
|---|---|
|
||||
| Gates (after ANY change) | `python3 controller/scripts/controller_gates.py` — from `controller/` |
|
||||
| Green gate | `go build ./... && go vet ./... && go test ./...` |
|
||||
| Build + deploy | the **`felhom-build-deploy`** skill — do not hand-roll it |
|
||||
|
||||
Guest 9201 is **bootstrap-managed — there is no compose file**;
|
||||
`felhom-controller-bootstrap.service` runs the tag written in `/etc/felhom-controller-image`. Catalog
|
||||
changes (`app-catalog-felhom.eu`) are picked up by controller sync ≤15 min, or via the "Sablonok
|
||||
frissítése" button.
|
||||
|
||||
## Working with CHANGELOG.md
|
||||
|
||||
**DO NOT read the full file** — it is large and will waste context.
|
||||
- Session start: use `CONTEXT.md` + `controller/README.md` for current state.
|
||||
|
||||
- Session start: `CONTEXT.md` + `controller/README.md` for current state.
|
||||
- Adding an entry: Read only the top ~30 lines for format, then Edit-insert after line 1.
|
||||
- History: Grep for topics instead of reading.
|
||||
|
||||
## End-of-session checklist
|
||||
|
||||
1. **Commit and push** all code changes
|
||||
2. **Build, push, and deploy** the new controller image (if controller code changed)
|
||||
3. **Update CHANGELOG.md** with what was done
|
||||
4. **Update CONTEXT.md** with decisions made, state and what's next
|
||||
5. **Update controller/README.md** if architecture or features changed
|
||||
6. **Verify** the deployment is working (check `docker ps` and logs)
|
||||
7. **Update REUSE.md** if you added/changed/deprecated a shared helper or pattern (same commit)
|
||||
1. **Commit and push** all code changes (explicit paths; no `git add -A`).
|
||||
2. **Build, push, and deploy** the new controller image, if controller code changed.
|
||||
3. **`CHANGELOG.md`** — always, whenever code changed and was pushed.
|
||||
4. **`CONTEXT.md`** — decisions made, state, what is next.
|
||||
5. **`controller/README.md`** — whenever a feature was added, modified or removed.
|
||||
6. **`REPORT.md`** — overwrite with this run's summary only.
|
||||
7. **`REUSE.md`** — if a shared helper or pattern was added/changed/deprecated (same commit).
|
||||
8. **Verify** the deployment (`docker ps` + logs).
|
||||
|
||||
<!--
|
||||
WHY THIS FILE IS SHORT (2026-08-06, instruction-trim task).
|
||||
Removed from here and rehomed, not lost:
|
||||
- the `## Layout (verified against the tree)` block -> derivable by `ls internal/`; REUSE.md
|
||||
carries the per-package seams and traps that the annotations were really for.
|
||||
- the `!!! IMPORTANT !!!` header -> its two requirements are checklist items 3 and 5. One voice,
|
||||
one place; a rule stated twice in one file is a rule that gets edited in one of them.
|
||||
- the host/access table -> documentation/operations/nodes.md is the single home. The copy here
|
||||
had drifted: it gave demo-felhom as plain root@192.168.0.162 (the LAN fallback, not the route),
|
||||
pinned "agent 0.93.0" against the project's own no-versions-in-docs rule, and claimed no drill
|
||||
VM was provisioned on demo-hp. Measured 2026-08-06: `qm list` on demo-hp shows VM 300
|
||||
`drill-r50` present. felhom-agent/CLAUDE.md was right; this file was wrong.
|
||||
- the "felhom-pve is back on the home LAN" block -> it was bookkeeping about a retired block; the
|
||||
record is in documentation/audits/AUDIT-vacation-remote-ops-2026-07-20.md.
|
||||
- the "Legacy: Windows workstation" block -> the workspace-root CLAUDE.md carries the full version.
|
||||
- the gates/logging/coupling/UI paragraphs -> .claude/rules/*.md, which load when a matching file
|
||||
is read instead of in every session.
|
||||
Full per-block accounting: felhom.eu/documentation/audits/LEDGER-instruction-trim-2026-08-06.md
|
||||
-->
|
||||
|
||||
+1111
-1
File diff suppressed because it is too large
Load Diff
@@ -1,182 +1,441 @@
|
||||
# REPORT — most recent implementation
|
||||
# REPORT — controller v0.215.0 → v0.216.0: disk-health severity ladder, escalation, and the alert that never sent
|
||||
|
||||
## felhom-controller v0.144.0 — „Megosztás": LAN SMB file sharing (R-7 slice 1) — 2026-07-18
|
||||
**Date:** 2026-08-14 · **Task class:** Implementation · **Repos touched:** `felhom-controller` (code),
|
||||
`felhom.eu` (documentation only — no hub code, no manifest bump, no ArgoCD sync)
|
||||
|
||||
**Class:** Implementation. **Shipped:** controller **v0.144.0** + new infra image
|
||||
**`gitea.dooplex.hu/admin/felhom-samba:1.0.0`**. Deployed + live-validated on demo guest 9201.
|
||||
---
|
||||
|
||||
### 1. Baselines
|
||||
## 1. Confirmed baselines used (as read at the start of the run)
|
||||
|
||||
| Repo | start `main` @ | Version |
|
||||
|---|---|---|
|
||||
| felhom-controller | `ac13966` (v0.143.0) | → **v0.144.0** |
|
||||
| felhom.eu | `7e0370f` | docs only (hub v0.66.0 / scripts v1.20.0 untouched) |
|
||||
| felhom-agent | `f222a7b` | NOT touched (v0.90.0) |
|
||||
| app-catalog-felhom.eu | — | NOT touched (deliberately not a catalog app) |
|
||||
| Repo | `main` @ start | Version | → Shipped |
|
||||
|------|----------------|---------|-----------|
|
||||
| felhom-controller | `3e3ee94b7bbe6b66663c468e22aa86616365a45a` | v0.214.0 | **v0.215.0**, then **v0.216.0** (a defect found live in v0.215.0 — §14) |
|
||||
| felhom.eu | `e0b56c976f8e4a7352309d78754fec448dd55f99` | n/a (docs only) | n/a |
|
||||
|
||||
*(The brief listed `a4a7de3d` for the controller; the live baseline was `ac13966` — the delta is the
|
||||
2026-07-18 build-root relocation, paths/docs only.)*
|
||||
Both trees verified clean (`git status --porcelain` empty, `HEAD == origin/main`) before any build.
|
||||
The controller hash matched the spec's stated baseline exactly. `MinAgent` stays **0.129.0** — no
|
||||
agent change; every field read here has been on the wire since agent v0.94.0/v0.95.0.
|
||||
|
||||
### 2. Files created / modified
|
||||
---
|
||||
|
||||
**Created:** `controller/infra-images/samba/{Dockerfile,entrypoint.sh}`,
|
||||
`controller/scripts/build-samba-image.sh`, `internal/settings/smb.go` (+test),
|
||||
`internal/infra/samba.go` (+test), `internal/stacks/samba.go` (+test),
|
||||
`internal/stacks/samba_classify.go` (+test), `internal/web/sharing_handlers.go` (+test),
|
||||
`internal/web/templates/sharing.html`.
|
||||
## 2. Files created / modified
|
||||
|
||||
**Modified:** `settings/settings.go` (2 fields), `infra/infra.go` (`SambaImage` pin),
|
||||
`stacks/infra.go`, `stacks/manager.go` (3 test seams), `stacks/metadata.go` (samba branch),
|
||||
`config/config.go` (`alwaysProtectedStacks`), `web/server.go`, `web/inframeta.go`,
|
||||
`cmd/controller/main.go` (`/api/sharing/` mux), `templates/layout.html`, `templates/icons.html`,
|
||||
`CHANGELOG.md`.
|
||||
**felhom-controller**
|
||||
- `controller/internal/agentapi/diskverdict.go` — modified (14-row ladder, `DiskPrior`, `UncorrectableSectors`, `TemperatureFailC`)
|
||||
- `controller/internal/agentapi/diskverdict_test.go` — modified
|
||||
- `controller/internal/agentapi/diskverdict_ladder_test.go` — **created**
|
||||
- `controller/internal/notify/notifier.go` — modified (severity, `DiskAlert`, `DiskAlertKind`, `Severity()`, 5 message shapes)
|
||||
- `controller/internal/notify/disk_health_test.go` — rewritten
|
||||
- `controller/internal/web/disk_health_state.go` — **created** (persistence + decision)
|
||||
- `controller/internal/web/disk_health.go` — modified
|
||||
- `controller/internal/web/disk_health_test.go` — rewritten
|
||||
- `controller/internal/web/server.go` — modified (seam signature)
|
||||
- `controller/cmd/controller/main.go` — modified (6h → 1h)
|
||||
- *(v0.216.0)* `controller/internal/web/disk_health.go` + `disk_health_test.go` — the R-335 dedup fix and its test
|
||||
- `CHANGELOG.md`, `CONTEXT.md`, `REUSE.md`, `controller/README.md`, `REPORT.md`
|
||||
|
||||
### 3. Commits
|
||||
**felhom.eu** (documentation only)
|
||||
- `documentation/audits/DIAG-smart-passed-trap-2026-08-14.md` — **created**
|
||||
- `documentation/audits/fixtures/smart-ST3000VX010-failing-2026-08-14.json` — **created** (raw `smartctl -a -j`, verbatim)
|
||||
- `documentation/audits/fixtures/smartd-history-sdg-2026-08-14.txt` — **created** (406 `smartd` journal lines)
|
||||
- `documentation/architecture/00-capability-map.md`, `documentation/backlog/ROADMAP.md`, `documentation/backlog/OPEN-ITEMS.md` — modified
|
||||
|
||||
| Hash | Phase |
|
||||
|---|---|
|
||||
| `f42f3e0` | Part 0 — felhom-samba image + build helper |
|
||||
| `b0c5ef4` | Part 1 — settings registry + renderers |
|
||||
| `0dcbea9` | Part 2 — lifecycle (ensure/reconcile/password/disable) + protection |
|
||||
| `1d26a69` | Part 4 — backup classification from the shares registry |
|
||||
| `4f08e5e` | Part 3 — „Megosztás" page + guarded picker |
|
||||
| `2eef9b2` | fix: `/api/sharing/` mux registration (found by live validation) |
|
||||
| `b409f5e` | fix: storage-ROOT vs share-TARGET validation split (found by live validation) |
|
||||
---
|
||||
|
||||
### 4. Part-4 Step-1 enumeration finding (verbatim, as required)
|
||||
## 3. Commits pushed to `main`
|
||||
|
||||
`GetStackClassifiedBinds` is **NOT inert** — it is consumed by `backup/offbox_capture.go:36`
|
||||
(offsite), `backup/tier2_capture.go:45` (tier-2), `appexport/fabplan.go:33,115` (.fab). The
|
||||
enumeration nevertheless never reaches a share-only stack:
|
||||
| Repo | Hash | What |
|
||||
|------|------|------|
|
||||
| felhom.eu | `848de81` | Part 0 — fixtures + findings doc |
|
||||
| felhom-controller | `bb50e12` | Parts 1–3 — ladder, severity, persisted state + tests |
|
||||
| felhom-controller | `c24f192` | Group L strengthened to two post-restart checks |
|
||||
| felhom-controller | `34d83f5` | Part 4 — cadence 6h → 1h (measured) |
|
||||
| felhom-controller | `8144a70` | Part 5 — CHANGELOG / CONTEXT / README / REUSE |
|
||||
| felhom.eu | `767960b` | Part 5 — capability map, ROADMAP, register rows R-328…R-334 |
|
||||
| felhom-controller | `90f2545` | **v0.216.0** — R-335, one physical disk evaluated once per run |
|
||||
| felhom.eu | `fa4748d` | R-335 register row |
|
||||
|
||||
- **Tier-2** — `RunAllTier2` (`tier2.go:399`) iterates `ListDeployedStacks()` and does **not** skip
|
||||
protected stacks. But `RunTier2` (`tier2.go:269`) resolves `GetAppDrivePath(stack)` then
|
||||
`os.Stat(RecoveryUnitPath(...))` and **returns nil before `tier2CaptureSet` is ever called** when no
|
||||
recovery unit exists. samba has neither an app drive path nor a recovery unit.
|
||||
- **Offsite** — `offbox.go:600` enumerates `settings.GetOffboxApps()` (apps with
|
||||
`AppBackup[name].Offbox`); samba is not an app-backup app.
|
||||
- **Volume dumps** — `backup.go:437–439` iterates `ListDeployedStacks()` and `continue`s on
|
||||
`cfg.IsProtectedStack(name)`; with samba protected this correctly skips it.
|
||||
---
|
||||
|
||||
**Verdict — design fork STOPPED and reported, not improvised** (per the task's Part-4 clause): putting
|
||||
share data into a live tier-2/offsite run requires teaching both engines about a stack with **no
|
||||
recovery unit and no single app drive path** — a structural change inside the engines, far beyond an
|
||||
enumeration tweak. Slice 1 therefore delivers the **correct classification seam** only.
|
||||
## 4. Per-test results — all twelve groups
|
||||
|
||||
**Consequence, stated plainly:** [R4]'s intent — a customer dropping family files onto `\\FELHOM` gets
|
||||
the product backup promise — is **classified but not executed**. Share data is in no live backup run
|
||||
today. Needs a Viktor ruling (suggested new item **R-7b**).
|
||||
| Group | Scenario | Test | Result |
|
||||
|-------|----------|------|--------|
|
||||
| A | real drive, 2nd observation | `TestDiskCheck_RealDrive_HibaAndOneCriticalEvent` + `TestLadder_RealDrive_ReachesHiba` | **PASS** |
|
||||
| B | the transient that cleared | `TestDiskCheck_FirstSightingIsWarnOnly` | **PASS** |
|
||||
| C | sustained → Hiba | `TestDiskLadder_SustainDrivesTheEscalation` + `TestLadder_SustainIsWhatFires` | **PASS** |
|
||||
| D | recovered, silent | `TestDiskCheck_RecoveryIsSilentAndClearsState` | **PASS** |
|
||||
| E | flap damping | `TestDiskCheck_FlapDamping` | **PASS** |
|
||||
| F | escalation beats damping | `TestDiskCheck_EscalationBeatsDamping` | **PASS** |
|
||||
| G | still getting worse | `TestDiskCheck_RealertWhenStillWorsening` | **PASS** |
|
||||
| H | below both bars | `TestDiskCheck_NoRealertBelowBothBars` | **PASS** |
|
||||
| I | heat | `TestLadder_Temperature` + `TestDiskCheck_TemperatureShape` | **PASS** |
|
||||
| J | no data never alarms | `TestDiskCheck_UnknownNeverAlarmsNorErasesPrior` + `TestLadder_UnknownNeverAlarms` | **PASS** |
|
||||
| K | severity routes | `TestNotifyDiskHealthDegraded_SeverityRoutes` | **PASS** |
|
||||
| L | state survives restart (seam) | `TestDiskCheck_StateSurvivesRestart_ProductionPath` | **PASS** |
|
||||
|
||||
### 5. Tests
|
||||
Supporting: `TestLadder_CountBackstopBoundary`, `TestLadder_ZeroPriorIsFailSafe`,
|
||||
`TestUncorrectableSectors`, `TestDegradedAttributes_NamesFailCounters`,
|
||||
`TestNotifyDiskHealthDegraded_{WarnShape,FailShapes,CopyDiscipline}`,
|
||||
`TestDiskAlertDecision_Table`, `TestDiskState_CorruptFileFallsBackToNoPrior`,
|
||||
`TestDiskCheck_DisappearedDiskIsForgotten`, `TestDiskCheck_UnreachableAgentIsInert` — all PASS.
|
||||
|
||||
Test functions **651 → 669** (+18). Green gate (`go build ./... && go vet ./... && go test ./...`) run
|
||||
after every phase: **23/23 packages ok**, zero failures. Template gates green
|
||||
(`template_id_gate.py`, `emoji_gate.py`).
|
||||
---
|
||||
|
||||
| Scenario | Tests | Result |
|
||||
|---|---|---|
|
||||
| A — enable + first share | `TestSambaReconcile_HappyPath`, `TestRenderSambaConfig_Golden`, settings CRUD | PASS |
|
||||
| B — read-only share | `TestRenderSambaCompose` (`read only = yes` + `:ro` bind) | PASS |
|
||||
| C — security gates | `TestSharingResolvePath_{GuardMatrix,SymlinkEscapeRefused,DecommissionedRootRefused,UniformRefusal}`, `TestSharingResolveStorageRoot`, `TestPathWithin_*`, name validation | PASS (symlink case SKIPs on Windows — no privilege) |
|
||||
| D — classification | `TestSambaClassifiedBinds_{TierMembership,ExcludesConfigMounts,NoShares}` | PASS |
|
||||
| E — delete/disable keep data | `TestSambaShareDeleteAndDisableKeepData` (byte-identical tree) | PASS |
|
||||
| Idempotency | `TestSambaReconcile_IdempotentNoComposeCall` (seam asserts 0 compose calls) | PASS |
|
||||
| Secrets | `TestSambaPasswordNeverPersisted`, `TestSambaPasswordRefusedWhenDisabled` | PASS |
|
||||
## 5. Red-proof outcomes — all twelve, individually
|
||||
|
||||
#### Red-proofs — all four run, confirmed, restored (`git diff` clean each time)
|
||||
Each mutation was applied by script, **asserted present in the source before the run** (the harness
|
||||
aborts with `MUTATION-NOT-APPLIED` if the target text is absent), the named test run, and the file
|
||||
reverted with `git checkout --`. The tree was confirmed clean after the sweep.
|
||||
|
||||
1. **Classification** — inverted Offsite→mandatory → `TestSambaClassifiedBinds_TierMembership` FAILED:
|
||||
`Felhőmentés ON must be mandatory, got "optional"` + `mandatory share must be in the OFFSITE set: []`.
|
||||
2. **`:ro` bind** — dropped the read-only bind mapping → `TestRenderSambaCompose` FAILED:
|
||||
`read-only share must get a :ro bind, missing "- …/filmek:…/filmek:ro"`.
|
||||
3. **Picker guard** — removed the containment assert in `sharingOwningRoot` →
|
||||
`TestSharingResolvePath_GuardMatrix` FAILED:
|
||||
`outside every registered root: must be refused, got "…\Temp\…\002"`.
|
||||
4. **Idempotency** — removed config change-detection → `TestSambaReconcile_IdempotentNoComposeCall`
|
||||
FAILED: `unchanged registry must perform NO compose call: calls went 1 → 2`.
|
||||
| # | Mutation applied | Target test | Outcome |
|
||||
|---|------------------|-------------|---------|
|
||||
| A | remove truth-table row 6 (the sustain rule) | `TestDiskCheck_RealDrive_HibaAndOneCriticalEvent` | **RED-PROOF PASSED — FINDING, see below** |
|
||||
| B | make row 9 return `Fail` | `TestDiskCheck_FirstSightingIsWarnOnly` | failed as required |
|
||||
| C | pass a zero `DiskPrior` in `RunDiskHealthCheck` | `TestDiskLadder_SustainDrivesTheEscalation` | failed as required |
|
||||
| D | let Rendben fall through the silence guard | `TestDiskCheck_RecoveryIsSilentAndClearsState` | failed as required |
|
||||
| E | compare against last **observed** verdict, not last **alerted** | `TestDiskCheck_FlapDamping` | failed as required |
|
||||
| F | let damping cover escalations | `TestDiskCheck_EscalationBeatsDamping` | failed as required |
|
||||
| G | remove the re-alert branch | `TestDiskCheck_RealertWhenStillWorsening` | failed as required |
|
||||
| H | make cooldown/doubling an **OR** instead of an AND | `TestDiskCheck_NoRealertBelowBothBars` | failed as required |
|
||||
| I | remove truth-table rows 3 **and** 13 | `TestLadder_Temperature`, `TestDiskCheck_TemperatureShape` | failed as required |
|
||||
| J | let UNKNOWN delete the prior record | `TestDiskCheck_UnknownNeverAlarmsNorErasesPrior` | failed as required |
|
||||
| K | restore `severity := "warn"` | `TestNotifyDiskHealthDegraded_SeverityRoutes` | failed as required |
|
||||
| L | skip loading the persisted state | `TestDiskCheck_StateSurvivesRestart_ProductionPath` | failed as required |
|
||||
|
||||
#### Password-leak check (rule 4)
|
||||
### A thirteenth red-proof, added after the deploy (R-335)
|
||||
|
||||
`TestSambaPasswordNeverPersisted` asserts the secret reaches the smbpasswd seam and appears in **none**
|
||||
of `settings.json`, `smb.conf`, `docker-compose.yml`. Live: the demo `settings.json` carries only
|
||||
`"user_set": true`. The password is STDIN-only (never argv) and never logged.
|
||||
| # | Mutation applied | Target test | Outcome |
|
||||
|---|------------------|-------------|---------|
|
||||
| M | delete the `if seen[key] { continue }` dedup guard | `TestDiskCheck_SameDiskTwiceIsEvaluatedOnce` | failed as required |
|
||||
|
||||
### 6. Deployment (live)
|
||||
Observed failure: `first sighting of an aliased disk must be silent, got 1: [{Label:felhom-backup … Kind:2 Sectors:8}]`
|
||||
— i.e. `Kind:2` is `DiskAlertFailSectors`, a **Hiba on a first sighting of 8 sectors**. Reverted.
|
||||
|
||||
### FINDING — red-proof A passed, and it is the spec's mutation that is at fault, not the code
|
||||
|
||||
The task specified group A's red-proof as *"remove truth-table row 6 → verdict is Warn"*. **That
|
||||
mutation cannot fail a test built on the real drive's values**: the real drive carries **352**
|
||||
unreadable sectors, so with row 6 deleted it still reaches Hiba via **row 8** (count ≥ 64). The test
|
||||
correctly stayed green, so the mutation proves nothing about row 6.
|
||||
|
||||
This was anticipated while writing the tests and is documented in the test's own comment rather than
|
||||
discovered afterwards. **Row 6 is genuinely pinned**, by two tests that hold the counters at **8**
|
||||
(far below the 64 backstop) and vary *only* the prior:
|
||||
|
||||
- `TestLadder_SustainIsWhatFires` (agentapi) — same `SmartSummary`, `DiskPrior{}` → Warn,
|
||||
`DiskPrior{SawUncorrectable:true}` → Fail.
|
||||
- `TestDiskLadder_SustainDrivesTheEscalation` (web) — the event-level twin.
|
||||
|
||||
Both were run under the row-6-deleted mutation and **both failed**, as recorded:
|
||||
`SAME 8 sectors, now sustained = 2 (Figyelmeztetés), want Fail/Hiba` and
|
||||
`severity = "warning", want critical` / `chip = "Figyelmeztetés", want Hiba`. So the invariant is
|
||||
covered; only the spec's chosen mutation was invalid.
|
||||
|
||||
### A second finding, from building red-proof L
|
||||
|
||||
The first version of the Group L seam test ran **one** check after the restart and **passed under the
|
||||
mutation** — because a controller that has forgotten its state is also silent on its first check. The
|
||||
test was strengthened to run **two** checks (commit `c24f192`), after which the mutation fails. This
|
||||
is the exact shape §10 warns about, caught by running the red-proof rather than assuming it.
|
||||
|
||||
---
|
||||
|
||||
## 6. Test count and suite state
|
||||
|
||||
- **Before:** 1391 test functions (at `3e3ee94`) · **After:** 1414 (+23)
|
||||
- `go build ./... && go vet ./... && go test ./...` — **all green**, no failures, no skips introduced.
|
||||
- `python3 controller/scripts/controller_gates.py` — **all 11 gates OK.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Cadence measurement (Part 4)
|
||||
|
||||
Measured on **demo-hp** (Tier 0, disposable), through `fetchDisks`' real path — the agent local API
|
||||
`GET /disks`, not the 60 s card cache. Ten consecutive calls, all **HTTP 200**:
|
||||
|
||||
```
|
||||
felhom-controller: gitea.dooplex.hu/admin/felhom-controller:0.144.0 Up (healthy)
|
||||
felhom-samba: gitea.dooplex.hu/admin/felhom-samba:1.0.0 Up
|
||||
0.840558 0.817000 0.815783 0.831992 0.809066
|
||||
0.832899 0.824788 0.804886 0.812539 0.833547 (seconds)
|
||||
```
|
||||
Anonymous pull of the samba image verified from guest 9201 without registry creds (Part-0 gate).
|
||||
|
||||
Binds (`docker inspect felhom-samba`) — Scenario B proof:
|
||||
| min | median | max | disk count |
|
||||
|-----|--------|-----|------------|
|
||||
| **0.804886 s** | **0.820894 s** | **0.840558 s** | **3 physical rows** across 2 devices (SanDisk X600 M.2 SATA SSD; Toshiba KXG50PNV1T02 NVMe, counted twice as `c11-scratch` + `felhom-backup`) |
|
||||
|
||||
**Branch taken: median < 5 s → `6*time.Hour` → `1*time.Hour`.** The median is ~6× under the bar. The
|
||||
detection argument is the real one: the observed benign excursion lasted about **one hour**, so a
|
||||
6-hourly sampler can land either side of it and then catch the terminal run half a day late.
|
||||
|
||||
No spin-up signature appeared in the timings (uniform ~0.82 s; demo-hp is all-flash), so the
|
||||
measurement did not suggest the spun-down-drive concern. That question is recorded as an Observation
|
||||
below and deliberately **not acted on**.
|
||||
|
||||
---
|
||||
|
||||
## 8. Live validation
|
||||
|
||||
Deployed to **demo-hp guest 9201** via the bootstrap path (`docker pull` →
|
||||
`/etc/felhom-controller-image` → `systemctl restart felhom-controller-bootstrap.service`).
|
||||
|
||||
```
|
||||
/opt/docker/stacks/samba/smb.conf -> /etc/samba/smb.conf rw=false
|
||||
/mnt/felhom-drives/hdd_1/media/filmek -> …/media/filmek rw=false (read-only share)
|
||||
/mnt/felhom-drives/hdd_1/shares/dokumentumok -> …/shares/dokumentumok rw=true
|
||||
samba-passdb volume -> /var/lib/samba rw=true
|
||||
gitea.dooplex.hu/admin/felhom-controller:0.215.0 Up 19 seconds (healthy) # 06:23Z
|
||||
gitea.dooplex.hu/admin/felhom-controller:0.216.0 Up 6 seconds (healthy) # after the R-335 fix
|
||||
```
|
||||
Listeners LAN-only (docker 172.x bridges absent — `bind interfaces only` working):
|
||||
`445` on `127.0.0.1` + `192.168.0.127`, `137/138` NetBIOS, `3702` WSD, `5357`.
|
||||
|
||||
### 7. Live validation
|
||||
### Leg 1 — no over-correction (the load-bearing check)
|
||||
|
||||
**Method:** the UI is password-gated and no browser leg was used, so validation drove the **exact
|
||||
endpoints the UI posts to** (curl with a real session cookie + `_csrf` token scraped from the rendered
|
||||
page) from inside guest 9201. No server logic skipped — only client rendering.
|
||||
Method: **endpoint-level** — authenticated `GET /dashboard` on the real controller
|
||||
(`https://felhom.enkisfelhom.hu/dashboard`, HTTP 200, 44 036 bytes), i.e. the exact endpoint the UI
|
||||
invokes; only rendering is skipped. No browser is available on DooPlex.
|
||||
|
||||
| Leg | Evidence |
|
||||
|---|---|
|
||||
| login + `GET /sharing` | 302 login; page renders (title + nav present) |
|
||||
| `POST /sharing/enable` | 303; `smb.enabled=true, server_name=FELHOM` |
|
||||
| `POST /sharing/password` | 303; `user_set=true`; log `household SMB password applied for user=felhom` |
|
||||
| `POST /sharing/shares` (new folder) | 303; `…/hdd_1/shares/dokumentumok` created **owned 1000:1000** |
|
||||
| `POST /sharing/shares` (existing, read-only) | 303; stored `read_only: true` |
|
||||
| Guard refusals (`/api/sharing/browse`) | `appdata`, `backups`, `/etc`, drive-root → **all 400, identical** `{"error":"Ez a mappa nem osztható meg."}` |
|
||||
| Guard accept | `…/hdd_1/media` → 200 with entries |
|
||||
| Generated `smb.conf` | hardened global verbatim; `[filmek] read only = yes`, `[dokumentumok] read only = no`, force-user block on both |
|
||||
| Windows 11 (192.168.0.110) | `Test-NetConnection 445` → **True** |
|
||||
| NetBIOS flat name (nmbd) | `nbtstat`: `FELHOM <00>/<03>/<20> Registered`; `ping FELHOM` → `Reply from 192.168.0.127` |
|
||||
| SMB round-trip (writable) | write → readback **BYTE-COMPARE: PASS** |
|
||||
| **Write-refused (read-only)** | write to `\\192.168.0.127\filmek` **rejected**; folder still empty on the box (non-effect proven) |
|
||||
| `force user` promise | SMB-written `r7.txt` owned **1000:1000** on the box |
|
||||
| WS-Discovery | ProbeMatch from `192.168.0.127` **PASS** |
|
||||
| **Explorer (human, Viktor)** | both shares open from the Network view; interactive Explorer **save** landed owned `1000:1000`; write into the read-only share **refused**, folder still empty |
|
||||
Card contents, parsed from the response body:
|
||||
|
||||
#### Two real bugs caught by live validation (fixed + regression-tested)
|
||||
| Disk | Chip | Class | Temp |
|
||||
|------|------|-------|------|
|
||||
| KXG50PNV1T02 NVMe TOSHIBA 1024GB | **Rendben** | `state-text-run` | 53 °C |
|
||||
| KXG50PNV1T02 NVMe TOSHIBA 1024GB | **Rendben** | `state-text-run` | 53 °C |
|
||||
| SanDisk X600 M.2 2280 SATA 128GB | **Rendben** | `state-text-run` | 44 °C |
|
||||
|
||||
1. **`/api/sharing/browse` unreachable** — the `/api/` subtree is routed on the main mux in `main.go`,
|
||||
so the case added to the web `ServeHTTP` switch was shadowed by the apiRouter catch-all and
|
||||
returned `401 authentication required`. Fixed by registering
|
||||
`mux.Handle("/api/sharing/", RequireAuth(CsrfProtect(ServeSharingAPI)))`, matching `/api/storage/`.
|
||||
2. **Share creation silently failed** — the "new folder" flow passed the storage ROOT through
|
||||
`sharingResolvePath`, which correctly refuses the drive root as a share TARGET. Split out
|
||||
`sharingResolveStorageRoot` (accepts exactly a registered live root — strictly tighter);
|
||||
`TestSharingResolveStorageRoot` asserts both halves.
|
||||
`Figyelmeztetés` = 0, `Hiba` = 0, `Nincs adat` = 0, `state-text-warn` = 0, `state-text-crit` = 0.
|
||||
**No healthy disk was over-corrected.**
|
||||
|
||||
### 8. NOT yet live-validated
|
||||
**Positive observable, at deploy:**
|
||||
`[INFO] [scheduler] Registered periodic job: disk-health-check (every 1h0m0s)` — the new cadence is
|
||||
in force, not merely compiled.
|
||||
|
||||
- ~~Explorer render~~ — **PASSED 2026-07-18 (Viktor).** Network → FELHOM → both shares open. He went
|
||||
further than the scripted leg: a real Explorer **save** into `dokumentumok` succeeded and landed on
|
||||
the box owned **1000:1000** (so `force user` holds for an interactive Explorer write, not just a
|
||||
scripted one), and a write into the read-only `filmek` was refused by Windows
|
||||
("Destination Folder Access Denied") with the folder left **empty** on disk. Slice 1 is fully
|
||||
PROVEN-LIVE; the capability-map row is flipped.
|
||||
- **Share data in a real backup run** — blocked by the §4 design fork (classified, not executed).
|
||||
- **Symlink-escape guard on Linux** — unit test SKIPs on Windows (no privilege).
|
||||
- Drive-disconnect behaviour live (unit-tested only); multi-root share sets; SMB throughput/signing.
|
||||
**Positive observable, per cycle** — two full hourly cycles observed after the deploy, from the
|
||||
container log:
|
||||
|
||||
### 9. Observations (not acted on)
|
||||
```
|
||||
2026/08/14 06:23:13 [INFO] [scheduler] Registered periodic job: disk-health-check (every 1h0m0s)
|
||||
2026/08/14 07:23:13 [INFO] [scheduler] Running job: disk-health-check
|
||||
2026/08/14 07:23:14 [INFO] [web] disk-health check complete: 3 disk(s) evaluated, 0 alert(s)
|
||||
2026/08/14 07:23:14 [INFO] [scheduler] Job disk-health-check completed (took 849ms)
|
||||
2026/08/14 08:23:13 [INFO] [scheduler] Running job: disk-health-check
|
||||
2026/08/14 08:23:14 [INFO] [web] disk-health check complete: 3 disk(s) evaluated, 0 alert(s)
|
||||
2026/08/14 08:23:14 [INFO] [scheduler] Job disk-health-check completed (took 843ms)
|
||||
```
|
||||
|
||||
- `monitor.EffectiveProtected` derives only from `cfg.Stacks.Protected` (golden-generated), so samba is
|
||||
**not** liveness-monitored — a dead samba container raises no "protected container missing" issue.
|
||||
Fixing it needs the golden controller.yaml to list samba, or the SMB-enabled flag threaded through 4
|
||||
`RunHealthCheck` call sites. Deferred deliberately; the shipped direction is the safe one (no false
|
||||
alarms while the feature is off).
|
||||
- A classified bind's `RelPath` is relative to its owning storage root; a share set spanning
|
||||
**multiple** roots has no single resolution root in the current `(Root, RelPath) + hddPath`
|
||||
vocabulary — part of the §4 fork.
|
||||
- `sambaWriteAtomic` is a fourth atomic-write helper (with `backup.atomicWrite`,
|
||||
`bootstrap.writeFileAtomic`, `setup.atomicWriteFile`) — a shared util is worth a cleanup pass.
|
||||
- A throwaway `r7.txt` from the round-trip test remains in the demo `dokumentumok` share (harmless;
|
||||
left so the Explorer leg has visible content).
|
||||
`grep -c disk_health_degraded` over the whole container log: **0**. Both cycles ran (849 ms / 843 ms,
|
||||
matching the §7 measurement), evaluated every disk, and emitted nothing. **Zero alerts from a check
|
||||
that demonstrably ran** — not silence.
|
||||
|
||||
Persisted state written by the first cycle (`/opt/docker/felhom-controller/data/disk-health-state.json`,
|
||||
428 bytes, on the `felhom-controller-data` docker volume, so it survives container recreation):
|
||||
|
||||
```json
|
||||
{"version": 1, "disks": {
|
||||
"path:/var/lib/vz": {"verdict": 1, "saw_uncorrectable": false, ...},
|
||||
"uuid:91d2dc2d-2d28-4929-9bdd-3e11fa2f41ae": {"verdict": 1, "saw_uncorrectable": false, ...}}}
|
||||
```
|
||||
|
||||
`verdict: 1` is `DiskVerdictOK` for both, `saw_uncorrectable: false`, never alerted.
|
||||
|
||||
**Reading those two artefacts against each other is what exposed R-335** — see §14.
|
||||
|
||||
### Leg 2 — the severity fix arrives (the point of the task)
|
||||
|
||||
Two synthetic `disk_health_degraded` events pushed for customer `demo-hp` **through the real hub
|
||||
event endpoint** (`POST https://hub.felhom.eu/api/v1/event`), from the guest's own controller using
|
||||
its own hub credentials — the genuine controller→hub path, not a hand-crafted operator call. Both
|
||||
returned `HTTP 200 {"ok":true}`. The hub DB was read with its `-wal` and `-shm` copied alongside
|
||||
`hub.db` (a `hub.db`-only read is stale).
|
||||
|
||||
**As STORED by the hub (`events`):**
|
||||
|
||||
| id | severity pushed | severity STORED |
|
||||
|----|-----------------|-----------------|
|
||||
| 2964 | `warning` | **`warning`** |
|
||||
| 2965 | `warn` | **`info`** ← coerced |
|
||||
|
||||
**`notification_log` rows for those two events:**
|
||||
|
||||
| id | event_type | severity | channel | status | error |
|
||||
|----|-----------|----------|---------|--------|-------|
|
||||
| 689 | `disk_health_degraded` | `warning` | `operator` | **`sent`** | *(none)* |
|
||||
| — | *(the `"warn"` push)* | — | — | **NO ROW EXISTS** | — |
|
||||
|
||||
**That pair is the proof.** The identical event, differing only in one word of the severity string,
|
||||
is the difference between *delivered to the operator* and *stored as an informational notice and
|
||||
delivered to nobody*. This is the first time this leg has been observed end to end.
|
||||
|
||||
Only the operator leg fired because **demo-hp has no `customer_notifications` row at all** (no
|
||||
customer email, no `enabled_events`), so no customer row was possible for either push — verified
|
||||
directly, not assumed. **One real email was sent to the operator**, as the task anticipated.
|
||||
|
||||
---
|
||||
|
||||
## 9. NOT yet live-validated — stated explicitly
|
||||
|
||||
**The Fail-from-counters path has never fired on real hardware.** Everything in §4/§5 exercises it
|
||||
against the committed fixture's values in unit tests only. The live legs above prove the *negative*
|
||||
(no false alert on three healthy disks) and the *severity wire* (end to end, through the hub) — they
|
||||
do **not** prove a live disk reaching Hiba. The fixture tests must not be read as a live proof.
|
||||
|
||||
Tracked as **R-332 (WATCHING)**. Closing condition: a live disk reaching Hiba from counters, or a
|
||||
deliberate injection through the real pipeline (agent `/disks` → controller check → hub event) — not
|
||||
a hand-set verdict.
|
||||
|
||||
**One item originally listed here has since been proven live** and is no longer part of this gap: the
|
||||
**persisted state surviving a controller restart**. The v0.215.0 → v0.216.0 redeploy destroyed and
|
||||
rebuilt the container, and the new one read back a `changed_at` written by the previous version rather
|
||||
than re-baselining — see §14. What remains unproven is the stronger half: an already-**alerted** disk
|
||||
not re-alerting after a restart, which needs a disk that has actually alerted. The drive that produced the fixture lives in DooPlex, which is Tier 2 and never a
|
||||
drill target; the demo boxes are all-flash and healthy.
|
||||
|
||||
---
|
||||
|
||||
## 10. Teardown
|
||||
|
||||
**This run provisioned nothing** — no VM, no guest, no hub customer record, no storage. Nothing was
|
||||
formatted, mounted, unmounted, repaired or written on any monitored disk; the only write is the
|
||||
controller's own `disk-health-state.json` inside its data volume.
|
||||
|
||||
Disposition of what the run did create:
|
||||
|
||||
- **Two synthetic hub events (`events` id 2964, 2965) and one `notification_log` row (id 689)** on the
|
||||
live hub. **Left in place deliberately.** Both messages are self-labelling
|
||||
(`"R-328 severity probe (…) - synthetic, no real disk fault"`), and deleting rows from the
|
||||
production hub DB is a riskier act than leaving two clearly-marked probe rows. Named here so they
|
||||
are not mistaken later for a real disk fault on demo-hp.
|
||||
- **One real operator email** resulting from row 689.
|
||||
- A local copy of `hub.db`/`-wal`/`-shm` in the session scratchpad only (not committed, not exported).
|
||||
|
||||
---
|
||||
|
||||
## 11. Register rows
|
||||
|
||||
| Row | State | Owner |
|
||||
|-----|-------|-------|
|
||||
| **R-328** — the severity drop: `"warn"` coerced to `info`, emailed to nobody | **CLOSED** (controller v0.215.0), proven live side by side | CC |
|
||||
| **R-329** — `app_start_failed` carries the identical defect | READY — **not fixed here**; needs a decision on whether it should notify at all | Viktor |
|
||||
| **R-330** — Phase 2: collect SMART attrs 187/199/188 + persist samples | READY — a declared wire change, hub models it in the same session under G-1 | CC |
|
||||
| **R-331** — Phase 3: growth-rate detection; revisit the static 64 | READY, blocked on R-330 | CC |
|
||||
| **R-332** — the Fail path has never fired on real hardware | **WATCHING** | CC |
|
||||
| **R-333** — NVMe temperature bands; agent `smartctl` has no `-n standby` | READY (S each) | Viktor decides (a); CC does (b) |
|
||||
| **R-334** — released with no golden carrying it (gate waiver) | READY — now applies to **v0.216.0** | CC bakes; **Viktor vouches** |
|
||||
| **R-335** — one physical disk walked twice per run, sustaining against itself | **CLOSED** (controller v0.216.0) | CC |
|
||||
|
||||
`smartd`-on-DooPlex-alerts-nobody is recorded in `DIAG-smart-passed-trap-2026-08-14.md` §8 as the
|
||||
same shape one layer out.
|
||||
|
||||
---
|
||||
|
||||
## 12. Observations — noticed, NOT acted on
|
||||
|
||||
1. **`app_start_failed` has the identical severity defect** (`notifier.go` ~L546, `"warn"`). Left
|
||||
untouched per scope. It needs a prior decision — should a stopped app email the customer at all? —
|
||||
because flipping the string alone converts a silent event into a mail flood on a crash-looping box.
|
||||
**R-329.**
|
||||
|
||||
2. **The 55/60 °C bands are spinning-disk bands being applied to NVMe, and this is close to biting.**
|
||||
Adopted unchanged from the operator's Prometheus config by explicit decision — but demo-hp's
|
||||
**healthy** Toshiba NVMe idles at **53 °C**, i.e. **2 °C below Figyelmeztetés and 7 °C below Hiba**,
|
||||
and NVMe routinely passes 60 °C under sustained write with no fault. As shipped, a healthy customer
|
||||
NVMe under load can be reported as **Hiba** — the single worst outcome this feature can produce, and
|
||||
the one leg 1 exists to guard. Not changed here because the threshold is a stated, settled operator
|
||||
decision; flagged rather than overridden. **R-333(a) — recommend splitting the bands by device
|
||||
class, or dropping them for NVMe and relying on `critical_warning`.**
|
||||
|
||||
3. **The agent runs bare `smartctl -a -j` with no `-n standby`**
|
||||
(`felhom-agent/internal/storage/hostops.go:368`), so every poll wakes a spun-down drive, and 6h → 1h
|
||||
multiplies that by six. Recorded, not acted on, per the task's instruction. demo-hp is all-flash so
|
||||
the measurement could not reveal it. Mitigating datum from the fixture: the failing drive logged
|
||||
only **3375 load cycles in 60505 power-on hours** (~one per 18 h), so this duty cycle barely spins
|
||||
down at all. **R-333(b).**
|
||||
|
||||
4. **`source ~/.config/credentials` prints two recovery codes to the terminal.** The file contains
|
||||
hyphenated keys (`R_DEMO-FELHOM`, `R_DEMO-HP`) that bash cannot assign, so sourcing it emits
|
||||
`command not found` errors **containing the secret values**. Anything that sources that file leaks
|
||||
them into logs, scrollback and transcripts. Not a code defect and out of scope; worth quoting
|
||||
values from it by other means, or renaming the keys.
|
||||
|
||||
5. **`golden_currency_gate.py` has no waiver parser.** Its own failure text says *"record a waiver in
|
||||
`OPEN-ITEMS.md` — never a bypass"*, but nothing reads such a waiver, so the only way past it is the
|
||||
bypass it warns against. See §13.
|
||||
|
||||
---
|
||||
|
||||
## 13. Deviations, stated plainly
|
||||
|
||||
- **`git push --no-verify` was used once**, on the `felhom.eu` docs push (`767960b`), and only there.
|
||||
Cause: `golden_currency_gate.py` correctly convicts the fact that controller **v0.215.0 is released
|
||||
and no golden carries it** (newest bake 0.214.0), so a *newly installed* machine would receive
|
||||
0.214.0 — without the severity fix. A golden bake was out of the task's scope, and its second half
|
||||
(vouching in the hub's day-0 artifact manifest) is operator-password-gated, so CC cannot complete it;
|
||||
a baked-but-unvouched golden is worse than none. Recorded as **R-334** with the bake+vouch owners
|
||||
named. CI re-runs the same entry point and will mail the operator. The running fleet is unaffected.
|
||||
- **One pre-existing test changed meaning by design:** `TestDiskVerdictFor`'s
|
||||
`critical_warning>0 → warn` case is now `→ fail` (truth-table row 4 — NVMe's own critical flag is a
|
||||
device declaration, not a drifting counter). `TestDiskHealthCheck_DegradationOnce` and its siblings
|
||||
were rewritten into the scenario groups because they encoded the pre-v0.215.0 single-alert behaviour
|
||||
the task deliberately replaces (Scenario C).
|
||||
|
||||
---
|
||||
|
||||
## 14. R-335 — a defect in v0.215.0, found live, fixed as v0.216.0
|
||||
|
||||
**How it was found.** Not by a test and not by review: by reading the release's own **positive
|
||||
observable** against the release's own **persisted artefact**. The hourly check logged *"3 disk(s)
|
||||
evaluated"*; `disk-health-state.json` held **two** records. Two artefacts that should have agreed did
|
||||
not.
|
||||
|
||||
**Cause.** demo-hp's `c11-scratch` and `felhom-backup` are the same physical NVMe (`/dev/nvme0n1`) and
|
||||
resolve to the same `diskKey`, so one disk was walked twice in a single run.
|
||||
|
||||
**Why it mattered.** `RunDiskHealthCheck` writes a disk's new record before the next entry reads it, so
|
||||
the **second** copy of an aliased disk consumed the **first** copy's write as its prior. The disk
|
||||
therefore **sustained against itself and reached Hiba on a first sighting** — defeating truth-table
|
||||
row 6, the single rule separating a one-hour benign excursion from a false critical alert — and would
|
||||
have emitted **two identical events** for one drive.
|
||||
|
||||
**Severity in practice: latent, not active.** Nothing fired on demo-hp because all three entries are
|
||||
healthy with zero counters. But any aliased disk developing one pending sector would have gone
|
||||
straight to Hiba, which is precisely the outcome §8 leg 1 exists to prevent. Aliasing is not exotic —
|
||||
it is the *normal* shape whenever a box has two PVE storage entries on one physical device.
|
||||
|
||||
**Fix (v0.216.0, `90f2545`).** Each `diskKey` is evaluated once per run. Both entries stay marked
|
||||
`seen`, so neither is mistaken for a disappeared disk, and the card still renders **both** storage
|
||||
rows — the dedup is about state and alerts, not display. Pinned by
|
||||
`TestDiskCheck_SameDiskTwiceIsEvaluatedOnce`, red-proof run and reverted (§5).
|
||||
|
||||
**Deployed:** `gitea.dooplex.hu/admin/felhom-controller:0.216.0 Up 6 seconds (healthy)`.
|
||||
|
||||
**Confirming cycle on v0.216.0 — CONFIRMED LIVE, 09:31:35Z:**
|
||||
|
||||
```
|
||||
live image: gitea.dooplex.hu/admin/felhom-controller:0.216.0 Up About an hour (healthy)
|
||||
2026/08/14 09:31:35 [INFO] [web] disk-health check complete: 2 disk(s) evaluated, 0 alert(s)
|
||||
grep -c disk_health_degraded: 0
|
||||
```
|
||||
|
||||
**`2 disk(s) evaluated` now matches the 2 persisted records.** The count and the artefact agree, which
|
||||
is the disagreement that exposed R-335 in the first place. Still zero alerts, still both card rows.
|
||||
|
||||
### The redeploy also proved persistence live — a gap §9 had listed as unproven
|
||||
|
||||
The 0.215.0 → 0.216.0 redeploy **replaced the container**, and the state file came back intact:
|
||||
|
||||
```json
|
||||
"path:/var/lib/vz": {"verdict": 1, "changed_at": "2026-08-14T07:23:14.640216851Z", ...}
|
||||
"uuid:91d2dc2d-…": {"verdict": 1, "changed_at": "2026-08-14T07:23:14.640216851Z", ...}
|
||||
```
|
||||
|
||||
That `changed_at` was written by **v0.215.0's first cycle at 07:23Z**, before the container was
|
||||
destroyed and rebuilt. The v0.216.0 container read it back and preserved it rather than stamping a
|
||||
fresh time — so the new container **loaded the pre-restart record instead of silently re-baselining**.
|
||||
That is Scenario L observed on real hardware, not just through the production-path unit test, and it
|
||||
is exactly the behaviour that was impossible before v0.215.0 (the baseline was in-memory).
|
||||
|
||||
It also incidentally confirms the unchanged-verdict path: `changed_at` is preserved across four checks
|
||||
and two controller versions because the verdict never changed, rather than being churned every cycle.
|
||||
|
||||
**What this still does NOT prove:** these disks are healthy and were never alerted, so the stronger
|
||||
half — *an already-ALERTED disk not re-alerting after a restart* — remains unit-tested only. R-332
|
||||
stands.
|
||||
|
||||
**Process note, recorded because it nearly cost the fix.** The red-proof harness reverts with
|
||||
`git checkout --`, which restores to `HEAD`. Running a red-proof against an **uncommitted** fix
|
||||
therefore *deletes the fix* along with the mutation — which happened here and was caught only by
|
||||
re-grepping the source afterwards. Commit the fix before red-proofing it, or snapshot outside git.
|
||||
|
||||
@@ -14,9 +14,14 @@
|
||||
| `PrimaryBackupPath` / `RecoveryUnitPath` / `RecoveryUnitComposePath` / `RecoveryUnitManifestPath` | controller/internal/appbackup/paths.go | `(nsRoot[, stackName]) string` | All backup dir layout | Take the NAMESPACE ROOT, not a bare drive path |
|
||||
| `AppDBDumpPath` / `AppVolumeDumpPath` / `AppDataDir` | controller/internal/appbackup/paths.go | `(nsRoot, stackName) string` | Per-app dump/data dirs | Same nsRoot contract. `AppDataDir`'s final segment is the app's real appdata dir NAME — NOT always the stack name (paperless-ngx → `paperless`); resolve via `AppDataDirNames` first (F-S2/F-S3) |
|
||||
| `AppDataDirNames` / `AppDataBindsPresent` | controller/internal/appbackup/paths.go | `(hddPath, stackName string, hddMounts []string) []string` / `(hddPath, hddMounts) bool` | Resolve the real `appdata/<name>` dir(s) from compose `${HDD_PATH}` binds (F-S2/F-S3) | `hddMounts` = ParseComposeHDDMounts shape. Deduped+sorted; falls back to `[stackName]` when no appdata bind. Tier-2 (`backup.Manager.tier2AppDataName`) refuses N>1; migrate (`stacks.Manager.ResolveAppDataDirNames`) loops N. `BindsPresent` drives the WARN-on-missing-declared-dir |
|
||||
| `UserdataDir` / `EnsureUserdataSkeleton` / `EnsureDirOwned` | controller/internal/appbackup/userdata.go | `(nsRoot)` / `(path, gid int)` | userdata/ tree w/ 2775 setgid gid-1000 convention | Linux-only chown via build-tag twin userdata_linux.go |
|
||||
| `UserdataDir` / `ImportDir` / `EnsureUserdataSkeleton` / `EnsureDirOwned` | controller/internal/appbackup/userdata.go | `(nsRoot)` / `(nsRoot)` / `(nsRoot, dirs []string)` / `(path, gid int)` | userdata/ tree w/ 2775 setgid gid-1000 convention. **R-75:** `ImportDir` is the CANONICAL drop-zone (`<nsRoot>/userdata/import`) and callers MUST resolve it against the SYSTEM namespace, never an app's HDD_PATH — use `stacks.Manager.GetImportRoot()`. `EnsureUserdataSkeleton` now takes the dir set: build it with `BuildUserdataSkeleton(DeriveUserdataDirs(stacksDir))`, or via `Manager.EnsureUserdataSkeleton` / `web.Server.ensureUserdataSkeleton`. | Linux-only chown via build-tag twin userdata_linux.go. **The set MUST stay sorted** — `fbNeedsRecreate` force-recreates FileBrowser on any byte diff and the naive map-order derivation measured 20/20 distinct (SPIKE P6). `UserdataSkeletonCarry()` is the old hardcoded list, retained forever so derivation can only ADD (zero removals). |
|
||||
| `BuildUserdataSkeleton` / `UserdataSkeletonCarry` / `DeriveUserdataDirs` | appbackup/userdata.go, stacks/skeleton_derive.go | `([]string)` / `()` / `(stacksDir)` | catalog-derived userdata skeleton (R-75) | Derives `${USERDATA_PATH}` binds only — `${IMPORT_PATH}` is NOT part of a drive skeleton (one root, system drive, `Manager.EnsureImportRoot`). Do NOT wire the catalog sync to `SyncFileBrowserMounts`. |
|
||||
| `appbackup.ValidateRelPath` / `ValidRoot` | controller/internal/appbackup/classify.go | `(root, path)` / `(root)` | THE single path-safety refusal set for every `${VAR}`-relative catalog path | Shared by `backup:` and `data_paths:`. **Do not write a second path validator.** |
|
||||
| `stacks.ValidateDataPaths` | controller/internal/stacks/datapaths.go | `(entries, binds, appName, logger)` | `data_paths:` annotation validation | ASYMMETRIC on purpose (Fork-3): malformed PATH ⇒ whole-block reject (data handling, `backup:` precedent); unknown ROLE ⇒ fails OPEN, one WARN (presentation, `Lifecycle` precedent). |
|
||||
| `web.fileBrowserLink` / `importFolderLink` | controller/internal/web/filebrowser_link.go | `(domain, sourceName, relPath)` | FileBrowser Quantum deep link | Template read out of the shipped router (SPIKE P2). **`url.PathEscape` per segment — NEVER `QueryEscape`** (space→`+` is a literal plus in a path). Let `html/template` do the attribute escaping; do not pre-escape. |
|
||||
| `HumanizeBytes` | controller/internal/appbackup/appdata.go | `(b int64) string` | Human byte sizes | Exported canonical; private clones exist (§6) |
|
||||
| `stablePathForName` / `agentWhere` | controller/internal/web/intermediary.go | `(name/registeredPath) string` | Map registry stable path `/mnt/felhom-drives/<n>` ↔ raw agent mount | Registry stores STABLE path; agent ops take the RAW mount — always convert |
|
||||
| `offsiteRestoreRootFor` | controller/internal/backup/offbox_verify_copies.go | `(drivePath string) string` | THE only place `backups/offsite-restore` is spelled | `offboxRestoreScratchDir` builds on it — the listing/delete surface MUST resolve byte-identical paths to what the restore wrote. Do not re-hardcode the segments (they were open-coded in 3 places before v0.147.0) |
|
||||
| `ProtectedHDDPaths` | controller/internal/stacks/delete.go | `(hddPath string) map[string]bool` | Never-delete set (root, appdata, backups, media, legacy felhom-data) | Consult before ANY recursive delete under a drive |
|
||||
|
||||
### Subprocess + timeout + exit-code discipline
|
||||
@@ -40,6 +45,9 @@
|
||||
| `jsonResponse` / `jsonError` | controller/internal/web/handler_export.go | `(w, v)` / `(w, msg, code)` | Export/import API | Third envelope shape — keep within export surface |
|
||||
| `limitBody` | controller/internal/api/router.go | `(w, req)` | Bound request bodies (1MB) | Apply before decode on any new POST |
|
||||
| `offboxRedirect` | controller/internal/web/offbox_handlers.go | `(w, r, msg string, isErr bool)` | Flash-message redirects | Flash = `?flash=` / `?flash_error=` query params, read by page handlers |
|
||||
| `offboxRedirectTo` | controller/internal/web/offbox_handlers.go | `(w, r, page, msg string, isErr bool)` | Same, to an EXPLICIT page | **TRAP (fixed v0.154.0): the separator is chosen, not `"?"`.** Targets may already carry a query — the R-48 wizard is `/backups/restore/app?name=<app>` — and a hardcoded `"?"` buries the flash inside the previous parameter's value |
|
||||
| `restoreOpInFlight` + `hasRecentRestoreResult` | controller/internal/web/restore_wizard.go | `(backup.RestoreOpStatus) bool` / `(st, app, now) bool` | THE "is a restore running / did one just finish" display reads | **TRAP (v0.154.0 shipped this bug): `Manager` has TWO running flags.** `IsRunning()` reads the CONCURRENCY flag, acquired inside the goroutine — and `RestoreOffboxScratch` never acquires it, so it is false for the whole verification restore. Display must read `RestoreStatus().Running` (set synchronously by `BeginRestoreOp`). Read the status ONCE per render or the strip and the suppression can disagree. `hasRecentRestoreResult` is app-bound and window-bounded — a process-wide result must not light another app's „Eredmény" |
|
||||
| `restoreWizardPath` / `deriveWizardStep` / `resolveWizardApp` | controller/internal/web/restore_wizard.go | `(app) string` / `(restoreWizardInput) restoreWizardView` / `([]OffboxAppRow, name) *OffboxAppRow` | R-48 offsite restore wizard: URL builder + the PURE step/unlock derivation + the app-resolution refusals | The step is **never** taken from the request. Precedence is load-bearing: op-running outranks a stale `?full_prep=`, else a commit button reappears mid-restore. Truth table + red-proof: `restore_wizard_test.go`. Adding a form here that posts anywhere new breaks `TestRestoreWizard_NoNewMutationEndpoints` **by design** — R-48 adds no mutation surface |
|
||||
| `redirectTier2` | controller/internal/web/tier2_config_handler.go | `(w, r, name, flash, flashErr)` | Tier2 page flash redirects | Same convention |
|
||||
| `validStackName` | controller/internal/web/validate.go | `(name string) bool` | Any stack name from a request | Single-segment, no `/ \ ..` — blocks path traversal into stacks/userdata |
|
||||
| `ValidateSegment` | controller/internal/appexport/validate.go | `(kind, s string) error` | Any attacker-controlled path segment (.fab manifest fields) | CTRL-001 guard; deliberately NOT for dotfile ConfigFiles |
|
||||
@@ -49,6 +57,10 @@
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
|---|---|---|---|---|
|
||||
| `backup.ErrOffboxSealedPackageHeld` + `IsOffboxSealedPackageHeld` + `sealedPackageHeld` + `OffboxAwaitingRecoveryKey` (R-241, v0.206.0) | controller/internal/backup/offbox.go | sentinel; `(error) bool`; `() bool`; `() bool` | **THE MINT GUARD** — a box never creates a repository key while the hub holds a sealed package for it | **The guard is a CONJUNCTION** (package held AND no key present). Widening it to "never mint" leaves a first-time box unable to start, waiting for a package that will never exist — pinned by `TestR241_ScenarioB_FirstTimeBoxStillMints`. **The refusal is a HOLDING state, not a failure:** `ApplyOffsiteTarget` catches the sentinel and still writes the transport, so `/recovery`'s synchronous tier-up (R-219) can bring the tier up the instant the key arrives; returning the error instead leaves `needsOffsiteCredential` true and the hub re-staging a consumed credential for ever. `OffboxAwaitingRecoveryKey` is **DERIVED, never stored** — and **`t.Enabled` is load-bearing in it**: a customer who switched off-site OFF is not awaiting anything (the Scenario-E carve-out `needsOffsiteCredential` makes two functions above; the first draft omitted it and an existing test caught it). A nil settings store reads as "no package held" — a transient read failure must never become a permanently-held tier |
|
||||
| `settings.HubEscrowKeySHA256` + `SetHubEscrowKeySHA256` / `GetHubEscrowKeySHA256`, and `OffsiteRecoveryOffer` **shape (c)** (R-241, v0.206.0) | controller/internal/settings/settings.go, controller/internal/backup/offbox.go | `(sha, checkedAt string) error` / `() (string, string)` | **THE DISCRIMINATOR the recovery screen asks** — does the hub hold a package for a key other than the one we use? | **The comparison was ALREADY computed on every ACK since SLICE 3 and persisted nowhere** — that is R-241's second half. Wire the recorder in `main.go`'s `EscrowAutoConfirmer` literal or shape (c) reads an empty hash for ever and the fix ships INERT (pinned by `TestMainWiresRecordEscrowKeyHash`). **§7.2 staleness, decided:** a KNOWN DIFFERENCE offers **however old the reading** — age is deliberately NOT gated on, because gating makes a box offline from the hub silently stop offering; an **ABSENT hash falls back to (a)/(b)** and does NOT offer, because `""` is the hub positively saying its package seals no key (legacy hash-less escrow), not an unknown. `CheckedAt` is for diagnosis, never a gate |
|
||||
| `backup.AbandonStatus` / `AbandonSweep` / `CancelAbandon` / `ClearAbandonPurgeIfConfirmed` / `ExtendAbandon` / `StopAbandon` + `AbandonGraceDays` (R-241, v0.206.0) | controller/internal/backup/offbox_abandon.go | see file | **The 14-day abandonment countdown** — the ONLY thing in the product that deletes a customer's off-site history | **BOTH HALVES OR NEITHER.** The set-aside store and the sealed package that protects it are two halves of one thing; removing only one leaves a package that opens nothing, or ciphertext nobody can decrypt. Not atomic across two machines, so it is a **two-phase commit**: delete the store, set `AbandonPurgeRequested`, and keep declaring it until the hub's ACK stops reporting a superseded package — the confirmation rides the SAME ACK as the request. **The countdown starts in `ResetOrphanedRepo`, NOT in the shared `resetOrphanedRepo`** — the helper is also the UNCLAIMED auto-reset, where nobody decided anything. **The recovery offer stays reachable for the whole grace** (a grace in which recovery is impossible is decorative). **Drive it with `SetOffboxClock`, never a shortened live timer** (§7.4). A transport failure leaves the countdown DUE so tomorrow retries; the operator levers REFUSE rather than no-op when nothing is running or the store is already gone |
|
||||
| `settings.SyncRecoveryOfferEpoch` / `PostponeRecoveryNoticeForEpoch` / `OptOutRecoveryRemindersForEpoch` + `web.recoveryBannerCookie` (R-241, v0.206.0) | controller/internal/settings/settings.go, controller/internal/web/recovery_handlers.go | `(offered bool, now) (RecoveryOfferView, error)` | **The offer EPOCH** — "once per entry into the offered state", not once ever | **Sync the epoch FIRST and UNCONDITIONALLY in `recoveryInterrupts`.** The first draft returned early when the offer was false, so the FALLING edge was never recorded, `RecoveryOfferActive` stayed true through a settled period, and the next entry counted as a continuation — **the exact defect the epoch exists to fix, reintroduced inside the fix**. Dismissals are recorded against the epoch they were made in, so a fresh entry resets them **by arithmetic**, with nothing to clear. **Three levers, three scopes, and NONE removes the entry point on `/backups/remote`:** the banner cookie is a browser SESSION cookie (no MaxAge — cleared on login) and persists nothing; the reminder opt-out is durable but silences the BANNER ONLY; "most nem" suppresses the full page only |
|
||||
| `atomicWrite` | controller/internal/backup/recovery_unit.go | `(path, data, perm) error` | Atomic file writes (backup pkg) | tmp+rename; no dir creation, no fallback |
|
||||
| `writeFileAtomic` | controller/internal/bootstrap/bootstrap.go | `(path, b) error` | controller.yaml writes from bootstrap | Always 0600 (holds local-api token + hub key) |
|
||||
| `writeConfig0600` | controller/internal/api/router.go | `(path, body) error` | config writes via API | ALWAYS chmods 0600 even pre-existing (F8); direct-write fallback on bind-mount EBUSY (non-atomic!) |
|
||||
@@ -56,8 +68,19 @@
|
||||
| `Settings.save` (unexported) | controller/internal/settings/settings.go | via mutator methods only | ALL settings.json persistence | tmp+rename, then `.bak` last-known-good AFTER rename succeeds. Never write settings.json by hand |
|
||||
| `settings.Load` | controller/internal/settings/settings.go | `(path, logger) (*Settings, error)` | Startup load | Corruption recovery: `.bak` restore → else preserve `.corrupt-<ts>` + safe defaults; never crash-loops |
|
||||
| `Manager.writeJournal` / `loadJournal` | controller/internal/stacks/migrate.go | `(j *MigrationJob)` | Migration crash journal | Enables `RecoverMigration` at startup |
|
||||
| `backup.SharesPseudoStack` / `DisplayStackName` | controller/internal/backup/shares_payload.go | `"_shares"` / `(key) string` | THE reserved key for the shares source (restic tag, `backups/secondary/_shares`, CrossDriveBackup record) + its display mapping | NEVER let the raw key reach a Hungarian surface — map at the notification/prose boundary ONLY; the persisted `EnlargedBlocked` set and the templates index by the RAW key |
|
||||
| `Manager.buildSharesPayload` / `classifiedShares` | controller/internal/backup/shares_payload.go | `() (dir, passdbOK, error)` / `() []classifiedShare` | the definitions+credential payload and the availability-filtered share set both tiers read | payload is SECRET-BEARING (0600 passdb.tar) — never log its bytes/name at INFO. `classifiedShares` is the single place a dead mount is dropped, so both jobs agree |
|
||||
| `Manager.selectTier2TargetFrom` | controller/internal/backup/tier2.go | `(stack, sourceDrive, fullSize, stateOnlySize) (*Tier2Target, error)` | tier-2 target choice with the source drive supplied EXPLICITLY | the seam the shares job reuses — NEVER fork the headroom math; `selectTier2Target` is now a thin wrapper over it |
|
||||
| `Manager.tier2ReconcileRoots` | controller/internal/backup/tier2.go | `(destBase, roots, legRels)` | staleness pruning with explicit dest roots | pure extraction from `tier2Reconcile` (which now calls it with `hdd`/`userdata`); reuse it rather than writing a second pruner |
|
||||
| `Manager.liveShareRootOK` / `scratchJoin` | controller/internal/backup/shares_restore.go | `(dst) bool` / `(scratch, abs) string` | THE place guard for shares restore + scratch path reconstruction | a snapshot is UNTRUSTED layout input: require a STRICT descendant of a live registered root, refuse `..` and the drive root itself. `scratchJoin` strips the volume name — plain `filepath.Join` splices a drive letter mid-path |
|
||||
| `infra.SambaContainerName` / `SambaPassdbVolume` / `SambaPassdbMount` | controller/internal/infra/samba.go | consts | single source of truth for the samba container identity | the compose renderer interpolates them; stacks/backup/monitor read them. The CONTAINER name (`felhom-samba`) is NOT the stack name (`samba`) — `EffectiveProtected` needs the container one |
|
||||
| `sambaWriteAtomic` | controller/internal/stacks/samba.go | `(path, data, mode) error` | samba smb.conf/compose writes | tmp+**fsync**+rename (the only one of these that fsyncs). Fourth atomic-write helper in the tree — see §6 |
|
||||
| `Loop.writeMarker` / `Recover` | controller/internal/quiesce/quiesce.go | `(m Marker)` / `()` | Quiesce crash-safety | Marker written BEFORE stopping stacks; Recover restarts stranded stacks at boot |
|
||||
| `quiesce.TieredBackend` + `Loop.resolveDueTiers` / `quiesceAndPollTiers` | controller/internal/quiesce/tiers.go, quiesce.go | `Tiers/DueFor/StartBackupFor/BackupStatusFor`; `resolveDueTiers(ctx) ([]dueTier,bool,error)` | THE R-82 multi-tier backup schedule — several whole-guest tiers (local daily + PBS weekly) reconciled into ONE quiesce window | **Both tiers due ⇒ ONE stop/start pair**, never two (two = two app outages for one night). Tiers run SEQUENTIALLY (vzdump holds a guest lock) and the app stays down until the LAST tier snapshots — resuming earlier loses app-consistency on the DR tier. Order is fast-first (agent advertises primary first) or downtime blows up. `ErrTiersUnsupported` (route 404) ⇒ pre-R-82 agent ⇒ degrade to the untargeted path and **STILL BACK UP** — never read it as "nothing due". |
|
||||
| `quiesce.failureBreaker` + `Loop.dropBackedOffTiers` / `noteTierFailure` / `noteTierSuccess` | controller/internal/quiesce/breaker.go, quiesce.go | `blocked/recordFailure/recordSuccess(target, now)`; `backoffFor(n) time.Duration` | **R-88** — a tier whose backups keep failing stops re-quiescing. Backoff 15m→30m→1h→2h→4h (cap), reset on success | **It gates the QUIESCE, not the backup** — the harm was never the failing backup, it was the app outage taken to attempt it, so backed-off tiers are dropped from the due set BEFORE any stack is stopped. **Per TARGET** — a broken offsite tier must never suppress a healthy local one (`TestBreaker_OneFailingTierDoesNotSuppressAHealthyOne`). **Never permanent** — the cap bounds the retry INTERVAL, it never stops retrying; a latched breaker is a silent backup outage, worse than the loop it replaces. **`TriggerNow` is never gated** (it already bypasses due-ness and the window gate), though a manual run still RECORDS its outcome. **`stillRunning` is NOT a failure** — a first full offsite snapshot legitimately runs for hours. State is **in-memory on purpose**: a restart forgets the backoff and re-attempts, which is the cheap direction to fail. Log the deferral ONCE when armed, never per tick. |
|
||||
| `quiesce.TierNotifier` + `Loop.SetTierNotifier` / `noteTierFailure` / `noteTierSuccess` | controller/internal/quiesce/breaker.go, quiesce.go | `BackupFailed(tier,msg,err)` / `BackupRecovered(tier,msg)`; `SetTierNotifier(n)` INIT-ONLY | **R-97a** — the whole-guest backup tier reports its outcome to the hub | A **seam, not an import** — quiesce keeps no dependency on `internal/notify` (same reason `windowStartFn` is injected). Wired by a setter because main.go builds the notifier AFTER the loop; `nil` = unprovisioned guest, not an error. **Edge-triggered:** failure fires only when the breaker ARMS (`n == 1`), never per retry — the cadence is 15m/30m/1h/2h/4h and an event per attempt is an inbox nobody reads. Recovery rides `recordSuccess`'s existing bool. **Event types are OPERATOR-ONLY** (`whole_guest_backup_failed`/`_recovered`, hub >= v0.78.0) — NOT `backup_failed`, which has a customerMessages entry AND sits in live `enabled_events`, so it would email the CUSTOMER about a backup they cannot act on. `WholeGuestBackupDetails.Tier` is load-bearing: the hub keys its per-tier cooldown on it. |
|
||||
| `quiesce.Loop.SuppressedStacks` + `markQuiesced` / `markUnquiesced` | controller/internal/quiesce/suppress.go | `() map[string]bool` (nil-safe on a nil *Loop) | **R-97b** — an app THIS controller stopped for a backup is not a fault | Consumed at the SINGLE derivation point `classifyRunStates` (which computes both the banner dead-list and the notifier Down-set — keep it one place). **Cycle-keyed, not state-based:** v0.164.0's `!= StateStopped` filter cannot see an app caught MID-RESTART (`starting`/`unhealthy`), which is how BookStack alarmed on 2026-07-27. The window (`quiesceAlarmGrace` = 180 s, derived from the deploy flow's 120 s health timeout and Mealie's 60 s start_period) **EXPIRES** — permanent suppression turns a loud false alarm into a silent real one. Open-ended while the cycle runs (a first offsite snapshot legitimately takes hours). |
|
||||
| `agentapi.BackupTiers` / `BackupDueFor` / `StartBackupFor` / `BackupStatusFor` | controller/internal/agentapi/backup_tiers.go | `(ctx[, target]) (…, error)` | The per-tier agent surface (agent >= v0.97.0) | `targetQuery("")` returns an EMPTY suffix so an untargeted call hits the pre-R-82 route byte-for-byte. `BackupTiers` maps a 404 to `ErrTiersUnsupported` — the documented ROUTE-PROBE capability signal, NOT a `featureProbes` row (the loop needs the tier LIST, not a yes/no). |
|
||||
|
||||
### Compose ops / stack lifecycle
|
||||
|
||||
@@ -65,17 +88,28 @@
|
||||
|---|---|---|---|---|
|
||||
| `Manager.DeployStack` | controller/internal/stacks/deploy.go | `(req DeployRequest) (string, error)` | Full deploy flow | Sets in-memory `Deployed` BEFORE compose up (slow-pull race), reverts on failure |
|
||||
| `Manager.RedeployFromEnv` | controller/internal/stacks/deploy.go | `(name, env map[string]string) error` | Re-up with changed env (migration flip, config edits) | `compose up -d`, never `restart` (restart won't pick up images/env) |
|
||||
| `Manager.StartStack/StopStack/RestartStack/UpdateStack` | controller/internal/stacks/manager.go | `(name string) error` | Lifecycle | Protected stacks refuse stop; all funnel through composeExec |
|
||||
| `Manager.PersistUnitRedeployConfig` (R-47, v0.153.0) | controller/internal/stacks/deploy.go | `(name, env map[string]string) error` | the PERSIST half of `RedeployFromEnv` — app.yaml + locked fields + in-memory flags, **starts nothing** | **TRAP: the restore paths must use THIS, never `RedeployFromEnv`.** RedeployFromEnv ends in a full `up -d`, which before the replay IS the H4 race. RedeployFromEnv is now literally this + the unchanged up-and-report tail |
|
||||
| `Manager.StartStackServices` (R-47, v0.153.0) | controller/internal/stacks/manager.go | `(name string, services []string) error` | scoped `compose up -d <svc>...` — the DB-only window a dump is replayed in | **REFUSES an empty list** (argument-less `up -d` is a FULL start — the one silent fall-through that would reintroduce the race). No `logPostStartStatus`: the app containers are absent on purpose. Never `RestartStack` here — it is a full up in disguise |
|
||||
| `appbackup.DBServiceNames` / `dbTypeForImage` (R-47, v0.153.0) | controller/internal/appbackup/dbservices.go | `(composePath string) ([]string, error)` | naming the compose SERVICE(s) holding a database, sorted | yaml.v3 `services:` MAP parse — **never a line scan** (immich's top-level `immich_ml_cache:` / `immich_postgres_data:` volume keys look exactly like services). `dbTypeForImage` is shared with `DiscoverDatabases`, which is what makes "a dump exists ⇒ a service can be named" hold. An error means CANNOT-TELL, never "no database" — callers refuse when a dump exists |
|
||||
| `Manager.StartStack/StopStack/RestartStack/UpdateStack` | controller/internal/stacks/manager.go | `(name string) error` | Lifecycle | Protected stacks refuse stop; all funnel through composeExec. **NOT writers of desired state (R-166)** — 14 call sites, only 2 are the customer; recording intent here would make a nightly backup indistinguishable from the customer pressing Stop. Use `SetDesiredState` at the intent point instead |
|
||||
| `Manager.SetDesiredState` / `DesiredStateOf` / `BackfillDesiredState` (R-166, v0.189.0) | controller/internal/stacks/desiredstate.go | `(name, desired string) error` / `(Stack) string` / `() int` | THE customer-intent record — `app.yaml` `desired_state`, tri-state `""`/`running`/`stopped` | **ONE OWNER: the customer's action.** Writers are the API action switch, `DeployStack`, `UpdateOptionalConfig`'s redeploy branch, and the `.fab` restore adapter — nothing else, ever. **`""` (absent) means UNKNOWN, never "running"**: every pre-v0.189.0 app.yaml reads absent, so treating it as running would start every deliberately-stopped app on upgrade. Write intent BEFORE the act and REFUSE the act if it fails (§8.2). Backfill is **running-only** — never infer `stopped` from zero containers, that inference IS the defect |
|
||||
| `Manager.DriveLive` (R-171, v0.190.0) | controller/internal/stacks/deploy.go | `(hddPath string) bool` | is an app's data drive a live mountpoint RIGHT NOW | Wraps the **same** `isMountPoint` seam the userdata belt uses (`manager.go`) — never write a second liveness check, the two would drift invisibly. The system/local path is legitimately not a mountpoint and returns true |
|
||||
| `bootrecon.StartGate` (R-171, v0.190.0) | controller/internal/bootrecon/bootrecon.go | `MayStart(stack) (bool, reason)` | THE one question the boot sweep asks before starting anything | **Fail-safe: cannot determine ⇒ return FALSE.** One seam for all three holders (absent drive · quiesce · an in-flight app-data operation) because they differ only in the reason string. Implemented in `main.go` (`bootDriveGate`) reusing `quiesce.SuppressedStacks()`, `AppStopGuard.HeldStacks()` and `Manager.DriveLive` — never re-derive any of them. Held apps go to `Result.HeldByDrive`, **never** `StillDown` (that is the dead-app alarm's bucket) |
|
||||
| the boot settle window (R-157 A, v0.190.0) | controller/cmd/controller/main.go | `bootReconcileSample` / `StableFor` / `Budget` | sample the fleet until it stops changing, then sweep ONCE | **settle + budget + one `DefaultRetryDelay` must stay under `deadAppBootGrace`** — pinned by `TestBootWindow_CommonCaseFitsInsideTheDeadAppGrace`, which is why the budget is 50 s and not 60 s. Sampling is READ-ONLY; sweeping per sample would never see a settled fleet (the sweep's own StartStack changes it). A late recovery is REPORTED (`recordLateRecovery`), never hidden by widening the grace |
|
||||
| `backup.AppStopGuard` (`Begin`/`End`/`Recover`) (R-166, v0.189.0) | controller/internal/backup/appstop_marker.go | `(opID, reason, stacks) error` / `()` / `() *AppStopRecovery` | THE crash marker for stop→work→start windows (volume dump, offbox reconstitute, `.fab` export) | Its **own** file (`appstop-state.json`), never quiesce's — one file, one writer. **A `defer` is NOT the mechanism** (Campaign 8 fault 10: SIGKILL runs no defer); the marker is. Written BEFORE the stop, cleared ONLY after a restart that succeeded; a FAILED restart deliberately KEEPS it. `Recover` RETURNS its outcome rather than notifying, because it must complete before the boot reconciler while the notifier does not exist yet |
|
||||
| `backup.ErrStartRefused` + `AppStopRecovery.Refused`/`Alarming()` (R-174, v0.191.0) | controller/internal/backup/appstop_marker.go | `errors.Is(err, ErrStartRefused)` / `() bool` | THE refusal-vs-failure split in the app-stop crash recovery | **A gated starter's refusal is NOT a restart failure.** `Recover`'s starter MUST be the gated `gatedAppStopStarter` (cmd/controller/main.go), never the raw `stacks.Manager` — that was the v0.189.0 defect, which started apps onto ABSENT drives at boot (R-171 one path over). A refusal goes to `Refused` (marker KEPT, silent), a real error to `Failed` (marker kept, ALARMS). Collapsing them routes a deliberate hold into `NotifyBackupFailed`, a customer-enabled type — the R-171 false alarm again. `main.go` must guard the notify with `Alarming()`, not `!= nil` |
|
||||
| `Manager.DeleteStack` / `RemoveStack` | controller/internal/stacks/delete.go | `(name, removeHDDData[, backupPaths])` | THE guarded removal paths | Orphan/protected/deploying/running checks + ProtectedHDDPaths filter before any RemoveAll |
|
||||
| `resolveContainerState` / `aggregateState` | controller/internal/stacks/manager.go | `(dockerState, dockerStatus)` / `([]ContainerInfo)` | State classification | `.State` says "running" even when unhealthy — `.Status` parse is the fix |
|
||||
| `Manager.logPostStartStatus` | controller/internal/stacks/manager.go | `(name, stackDir, env)` | Async post-start verification | compose up exits 0 on crash-loops; this is the detection. Goroutine + 3s, never blocks |
|
||||
| `Manager.EnsureBaseStack` | controller/internal/stacks/infra.go | `() error` | Traefik/cloudflared/FileBrowser infra convergence | Renders from `internal/infra` templates |
|
||||
| `appbackup.ClassifyBinds` / `ValidateBackupSpec` | controller/internal/appbackup/classify.go | `(spec, binds) ([]ClassifiedBind, bool)` / `(spec, binds) error` | Backup-classification (Task 2, referential coupling) — pure | Two-level default: explicit wins over `:ro`; unlisted writable→mandatory, unlisted `:ro`→excluded; nil spec→legacy/false. Validate REJECTS the WHOLE block on any defect (whole-block semantics). INERT — no tier consumes it yet |
|
||||
| `ParseComposeClassifiableBinds` | controller/internal/stacks/classify_binds.go | `(composePath) []appbackup.ComposeBind` | `${VAR}`-relative binds + `:ro` for classification | Do NOT use `ParseComposeHDDMounts`/`ExportDataMounts` as classifier input (§traps) — they resolve absolutes, drop `:ro`, or union the userdata ROOT. Short-syntax only |
|
||||
| `Metadata.EffectiveLifecycle` / `CanInstall` / `IsAbandoned` + `web.lifecycleBadge` / `web.visibleCatalogStacks` | controller/internal/stacks/metadata.go, controller/internal/web/metabadge.go, controller/internal/web/handlers.go | `meta.CanInstall() bool` | app lifecycle: `available` / `hidden` / `abandoned` (v0.158.0) | THE single interpretation of `.felhom.yml` `lifecycle:` — every surface must go through these, never compare the raw string. Listing drops `!Deployed && !Protected && !CanInstall()`; `api.deployStack` refuses server-side BEFORE any mutation (hiding a button is not a gate), `stacks.DeployStack` repeats it for non-API callers. **Unknown value fails OPEN** (→ available + one WARN) — opposite to the gate on purpose: a typo must never pull a working app out of every catalog. **NEVER let lifecycle reach orphan detection** (`getCatalogTemplateSlugs`) — a withdrawn template stays in the tree, or every deployed instance reads as `Elavult` and gets a Törlés button. Badges: `MetaBadge` + `meta_badge` partial, built generic for R-56 difficulty labels |
|
||||
| `Manager.ClassifiedBinds` + `StackDataProvider.GetStackClassifiedBinds` | controller/internal/stacks/metadata.go, appbackup/appdata.go | `(name) ([]appbackup.ClassifiedBind, bool)` | Per-stack classification through the REAL LoadMetadata validate path | The wired seam Task 3 consumes; LoadMetadata is the SINGLE validation choke point (bad block → nil + one ERROR → legacy) |
|
||||
| `backup.Manager.DumpAppVolumesSafe` | controller/internal/backup/backup.go | `(stackName) error` | Volume tar of a live app | Stops → dumps → restarts; surfaces BOTH errors (app may be left stopped). Check `GetDockerVolumes()!=0` + `IsProtectedStack` BEFORE calling — it stops the stack before its own volume check (see `runVolumeDumps`) |
|
||||
| `backup.Manager.ListRestorePoints` | controller/internal/backup/restore_points.go | `(stackName) ([]RestorePoint, bool)` | Restorable keep-side backups (the /api/backup/snapshots payload) | ONE point per app (the current unit); tier always 1 — never list Tier-2 (not restorable via /backup/restore) |
|
||||
| `backup.Manager.RestoreTier2Files` | controller/internal/backup/tier2_restore.go | `(stackName) (filesRestored int, err error)` | In-place ADDITIVE-ONLY class-C file restore from the recorded Tier-2 copy (`POST /backup/tier2/restore`) | Never overwrites/deletes live files; refusals (Hungarian) before any stop; source = recorded `DestinationPath`, never re-selected |
|
||||
| `backup.Manager.RestoreTier2Files` | controller/internal/backup/tier2_restore.go | `(stackName) (filesRestored int, err error)` | In-place ADDITIVE-ONLY class-C file restore from the recorded Tier-2 copy (`POST /backup/tier2/restore`) | Never overwrites/deletes live files; refusals (Hungarian) before any stop; source = recorded `DestinationPath`, never re-selected. **C9-F1 (v0.183.0): reads `hdd/` + `userdata/` ONLY — never `recovery-unit/`.** For 43 of 53 catalog apps that is a guaranteed no-op, so it now refuses with `ErrTier2NoRestorableData` BEFORE stopping the app. Ask `Tier2RestoreCoverage` first |
|
||||
| `backup.Manager.Tier2RestoreCoverage` | controller/internal/backup/tier2_restore.go | `(stackName) (Tier2Coverage{Legs, HasUnit}, error)` | Answers what a Tier-2 restore CAN and CANNOT return for an app, from the RECORDED copy on disk | **C9-F1.** `Legs` = subtrees the restore reads; `HasUnit` = the copy also holds DB dumps + volume tarballs it will NEVER read. Use it to refuse up front and to decide whether the success message must disclose uncovered data. Judged from the copy, not the catalog, so a retemplated app is judged by what it actually has |
|
||||
| `Manager.acquireRunning`/`releaseRunning`, `acquireMigrating` | controller/internal/backup/backup.go, controller/internal/stacks/migrate.go | `() error` | Single-flight for long ops | Copy this mutex-flag pattern for any new long-running manager op |
|
||||
|
||||
### Secrets hygiene
|
||||
@@ -87,10 +121,13 @@
|
||||
| `SaveAppConfig` / `LoadAppConfigDecrypted` | controller/internal/stacks/deploy.go | `(stackDir, cfg, encKey, sensitiveVars)` | app.yaml persistence | Encrypts only `SensitiveEnvVars(meta)`; never write app.yaml directly |
|
||||
| `generateValue` / `randomAlphanumeric` | controller/internal/stacks/deploy.go | `(spec "password:N\|hex:N\|base64key:N\|static:v")` | Auto-generated secrets | crypto/rand-backed; reuse the spec grammar |
|
||||
| `Manager.GenerateSecretForField` | controller/internal/stacks/deploy.go | `(stackName, envVar) (string, bool)` | Replacement value for a RESETTABLE secret from its catalog `generate` spec (O4 restore path via `backup.SetSecretGenerator`) | REFUSES `data_key` fields, spec-less and non-secret fields; never log the value |
|
||||
| `reconcileRestoreSecrets` | controller/internal/backup/restore_unit.go | `(nonSecretEnv, recoveredSecrets, secretNames, dataKeyNames)` | Recovery-unit restore env merge | Units are secret-FREE by design; secrets come from live app.yaml |
|
||||
| `reconcileRestoreSecrets` | controller/internal/backup/restore_unit.go | `(nonSecretEnv, unitSecrets, guestSecrets, secretNames, dataKeyNames)` | Recovery-unit restore env merge | **Precedence: UNIT WINS over guest** (the unit's secrets match the data being restored; the guest's are merely newest). Pure — new sources arrive as ARGUMENTS. Fail-closed data-key gate lives here |
|
||||
| `stacks.PortableSecretEnvVars` | controller/internal/stacks/deploy.go | `(meta) []string` | **THE D5 secret boundary**: which secrets may travel on a customer drive | `type: secret` travels, `type: password` NEVER, minus the `nonPortableSecrets` code register. Withholding the password class is what licenses plaintext — do not relax one without the other |
|
||||
| `buildUnitAppYaml` / `readUnitEnv` | controller/internal/backup/{recovery_unit,restore_unit}.go | `(info) []byte` / `(path, portableNames)` | The ONE place the unit's app.yaml is written / split back | Split is driven by the MANIFEST's portable names, never guessed from key names; write 0600; empty `portableNames` = schema-1 unit ⇒ everything is plain config |
|
||||
| `EncryptFile` / `DecryptFile` / `IsEncryptedFAB` | controller/internal/appexport/crypto.go | password-based file crypto | .fab export bundles | scrypt-derived AES+HMAC keys |
|
||||
| `maskRepoURL` | controller/internal/sync/sync.go | `(url) string` | Logging git URLs | Strips embedded credentials |
|
||||
| `metrics.RedactLine` | controller/internal/metrics/redact.go | `(s string) string` | ANY log line shipped off-box (issue context, log tails) | Masks password/passwd/secret/token/api-key/authorization/bearer values + 64-hex; apply BEFORE the line leaves the box — controller-side redaction is authoritative |
|
||||
| `settingsRetrievalPasswordRevealHandler` | controller/internal/web/handlers.go | `POST /settings/retrieval-password/reveal` | **THE PATTERN for showing a secret in the UI** — an XHR that returns only the value | **Never template a secret into a page and hide it with CSS.** `display:none` / `hidden` / `type="password"` stop a browser DRAWING the value; the plaintext is still in the response body, so a `curl` of the page returns it, and it reaches caches, history and any screen-share of the source. R-249 shipped exactly that for two months and was found by it landing in a transcript. The page carries a **boolean** (`HasRetrievalPassword`); the value comes from a POST (CSRF-covered, uncacheable) and the reveal is **logged as an act**. `escrow_handlers.go` states the same rule for R. **Test on the RESPONSE BODY** — a test asserting what the customer *sees* cannot see this class at all. **Both R-254 sites are now FIXED the same way** — `POST /apps/<slug>/initial-credentials/reveal` (re-reads the container, never a cached copy) and `POST /stacks/<name>/auto-field/reveal` (authorised on the field being a `type: secret` auto-field of that stack). **Per-secret, never one generic reveal-any-named-secret endpoint.** The PRE-DEPLOY hidden input is deliberate and untouched — a form must carry what it submits (README §318). Enforced by `scripts/secret_in_markup_gate.py`, whose measured blind spot (a secret under a neutral page-data key) is in its docstring; runtime body-assertion covers 4 of 27 pages — R-255. |
|
||||
|
||||
### Storage registry + mount detection
|
||||
|
||||
@@ -122,28 +159,42 @@
|
||||
| `Client.AddNetStorage/ListNetStorage/RemoveNetStorage` | controller/internal/agentapi/client.go | NAS mounts (A1) | Network storage | Password passes through to agent's 0600 cred file; controller NEVER persists it |
|
||||
| `agentapi.StatusError` | controller/internal/agentapi/client.go | `{Path, Code}` typed non-2xx GET error | Distinguishing HTTP statuses from transport errors (`errors.As`) | NEVER string-match agent error text — the capability probe keys on `Code==404` |
|
||||
| `SupportCache.Supports` / `Client.Supports` | controller/internal/agentapi/features.go | `(ctx, prober, Feature) SupportState` | Agent-capability gate for COUPLED features (route probe, TTL 5m) | 404 ⇒ No; transport/5xx ⇒ Unknown (NEVER refuse on Unknown). New coupled feature = new `featureProbes` row + gate call at the entry point + `MinAgent:` in the CHANGELOG header (publish-train-rules.md). Web layer: `Server.netFeatures` through the `netAgent` seam |
|
||||
| `agentapi.DiskVerdictFor` / `DiskVerdict.Label` / `DegradedAttributes` / `UncorrectableSectors` / `DiskPrior` / `TemperatureFailC` | controller/internal/agentapi/diskverdict.go | `(*SmartSummary, DiskPrior) DiskVerdict` | THE shared disk-health verdict (card chip + hourly check) — v0.169.0, 14-row ladder v0.215.0 | Pure — no clock, no I/O; history arrives as `DiskPrior`. nil/UNKNOWN → `DiskVerdictUnknown` (Nincs adat, NEVER alarms, row 1 is first for that reason). **Never trust `smart_status.passed`**: attrs 187/197/198 carry `thresh: 0`, so it cannot fail on unreadable sectors. A zero `DiskPrior` is the fail-safe (first sighting can only reach Figyelmeztetés). **Four labels, no fifth** — predicted failure is „Hiba". Do NOT recompute the verdict inline anywhere, and do NOT re-literal 60 °C — use `TemperatureFailC` |
|
||||
| `Server.resolveBackupTargetState` / `backupTargetView` | controller/internal/web/backup_target_offer.go | `(ctx)` → state / `*BackupTargetView` (nil = render nothing) | The whole-system backup-target answer: healthy · degraded-never-configured · **TargetAbsent** (configured, drive gone) · unknown | Test seams `Server.tiersFn` + `Server.disksFn` (nil → the real client). **`degradedMessageFor` is the ONE place that decides customer copy** — add a state there, never in a template. `backupTargetView` returns **nil** for healthy AND unknown so a template typo cannot decorate a working box. R-112: this state had NO consumer for two releases; the render is server-side on `backups.html`, and the seam test drives `backupsHandler` and asserts rendered HTML |
|
||||
| `Server.cachedDisks` / `RunDiskHealthCheck` | controller/internal/web/disk_health.go | `(ctx)` | Card fetch (60s TTL) / the hourly degradation check | Card uses the 60s TTL cache (anti-smartctl-storm); the CHECK fetches FRESH (`fetchDisks`). Test seams: `Server.disksFn` (source) + `Server.diskNotifyFn(notify.DiskAlert)` (sink). State is PERSISTED (v0.215.0) — a restart no longer re-baselines |
|
||||
| `diskAlertDecision` / `diskAlertKindFor` / `Server.priorFor` / `Server.cardPriorFor` | controller/internal/web/disk_health_state.go | pure + `(key) agentapi.DiskPrior` | Whether an observation emits, and which message shape | Compares against the **last ALERTED** verdict, not the last observed — that is what collapses a flap to one alert. Re-alert needs doubling **AND** 24h (an AND). **`priorFor` is for the CHECK, `cardPriorFor` for the CARD** — they differ by one observation and mixing them makes the chip read one level more severe than the email |
|
||||
| `diskRecord` / `writeDiskState` / `Server.loadDiskStateLocked` | controller/internal/web/disk_health_state.go | `disk-health-state.json` in `cfg.Paths.DataDir` | Persisted per-disk observation + alert history | Atomic tmp+rename (the `selfupdate.SaveState` shape, copied not imported). Missing file = normal; corrupt = LOG and fall back to no-prior, **never fatal**. Written ONCE per check run. Keyed by `diskKey`. **One record per disk, NOT a sample series** — history is Phase 2/3 in `metrics.MetricsStore` |
|
||||
|
||||
### Notifications / hub sync
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
|---|---|---|---|---|
|
||||
| `Notifier.PushEvent` | controller/internal/notify/notifier.go | `(eventType, severity, message, details)` | Hub events | Async goroutine, 3 attempts/3s backoff. NEW event types MUST be added to hub `allowedEventTypes` or POST /event 400s; hub only emails `warning`/`error` from this path |
|
||||
| `Notifier.PushEvent` | controller/internal/notify/notifier.go | `(eventType, severity, message, details)` | Hub events | Async goroutine, 3 attempts/3s backoff. NEW event types MUST be added to hub `allowedEventTypes` or POST /event 400s. **SEVERITY IS AN EXACT WIRE CONTRACT: `{"info","warning","error","critical"}` and nothing else.** The hub silently COERCES any other string to `"info"` (`hub/internal/api/handler.go`, the ingest severity switch) and `severityNotifies` (`hub/internal/notify/dispatcher.go`) emails only warning/error/critical — so a typo'd severity is stored and delivered to NOBODY, with no error anywhere. **`"warn"` is not a severity.** It shipped on `disk_health_degraded` (fixed v0.215.0, R-328) and is STILL live on `app_start_failed` (R-329) |
|
||||
| `notify.DiskAlert` / `DiskAlertKind` / `DiskAlertKind.Severity()` | controller/internal/notify/notifier.go | `NotifyDiskHealthDegraded(DiskAlert)` | The disk-health alert payload + its five Hungarian message shapes | The notifier owns customer copy — pass a `DiskAlert`, never a pre-formatted string, or Hungarian scatters across packages. `Severity()` is the ONE mapping kind→hub severity and is exported so any package can assert the contract instead of duplicating the literal |
|
||||
| `Notifier.Notify*` convenience methods | controller/internal/notify/notifier.go | typed wrappers (backup/DB/storage/channel/DR…) | Standard events | Add a typed wrapper rather than raw PushEvent calls |
|
||||
| `report.BuildReport` / `Pusher.Push` | controller/internal/report/builder.go + pusher.go | periodic hub report | Box→hub reporting | ACK carries `config_version` → `ConfigRefresher.Reconcile` |
|
||||
| `report.Trigger` (`NewTrigger`/`Fire`/`Run`) | controller/internal/report/trigger.go | `Fire()` after a hub-relevant user action | THE out-of-cycle report push (v0.139.0) — fire via `api.Router.reportPushNow` / `web.Server.reportTriggerNow`, both nil-safe | Coalesce-and-eventually-fire (trailing edge; quiet 2s, min spacing 15s). NEVER add retries (Pusher owns them); NEVER reuse the `internal/sync` REFUSE-debounce for hub pushes (a refused fire loses the update until the next cycle). Fire only AFTER a successful local commit |
|
||||
| `report.SetPendingLogTails` + `buildLogTailsSection` | controller/internal/report/logtail.go | ACK `log_tail_requests` → next report `log_tails` | THE pull-based ACK-flag pattern (hub asks, controller pushes next cycle) — copy for any new hub→box request | Consume-once drain at BuildReport; failed push re-arms from the hub's still-pending request; NEVER add a hub→controller push channel |
|
||||
| `metrics.FetchContainerLogTail` | controller/internal/metrics/logscanner.go | `(name, tailLines) (string, error)` | Raw per-container `docker logs --tail=N` | 15s timeout; caller caps/redacts (capTailLines) |
|
||||
| `ConfigRefresher.Reconcile` | controller/internal/report/config_refresh.go | `(ackVersion int)` | Pull-based config refresh | Re-pulls controller.yaml (re-merging local_api), then graceful self-restart; first-run = baseline, no restart |
|
||||
| `offsiteapply.SettleProvider` / `SettleFunc` / `Bridge.AwaitSettle` / `ReconcileWhenSettled` (R-71a, v0.162.0) | controller/internal/offsiteapply/offsiteapply.go + seams.go | `SettleState() (version, floor string, updateRunning, floorKnown bool)` | THE settle-gate: defers the offsite one-time-password consume past a managed day-0 floor-update (the F10 race). Wire the `SettleFunc` adapter over `updater.GetFloor()`/`IsUpdateRunning()` — **the updater's knowledge is the ONE floor source; never fetch the floor a second way**. Gate ONLY the bridge goroutine, and only when an updater exists (nil `Settle` = reconcile immediately). Bounds `settlePoll`/`settleFloorSubBound`/`settleOverallBound`; the floor is in-memory (report-ACK-derived, ~5–10 s), NOT persisted → unknown until the first ACK on any restart. Inject `Now`/`Sleep` in tests (no real sleeps). B′: at/above-floor GOes on the first poll, zero wait. Do NOT touch the consume/persist order or the 404 contract — ordering only |
|
||||
| `bootstrap.MaybeIngest` / `RefreshConfig` | controller/internal/bootstrap/bootstrap.go | bootstrap.json → controller.yaml | Day-0 + refresh | Overwrites controller.yaml, NEVER settings.json |
|
||||
| `api.GracefulSelfRestart` | controller/internal/api/selfrestart.go | `(logger)` | Controller self-restart | Detached exit; bootstrap unit re-runs the image |
|
||||
| `Settings.AddPendingEvent/DrainPendingEvents` | controller/internal/settings/settings.go | offline event queue | Events while hub unreachable | — |
|
||||
| `Manager.SetUnitNotify` + `UnitSpace` (R-158/R-167, v0.191.0) | controller/internal/backup/recovery_unit.go | `(func(stack string, err error, *UnitSpace))` | THE per-app Tier-1 recovery-unit capture failure alert — fires PER APP from `captureAllRecoveryUnits`, loop continues | **OPERATOR-TIER** (`recovery_unit_capture_failed`, in the hub's `operatorOnlyEvents`). **NEVER route it to `backup_failed`** — that type is in `DefaultEnabledEvents` and carries Hungarian copy, so it emails the CUSTOMER about a failure they cannot act on (D-c; R-158's own proposal said `backup_failed` and D-c overrides it). `UnitSpace` is **nil when the target filesystem is unreadable** and renders as *"unavailable"*, never as zeros — "0 GB free" and "we could not look" are opposite diagnoses. No controller-side cooldown: the hub owns it |
|
||||
| `Manager.beginRunSummary` / `noteFailure` / `noteAttempted` / `emitRunSummary` / `SetRunSummaryNotify` (R-182, v0.194.0) | controller/internal/backup/runsummary.go | `(kind, runID) func()` / `(app, leg, reason)` / `(RunSummary)` | **THE per-run operator digest.** One `backup_run_failures` event at the end of a run listing every failed app, its leg and its reason — emitted ONLY when something failed | **The RECORD and the NOTIFICATION are different things and must stay so.** The per-app `recovery_unit_capture_failed` event is the record (hub routes it *record-only*, stored + logged every time); this digest is the notification. Before R-182 one event was both, and did neither: nine arrived, two were mailed, seven vanished before `LogNotification`. **Lifetime is `admissionSet`'s exactly** — absent collector means "no run in flight", never a stale answer. **A refusal is noted ONCE, inside `admitApp` where the verdict is taken**, not at the three legs that consult it: R-181's one-verdict-covers-all-three contract makes per-leg noting produce "2 of 1 apps failed". **Deliberate skips (disconnected / decommissioned) must NEVER be noted** — they have their own alert and a nightly digest about an unplugged drive is an ignored digest. **A clean run emits NOTHING**; silence is safe only because the hub's deadline check (`monitor/deadline.go:396,417`) raises a missed backup from report freshness independently — if that is ever weakened this design loses its footing. **`run_id` is unique per real run** (so the hub's 1-h cooldown cannot collapse a manual run into the nightly one) and **deliberately EMPTY on the periodic refresh sweep**, which must stay under that cooldown or a polled status page becomes a mail flood |
|
||||
| `Manager.admitApp` / `beginAdmissionRun` / `decideAdmission` / `estimatedWriteBytes` (R-181, v0.193.0) | controller/internal/backup/admission.go | `(stackName) bool` / `() func()` | **THE reserve gate. Call it before ANY per-app backup write** — one verdict per app per run, covering the DB dump, the volume dump and the unit capture (all three write under one per-app root) | **Decided LAZILY at the app's first write, never once at run start** — app A's dump can put app B under the reserve, so a run-start verdict reads a disk that no longer exists. **Never re-decided between an app's own legs**: that is exactly the split R-181 closed (bulk written, capture refused). **Reset per run** via the closer `beginAdmissionRun` returns. **Must sit ahead of `DumpAppVolumesSafe`**, which stops the stack as its first act — a refusal decided inside it has already bounced the app. Fires **exactly one** `unitNotify` per refused app per run. Nil admission set (periodic status refresh) → decides fresh, which is still once per app per sweep. Wiring pinned by an **AST walk** in `TestAdmission_IsWiredIntoEveryProductionWriteLeg`, not `strings.Contains` |
|
||||
| `Manager.floorVerdict` + `FloorUsedPercent`/`FloorFreeGiB` / `ErrCaptureFloor` / `floorReason` (R-165 B2 v0.192.0, size term R-181 v0.193.0) | controller/internal/backup/recovery_unit.go | `(*UnitSpace, estGiB float64) (*UnitSpace, floorReason)` | The pure two-question predicate behind `admitApp`: is the filesystem already below the reserve (`floorHeadroom`), and would THIS app's write take it below (`floorSize`)? | **Headroom is about the FILESYSTEM, never a per-unit cap** — a size cap is R-163 rebuilt inside one volume; the size term bounds the *delta*, not the unit. **REFUSES, never deletes:** nothing here is generational (a unit is one fixed path per app, a DB dump one fixed name), so pruning could only destroy a DIFFERENT app's only local copy — **never repurpose `pruneStalePrimaryDirs`**, which removes ORPHANED dirs from an app that moved drives and has no notion of age. Two terms (97% / 1 GiB) in `fillwatch`'s shape, deliberately BEYOND its critical band (95% / 2 GiB) so the customer is always warned first — pinned by `TestFloorSitsBelowTheCriticalWarningBand`. **`estGiB == 0` degrades to headroom-only on purpose** — refusing an app with no history makes the FIRST backup the one that can never happen. A nil reading neither refuses nor warns (§8.4). Inject `unitSpaceFn` in tests rather than manufacturing occupancy on a real disk |
|
||||
| `fillwatch.Watcher` (`New`/`SetNotify`/`Check`) (R-167, v0.191.0) | controller/internal/fillwatch/fillwatch.go | `(statePath, logger, targetsFn, usageFn)` → `Check() error` | THE customer fill warning — warns BEFORE a filesystem fills, per FILESYSTEM (never per app: one full disk holding ten apps would fire ten times) | Emits the **pre-existing** `disk_warning`/`disk_critical` pair, which was allowlisted + copy'd + default-enabled with **no producer in any repo** until now — do NOT mint a new type beside it. **Two threshold terms, whichever trips first** (85% / 5 GiB; critical 95% / 2 GiB) because a percentage alone lies at both ends of this fleet's size range. **Edge-triggered on ESCALATION ONLY**, state persisted; de-escalation is silent and re-arms. Hysteresis dead zone between clear (75% / 7 GiB) and warn — pinned by `TestThresholdsKeepTheirHysteresisGap`. **A nil usage read is NEVER a warning** (§8.4). The hub has **no `customerMessages` entry** for either type on purpose — an entry would override the dynamic message and discard the drive label + free space |
|
||||
|
||||
### Scheduler / time / UI
|
||||
|
||||
| Symbol | File | Short signature | Use for | Gotchas |
|
||||
|---|---|---|---|---|
|
||||
| `Scheduler.Every` / `Daily` | controller/internal/scheduler/scheduler.go | `(name, interval/"HH:MM", fn)` | ALL background jobs | Daily is Europe/Budapest, DST-safe (`nextDailyRun` avoids Add(24h)); register in main.go block (§5) |
|
||||
| `getBudapestLocation` | controller/internal/scheduler/scheduler.go | `() *time.Location` | Local-time math | web has its own `getTimezone` (§6) |
|
||||
| `Scheduler.UpdateDaily` | controller/internal/scheduler/scheduler.go | `(name, "HH:MM") bool` | Retime a daily job at runtime (no restart) | Per-job buffered `resched` chan + select case in `runDailyJob`; false (WARN) on invalid time / unknown-or-non-daily name; read `Schedule` under the mutex in the loop |
|
||||
| `backupwindow.*` (LegTimes / GateWindow / EffectiveWindow / ParseHHMM / FmtHHMM / Valid) | controller/internal/backupwindow/backupwindow.go | pure `string`↔`int` | Backup-window arithmetic (v0.168.0) | Offsets (W+60m/W+105m, gate W+2h..W+6h) are CONSTANTS — derived, never stored; wrap-safe modulo 1440; `EffectiveWindow(settings, yaml)` = settings>yaml>"02:30" |
|
||||
| `getBudapestLocation` | controller/internal/scheduler/scheduler.go | `() *time.Location` | Local-time math | web has its own `getTimezone` (§6); quiesce has its own `budapestLocation` (window gate) — 3rd copy, see §6 |
|
||||
| `Server.templateFuncMap` | controller/internal/web/funcmap.go | template.FuncMap | ALL template functions | `stateColor` outputs v2 suffixes `run/progress/warn/neutral/off`; stopped = NEUTRAL not red (operator-approved); `stateLabel` copy is frozen byte-identical (unit-tested) |
|
||||
| `timeAgoStr` | controller/internal/web/funcmap.go | `(s RFC3339 string) string` | Ago-format for STRING timestamps | Exists because `timeAgo(time.Time)` 500'd on strings (v0.93 bug) |
|
||||
| `Server.baseData` / `executeTemplate` | controller/internal/web/handlers.go + server.go | page-data plumbing | New pages | baseData injects nav/alerts/version; templates must pass `controller/scripts/template_id_gate.py` + `controller/scripts/emoji_gate.py` |
|
||||
@@ -167,6 +218,8 @@
|
||||
| Platform split | controller/internal/system/mounts_linux.go + mounts_other.go | `_linux.go`/`_other.go` twins; other = permissive no-op stubs for dev on Windows |
|
||||
| Debounced trigger + status (REFUSE-style — a too-soon fire is refused/lost) | controller/internal/sync/sync.go | `TriggerSync` 30s debounce, `Status()` snapshot struct, post-sync hook fan-out |
|
||||
| Coalescing trigger (trailing edge — a burst collapses but the LAST state always fires) | controller/internal/report/trigger.go | buffered-1 chan + non-blocking `Fire()` + single worker (quiet window → drain → min-interval → fire once); shape from hub `wgsync/reconciler.go` |
|
||||
| Detached job + status poll (single-flight, phase strings) | controller/internal/web/storage_init_job.go | acquire/release/set/**deep-copied** snapshot; phases mapped to Hungarian in the template; 1–3 s poll; terminal state **PROBED, not inferred**. Clones: `netstorage_job.go`, `samba_ensure_job.go` (v0.147.0). **Five of these now exist and agree on nothing — R-45 will unify them; prefer extending an existing one over a sixth** |
|
||||
| Streaming subprocess progress | controller/internal/backup/offbox_progress.go | `offboxStreamRunner` seam (stdout scanned line-by-line, stderr buffered, output tail-bounded) + a PURE line parser + a mutex-guarded published snapshot. Traps it encodes: a source reporting nothing is **normal** (restic sends 0 bytes for a whole incremental run) and the progress source may only update on unit completion — degrade bytes → files → current item + elapsed, never fake a percentage |
|
||||
| Post-start async verification | controller/internal/stacks/manager.go `logPostStartStatus` | goroutine + sleep, INFO log, never blocks/fails the operation |
|
||||
| Startup wiring order | controller/cmd/controller/main.go | init-only setters (`SetStackProvider` M2 contract: exactly once, before scheduler/HTTP), scheduler registration block |
|
||||
|
||||
@@ -193,7 +246,28 @@
|
||||
| `Server.agentLogsFn` (func seam) | controller/internal/web/server.go | nil → `agentClient().DebugLogs` (agent GET /debug/logs) | injected in controller/internal/web/observability_test.go (incl. the pre-0.83 typed-404 notice path) |
|
||||
| `escrowAgent` + `Server.escrowAgentFn/escrowStageFn/escrowStaleFn` | controller/internal/web/escrow_handlers.go (+ server.go fields) | `*agentapi.Client` / `PushOffboxPasswordForEscrow` / `report.EscrowAutoConfirmer.StaleBlob` (SetEscrowStale) | `fakeEscrowAgent` + fn injections in escrow_wizard_test.go — call-ORDER assertions (stage BEFORE trigger) + agent-never-called gates. The claim leg is the ONLY surface R crosses: no-store, never logged, never templated |
|
||||
| `offboxCeremonyWaitState` + `escrowCeremonyGraceWindow` | controller/internal/web/handlers.go | pure pick: (awaiting, timedOut) from `OffboxTarget.{EscrowState,CeremonyCompletedAt}` — the v0.138.0 "megerősítésre vár" card. Stamp SET on claim (escrow_handlers.go), CLEARED on the flip (main.go Flip + offbox_handlers.go manual confirm) | escrow_wait_state_test.go truth table (escrowed/unstamped/unparseable → plain CTA; boundary via `>=`) |
|
||||
| `Manager.sambaUpFn` / `sambaPasswdFn` / `sambaRunFn` (func seams) | controller/internal/stacks/manager.go (fields) + samba.go | nil → `composeUp` / `docker exec smbpasswd` (STDIN) / `containerRunning("felhom-samba")` | injected in controller/internal/stacks/samba_test.go — the idempotency test asserts the up-seam is called **zero** times when config is unchanged; the passwd seam means no unit test ever handles a real secret or touches docker |
|
||||
| `Manager.sambaUpFn` / `sambaPasswdFn` / `sambaRunFn` / `sambaAddrFn` (func seams) | controller/internal/stacks/manager.go (fields) + samba.go | nil → `composeUp` / `docker exec smbpasswd` (STDIN) / `containerRunning("felhom-samba")` / `docker exec felhom-samba ip -4 -o addr show eth0` | injected in controller/internal/stacks/samba_test.go — the idempotency test asserts the up-seam is called **zero** times when config is unchanged; the passwd seam means no unit test ever handles a real secret or touches docker. **`sambaRunFn` has an EXPORTED setter (`SetSambaRunProbe`)** — internal/web's status-contract tests need a live-container world from another package. `sambaAddrFn` backs `SambaLANAddress()` (v0.151.0); its parse is separately pinned in samba_lanaddr_test.go and it returns "" on any failure — the page omits a line rather than printing a wrong address |
|
||||
| `Manager.SambaLANAddress()` | controller/internal/stacks/samba.go | `() string` — the guest's LAN IPv4 for the Megosztás connect card (v0.151.0, S-2) | Read from the SAMBA container's netns (`network_mode: host`), never `net.InterfaceAddrs()` — the controller is on a docker BRIDGE and would answer 172.x (the same trap `setup.DetectLocalIPs` needs `HOST_IP` for). **NEVER cache/persist it** — the guest holds it by DHCP (S-5); callers re-derive per render. `""` = omit the line |
|
||||
| `Server.sambaAddrFn` (func seam) | controller/internal/web/server.go (field) + sharing_handlers.go `sambaLANAddress()` | nil → `stackMgr.SambaLANAddress()` | The web-side half of the connect card. Tests inject a COUNTED fn — the fresh-per-render assertion is what stops anyone memoizing a DHCP lease |
|
||||
| `Manager.guestNetExecFn` (func seam) + `GuestGateway()` / `GuestNetSnapshot()` | controller/internal/stacks/manager.go (field) + guestnet.go | nil → `docker exec felhom-samba <args>` — ONE seam for all R-66 guest-netns reads (route/link/addr/resolv.conf); tests script canned outputs per argv | guestnet_test.go. **The netns door rule:** the controller's OWN netns is the docker bridge, so any in-process read (`net.Interfaces`, `/proc/net/route`, its own `/etc/resolv.conf` = 127.0.0.11) is the S-2 wrong answer — guest-net reads MUST go through the samba (`network_mode: host`) exec door. Megosztás off ⇒ door closed ⇒ "" / per-item error strings; NEVER substitute an in-process value. Same S-5 law as SambaLANAddress: live per render, never cached/persisted. Parsers (`parseDefaultRoute`, `parseGuestInterfaces`, `parseResolvConf`) are pure + separately pinned |
|
||||
| `buildFileBrowserPaths` + `fbPathDeps` (R-67, v0.160.0) | controller/internal/web/handlers.go | pure assembly of one FileBrowser sync pass: (mount lines, config source paths) from the registry, with per-kind gates | filebrowser_network_test.go. **Two storage classes, two DIFFERENT gates:** drives keep the drive-absent gate + userdata scoping + skeleton (byte-identical to pre-R-67 — tested); network shares bind the share ROOT `:rslave` with the STUB gate instead (`classifyFSPath`; stub ⇒ excluded from mounts AND sources — an exposed stub swallows uploads the real mount later shadows; idle autofs / unknown ⇒ include, fail open). NEVER call `EnsureUserdataSkeleton` toward a network path (red-proven); never force-wake an idle trigger in the sync (doctrine) |
|
||||
| `Settings.RefuseAsAppNamespace` (R-108, v0.187.0) | controller/internal/settings/settings.go | `(path) (refuse bool, hungarianReason string)` — may an app's DATA NAMESPACE live here? | **THE single predicate for every placement surface** (deploy POST `api/router.go`, per-app migrate list + `handleStorageMigrateApp`, `handleStorageDecommission` mode=migrate TARGET). **Network storage is refused** because an app's namespace root IS its backup root (`namespaceRoot` returns a non-system drive path as-is → `<HDD_PATH>/backups/primary/<stack>/`), and on a share that lands inside FileBrowser's share-ROOT `download:true` bind — which CANNOT be narrowed (R-67 `:rslave` = automount wake; and apps on a share store at `<share>/<app>`, so there is no `userdata/` to scope to and creating one would write Felhom convention onto a customer's NAS). **DISTINCT from `refuseNetworkLifecycle`** — that asks "may a DRIVE lifecycle op run on this path" and is applied to the op's SUBJECT; this asks "may an app live here" and is applied to a placement TARGET. Migrate needs BOTH. **FAILS CLOSED:** `/mnt/felhom-drives` holds both kinds, so a path prefix cannot classify — `Kind` exists only on a REGISTERED path, therefore an unregistered path under that root is un-classifiable and REFUSES. Empty path = SSD-resident = allowed; nil receiver refuses. network_app_namespace_test.go, 4 red-proofs |
|
||||
| `Server.guestGatewayFn` / `guestNetFn` (func seams) | controller/internal/web/server.go (fields) + sharing_handlers.go accessors | nil → `stackMgr.GuestGateway` / `stackMgr.GuestNetSnapshot` | network_card_test.go — the counted-fn freshness test (2 renders ⇒ 2 resolves) is what stops anyone memoizing a DHCP lease; the Hálózati név row is gated on `smb.Enabled` (red-proven: gate dropped ⇒ \\FELHOM rendered while samba is down) |
|
||||
| `sambaEnsureState.consumeIfRunning()` | controller/internal/web/samba_ensure_job.go | serve-once `snapshot()` for terminal `running` only | `/sharing/status` carries a job EDGE (`phase`) and a service LEVEL (`running`) in one envelope — never let a level reach the phase channel, and never re-serve a consumed edge: the client answers `phase=="running"` with `location.reload()`, so both mistakes produce an infinite page reload (S-1/S-4, DIAG-sharing-2026-07-20.md). `failed`/`needs_password`/in-flight are NOT consumed |
|
||||
| `infra.SambaHostInterface` | controller/internal/infra/samba.go | the guest LAN nic name (`eth0`) | Single source for smb.conf's `interfaces =`, the container's `FELHOM_IFACE`, and the LAN-address read — if they name different nics, the service and the address the page prints drift apart |
|
||||
| `Manager.sambaImgFn` (func seam) | controller/internal/stacks/manager.go (field) + samba.go | nil → `docker image inspect <infra.SambaImage>` | drives the 4b card's pulling-vs-starting decision, which MUST be taken before `compose up` (afterwards the image is always present) |
|
||||
| `Manager.offboxStreamRunner` + `SetOffboxStreamRunner` | controller/internal/backup/offbox_progress.go | nil → `defaultOffboxStreamRunner` (real `restic`, stdout scanned live) | streaming sibling of `offboxRunner`; fakes emit canned `--json` status lines in offbox_progress_test.go, so the whole progress path runs with no restic, network or repo |
|
||||
| `Manager.offsitePreDumpFn` + `SetOffsitePreDumpFn` (R-44, v0.148.0) | controller/internal/backup/offbox_reconstitute.go (seam) + offbox.go (call site) | nil → `runDBDumpsInternal` under the SAME running flag | THE dumps-before-capture ordering seam. Extracted so the order is observable without Docker/restic — an ordering guarantee no test can see is one refactor from silently reverting to the DIAG-immich-restore-2026-07-19 behaviour. Red-proof: moving the capture first yields `[capture dump]` |
|
||||
| `Manager.offboxFullPlaceCopier` + `SetOffboxFullPlaceCopier` (R-43) | controller/internal/backup/offbox_reconstitute.go | nil → `rsyncRestoreOverwrite` (`-a --itemize-changes`; **no** `--ignore-existing`, **no** `--delete`) | **TRAP: do NOT reuse `offboxPlaceCopier` here.** The two copiers have OPPOSITE semantics for an existing file — `--ignore-existing` is exactly what a full restore must not do, and conflating them is how a missing-only merge came to be labelled a restore. Never `rsyncMirror` (`--delete`) in any restore direction |
|
||||
| `Manager.safetyDumpFn` + `SetSafetyDumpFn` (R-43) | controller/internal/backup/offbox_reconstitute.go | nil → `DumpOne` | the pre-restore undo. Invariant: the `pre-restore-`-prefixed dump must be verified ON DISK before anything is stopped/overwritten/replayed; failure ⇒ refuse with zero changes. Red-proof requires removing BOTH guards (the `err != nil` return and the `os.Stat`) — removing one leaves the other holding |
|
||||
| `reimportDBDumpsFrom(ctx, stack, dumpDir)` | controller/internal/backup/restore_db.go | explicit-dir sibling of `reimportDBDumps` (which passes `AppDBDumpPath`) | offsite reconstitution replays from the SCRATCH unit: the live unit is deliberately never overwritten, so replaying from it would replay the current DB over itself and restore nothing |
|
||||
| The DB-only replay window (R-47, v0.153.0) | controller/internal/backup/{offbox_reconstitute,restore_unit}.go | both restore paths: stop → place/volumes → `StartStackServices(dbServices)` → replay → `StartStack` (full) | **THE ordering invariant.** Replaying while the whole stack is up lets the app's own schema management race the dump — measured at 2 s on 2026-07-19 (H4), replay aborted `already exists`. Fail-closed: a dump with NO identifiable DB service refuses BEFORE the first mutation. Every exit from the window (replay error, DB-only start error) MUST still do a best-effort full start, or a failed restore becomes an outage. `hasReplayableDump` excludes `pre-restore-` safety dumps — counting them would arm the window for an app with nothing to replay |
|
||||
| `Manager.OffsiteScratchPair` / `OffsitePairInfo` | controller/internal/backup/offbox_reconstitute.go | reads the restored scratch unit's manifest (`offsite_run_id` / `dumps_at`) + the R-44 sniff | the confirm-dialog honesty surface. All warn-level: a pre-v0.148 (unstamped) pair and an empty-looking dump are SURFACED, never blocked — a false positive that refused a legitimate restore would be worse than the skew |
|
||||
| `appbackup.DumpValidation.LooksEmpty` (R-44 sniff) | controller/internal/appbackup/dbdump.go | computed in ValidateDump's existing single pass; `userTableNames` is EXACT-match | size and table count are both useless as emptiness heuristics (the 2026-07-19 dump: 52MB, 60+ tables, zero users — all geodata). **TRAP: never widen to a substring match on "user"** — it would flag `user_metadata` / `album_user` / `user_audit` on every healthy single-user box. A row wider than the read buffer still counts as a row |
|
||||
| `Manager.execFn` (func seam) + `restartPolicyLookup` / `inspectRestartPolicyFn` (R-51, v0.156.0) | controller/internal/stacks/manager.go | nil → real `exec.Command` / `docker inspect -f {{.HostConfig.RestartPolicy.Name}}` | `scriptedDocker` in controller/internal/stacks/degraded_test.go drives the WHOLE production path (docker ps → aggregateState → docker inspect) — an aggregateState-only test proves the function, not the caller. Policy answers are cached per container+state and pruned to the live `docker ps` set; a FAILED inspect is deliberately never cached (a hiccup must not pin a container to "unknown") and reads as SUPERVISED, i.e. fail-closed — the opposite of `IsDownState`'s fail-open, because there the state is ambiguous while here a member is known dead |
|
||||
| `bootrecon.StackProvider` (R-52, v0.156.0) | controller/internal/bootrecon/bootrecon.go | `*stacks.Manager` (GetStacks/StartStack/RefreshStatus) | `fakeStacks` counts StartStack per app; the load-bearing assertion is the NEGATIVE — a zero-container stack (a UI Stop = `compose down` = containers removed) must record **0** starts, while a boot orphan (containers present, Exited) records exactly 1. `Reconciler.sleep` is injected so the 30 s gap costs nothing |
|
||||
| `bootReconcileFn` + `runBootReconcile` (package-main seam, v0.156.0) | controller/cmd/controller/main.go | `bootrecon.New(mgr, logger).Run` | controller/cmd/controller/bootrecon_wiring_test.go. **The wiring itself is asserted by an AST walk** over `func main()`, not a `strings.Contains` — the substring version passed its own red-proof because a commented-out call still contains the string. Comments are not callers |
|
||||
| `classifyRunStates` (pure fix-3 derivation, v0.164.0) | controller/cmd/controller/main.go | `([]stacks.Stack, quiesced, failedRestart map[string]bool, now time.Time)` → `(dead []web.DeadApp, states []notify.AppRunState)` | classify_runstates_test.go. **THE single fix-3 rule: down = `(IsDownState(st.State) || st.CrashLooping(now)) && !userStopped && !quiesced`.** C9-F2 (v0.183.0) added the crash-loop term: `restarting` is NOT in `IsDownState` and must not be — adding it alarms on every deploy and update fleet-wide — so a SUSTAINED restarting run (`stacks.crashLoopAfter` = 5 m, above the 120 s deploy timeout, Mealie's 60 s start_period AND R-97b's 180 s grace) becomes down instead. `now` is injected so the threshold is a testable contract. A deliberate UI stop (`compose down` → zero containers → StateStopped, I1) must not alarm — banner OR email — while faults (Exited/Degraded) alarm byte-identically; I2 (P2 census: all catalog services `unless-stopped`) is why a crash never rests at stopped. **Do NOT touch `IsDownState`** (other callers rely on stopped=down) and do NOT filter in `buildDeadAppAlerts`/`NotifyAppStartFailures` — one derivation point. If I1 or I2 changes, revisit the suppression |
|
||||
| `report.SetPendingControllerLog` / `SetControllerLogSource` | controller/internal/report/selftail.go | ACK-armed consume-once self-log pull (the logtail.go shape) | selftail_test.go; source = `logBuffer.Lines`, wired once in main.go |
|
||||
| `util.ParseVersion` / `util.Version.Compare` | controller/internal/util/version.go | THE one semver comparator (house rule: never a second) — selfupdate aliases it; agentapi's MinAgent comparison uses it | rejects pre-release/dev/latest (callers fall back, never trust); numeric compare (0.100 > 0.81) |
|
||||
| `agentapi.AgentVersionReporter` + `featureMinAgent` | controller/internal/agentapi/features.go | version-first Supports (v0.82.0 header channel); probe = fallback for header-less agents | a coupled feature adds BOTH a featureProbes row AND a featureMinAgent row; v0.116.0: `SupportsWithSource` also reports HOW the verdict was reached (version/probe-cache/probe) for the gate log line |
|
||||
@@ -211,7 +285,7 @@
|
||||
| `dumpVolumesSafe` (func seam) | controller/internal/backup/backup.go | nil → real `DumpAppVolumesSafe` | injected in controller/internal/backup/volume_dumps_test.go (gating tests without Docker) |
|
||||
| `generateSecret` (func seam) | controller/internal/backup/backup.go | `stacks.Manager.GenerateSecretForField` via `SetSecretGenerator` (main.go) | injected in controller/internal/backup/restore_secrets_gen_test.go |
|
||||
| `restoreFilesCopier` (func seam) | controller/internal/backup/backup.go | nil → real `rsyncRestoreMissing` | injected in controller/internal/backup/tier2_restore_test.go (orchestration without rsync) |
|
||||
| `tier2Mirror` (func seam) | controller/internal/backup/backup.go | nil → real `rsyncMirror` | both RunTier2 rsync legs; injected in controller/internal/backup/tier2_appdata_test.go (resolve→mirror without rsync) |
|
||||
| `tier2Mirror` (func seam) | controller/internal/backup/backup.go | nil → real `rsyncMirror` | both RunTier2 rsync legs; injected in controller/internal/backup/tier2_test.go (resolve→mirror without rsync) |
|
||||
| `migSeams.resolveNames` (func seam) | controller/internal/stacks/migrate.go | nil → real `ResolveAppDataDirNames` (compose-derived) | injected in controller/internal/stacks/migrate_fs3_test.go (F-S3 appdata dir-name resolution) |
|
||||
|
||||
Cross-repo edges:
|
||||
@@ -229,7 +303,7 @@ Cross-repo edges:
|
||||
- **Docker volume tar streaming (v0.125.0)**: `appexport.dockerExec` (seam, package var) + `withVolumeHelper`/`exportVolumeTar`/`importVolumeTar` — stream volume content via `docker cp` through a stopped helper container. NEVER `docker run -v <controller-local path>` — the daemon resolves `-v` host-side and strands the data when the controller is containerized (the v0.124.0 HIGH finding); `controller/scripts/docker_run_volume_path_gate.py` enforces (every `"-v"` allowlisted with its WHY).
|
||||
- **Guarded file download (v0.124.0)**: `handler_export_download.go` — the canonical shape for streaming a server-side file to the browser: accept a BASENAME only (shape regexp + no separators/`..`), `filepath.Join` then assert `filepath.Dir(path) == dir`, `io.Copy` (never ReadAll), `Content-Disposition: attachment`, remove after a successful stream, TTL sweep (`sweepFabDownloads(dir, now, maxAge, logger)` — now injected for tests). Red-proof the guard by loosening to prefix-matching (the `..` case must fail).
|
||||
- **Backups sub-page data**: `backupsCommonData(page, title, r)` + `backupsOffboxData(data)` (handlers.go) — the ONLY builders for the four `/backups*` pages; a new backups section extends these, never re-derives in a page handler. (The one-shot v0.124.0 move gate `backups_split_move_check.py` was retired in v0.126.0.)
|
||||
- **App-list row (v0.126.0)**: `app_list_row`/`app_list_row_end` in `templates/app_row.html` is THE canonical list pattern — icon+name(+secondary) left, caller action block right; open with `dict "Slug" ... "Name" ...` (optional `Secondary`/`RowClass`/`Href`/`FallbackIcon`), close with `app_list_row_end`. Do NOT hand-roll app rows — `scripts/app_row_dedup_gate.py` enforces single-sourcing (the backups_apps expander header is the one allowlisted aligned copy). Infra display identity: `inframeta.go` map + `infraMeta` func (filebrowser is the only Linked stack).
|
||||
- **App-list row (v0.126.0)**: `app_list_row`/`app_list_row_end` in `controller/internal/web/templates/app_row.html` is THE canonical list pattern — icon+name(+secondary) left, caller action block right; open with `dict "Slug" ... "Name" ...` (optional `Secondary`/`RowClass`/`Href`/`FallbackIcon`), close with `app_list_row_end`. Do NOT hand-roll app rows — `controller/scripts/app_row_dedup_gate.py` enforces single-sourcing (the backups_apps expander header is the one allowlisted aligned copy). Infra display identity: `inframeta.go` map + `infraMeta` func (filebrowser is the only Linked stack).
|
||||
- **Consequential-action confirm (LIGHT)**: `felhomConfirm(el, question, onYes)` in layout.html (v0.123.0) — the trigger swaps in place to "kérdés + Igen/Mégse"; form buttons opt in with `data-confirm="…"` (delegated listener, `requestSubmit` keeps formaction/name-value). NEVER native `confirm()`/`prompt()` (OS-modals freeze browser automation — drill F-11; `native_confirm_gate.py` enforces). Heavy destructive flows keep the `.confirm-overlay` `openDialog` pattern.
|
||||
- **New hub event**: typed `Notify*` wrapper on Notifier + hub allowlist entry (cross-repo).
|
||||
- **New app integration**: `integrations.Manager.RegisterHandler` with `IntegrationKey(provider, target)`.
|
||||
@@ -247,7 +321,7 @@ Cross-repo edges:
|
||||
| dir-size ×6 | controller/internal/stacks/delete.go `getDirSizeBytes`/`getDirSizeHuman`; controller/internal/backup/tier2.go `dirSizeBytes` (du -sb); controller/internal/appexport/estimate.go `dirSize`+`duBytes`; controller/internal/appexport/export.go `calcDirSize`; controller/internal/web/handlers.go `dirSizeHuman` |
|
||||
| timeAgo switch body ×2 | controller/internal/web/funcmap.go `timeAgo` vs `timeAgoStr` (identical formatting logic) |
|
||||
| CSRF ×2 | controller/internal/web/csrf.go (session HMAC) vs controller/internal/setup/csrf.go (cookie double-submit) — intentional (pre-auth wizard) but unlabeled |
|
||||
| Budapest timezone loader ×2 | controller/internal/scheduler/scheduler.go `getBudapestLocation` vs controller/internal/web/funcmap.go `getTimezone` |
|
||||
| Budapest timezone loader ×3 | controller/internal/scheduler/scheduler.go `getBudapestLocation` vs controller/internal/web/funcmap.go `getTimezone` vs controller/internal/quiesce/quiesce.go `budapestLocation` (v0.168.0 window gate — Budapest wall-clock, kept local to avoid a scheduler↔quiesce import edge) |
|
||||
| JSON writers ×5, 3 envelope shapes | api `writeJSON`; web `writeDiskJSON`, `jsonResponse`/`jsonError`, `writeDebugJSON` |
|
||||
| Safe-name validators ×4 | controller/internal/web/validate.go `validStackName`; controller/internal/api/router.go `validStackParam` (same body — api↔web import cycle); controller/internal/backup/offbox.go `isSafeStackName`; controller/internal/appexport/validate.go `ValidateSegment` (strictest) |
|
||||
| DB wait/import ×2 | controller/internal/appbackup/dbdump.go `waitDBReady`/`ImportDump` vs controller/internal/appexport/restore.go `waitForDB`/`importDBDump` |
|
||||
|
||||
+13
-11
@@ -14,7 +14,9 @@ destructive section (7) is last and gated.
|
||||
bootstrap-managed. Public dashboard/API: **https://felhom.demo-felhom.eu** (no dashboard password →
|
||||
the API is open; drive it via the PUBLIC URL, not the container IP).
|
||||
- **Versions at writing:** controller **v0.60.0**, agent **v0.30.0**, hub v0.11.0.
|
||||
- `SSH=/c/Windows/System32/OpenSSH/ssh.exe`; host root via SSH alias `felhom-pve`; `export MSYS_NO_PATHCONV=1` for `pct exec`.
|
||||
- Run from DooPlex (192.168.0.180); host root via SSH alias `felhom-pve` — plain `ssh felhom-pve`.
|
||||
(Legacy Windows workstation: needed `SSH=/c/Windows/System32/OpenSSH/ssh.exe` and
|
||||
`export MSYS_NO_PATHCONV=1` for `pct exec`.)
|
||||
- **Findings log:** record every observation (✓/✗ + notes on UX friction, latency, confusing labels,
|
||||
error handling) in a new `REPORT-e2e-live-drive-<date>.md`. Each step says what "good" looks like and
|
||||
what to watch for.
|
||||
@@ -27,12 +29,12 @@ destructive section (7) is last and gated.
|
||||
|
||||
1. Controller healthy + version:
|
||||
- `curl -s https://felhom.demo-felhom.eu/api/health` → `{"ok":true,...}`.
|
||||
- `$SSH felhom-pve "pct exec 9201 -- docker ps --filter name=felhom-controller --format '{{.Image}} {{.Status}}'"` → `:0.60.0 Up ... (healthy)`.
|
||||
- `ssh felhom-pve "pct exec 9201 -- docker ps --filter name=felhom-controller --format '{{.Image}} {{.Status}}'"` → `:0.60.0 Up ... (healthy)`.
|
||||
- Dashboard loads (Hungarian UI), no error banners.
|
||||
2. Agent healthy + version: `$SSH felhom-pve "systemctl is-active felhom-agent; /usr/local/bin/felhom-agent --version"` → `active`, `0.30.0`.
|
||||
2. Agent healthy + version: `ssh felhom-pve "systemctl is-active felhom-agent; /usr/local/bin/felhom-agent --version"` → `active`, `0.30.0`.
|
||||
3. Headroom (deploys pull images — **bound to ≤3 small apps**):
|
||||
- Docker-data volume free: dashboard storage bars, or `$SSH felhom-pve "pct exec 9201 -- df -h /var/lib/docker /"`. Need comfortably above the v0.58 reserve (`max(5GB,10%)`) or deploys will be gated **507**.
|
||||
- RAM: `$SSH felhom-pve "pct exec 9201 -- free -h"`.
|
||||
- Docker-data volume free: dashboard storage bars, or `ssh felhom-pve "pct exec 9201 -- df -h /var/lib/docker /"`. Need comfortably above the v0.58 reserve (`max(5GB,10%)`) or deploys will be gated **507**.
|
||||
- RAM: `ssh felhom-pve "pct exec 9201 -- free -h"`.
|
||||
- Disk list sane: `curl -s https://felhom.demo-felhom.eu/api/disks` → felhom-usb (user-data, data_bearing), local/local-lvm (system), felhom-pbs (backup).
|
||||
4. Record current deployed apps (so cleanup is unambiguous): dashboard "Alkalmazások", or `curl -s https://felhom.demo-felhom.eu/api/stacks/rescan` then the stacks list. (actualbudget is expected already deployed.)
|
||||
- **Good:** all green/healthy; free space well above reserve. **Watch for:** any app stuck "Telepítés alatt" (deploying) from a prior run — note and resolve before starting.
|
||||
@@ -49,7 +51,7 @@ Pick **two small apps** not currently deployed (suggest: `vikunja`, `mealie` —
|
||||
- **Good:** progresses config→containers→health; ends `running`/healthy within ~120s; the card flips to deployed; no "Telepítés" button reappears mid-pull (in-memory Deployed=true during pull).
|
||||
- **Watch for:** stuck at a step, health-probe never going green (check the app's healthcheck tool exists), confusing Hungarian labels, the deploy gate returning **507** (insufficient Docker-data headroom — expected if low on space; note the banner wording).
|
||||
2. Deploy app #2; same checks.
|
||||
3. Confirm on disk the durable record is correct (CTRL-T2-1, happy case): `$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → `deployed: true` (only after success).
|
||||
3. Confirm on disk the durable record is correct (CTRL-T2-1, happy case): `ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → `deployed: true` (only after success).
|
||||
- **Good:** `deployed: true` on disk after a successful deploy. **Watch for:** secrets appearing in plaintext in app.yaml (they must be `enc:`-prefixed — H10/encryption check).
|
||||
|
||||
---
|
||||
@@ -59,10 +61,10 @@ Pick **two small apps** not currently deployed (suggest: `vikunja`, `mealie` —
|
||||
Goal: prove a crash during the image-pull window leaves the stack **NOT-deployed and redeployable**, not ghost-stuck.
|
||||
|
||||
1. Pick a **third app with a non-trivial image pull** (so the pull window is a few seconds — e.g. `paperless-ngx` if space allows, else `mealie`). Start the deploy (UI Telepítés or API POST), and **immediately** — while it is still pulling (status `deploying`, before `running`) — kill the controller:
|
||||
- `$SSH felhom-pve "pct exec 9201 -- docker kill felhom-controller"` **[operator: time this during the pull]**
|
||||
- `ssh felhom-pve "pct exec 9201 -- docker kill felhom-controller"` **[operator: time this during the pull]**
|
||||
- The bootstrap service (`felhom-controller-bootstrap.service`) restarts it within seconds. Confirm back up: `curl -s https://felhom.demo-felhom.eu/api/health`.
|
||||
2. After restart, check the stack state:
|
||||
- On disk: `$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → **`deployed: false`** (transitional — the fix).
|
||||
- On disk: `ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller cat /opt/docker/stacks/<app>/app.yaml | grep deployed"` → **`deployed: false`** (transitional — the fix).
|
||||
- UI/API: `GET /api/stacks/<app>` → state `not_deployed` (the card shows **Telepítés**, not a ghost "deployed").
|
||||
3. **Redeploy** the same app — it must be **allowed** (no "already deployed; use update instead" refusal) and complete normally.
|
||||
- **Good:** post-crash the app reads not-deployed and redeploys cleanly. **PRE-FIX behaviour (must NOT occur):** app.yaml `deployed: true` with no containers, and redeploy refused — that's the ghost-stuck regression the fix removes.
|
||||
@@ -78,9 +80,9 @@ Goal: prove a crash during the image-pull window leaves the stack **NOT-deployed
|
||||
- **Good:** imports, recreates the stack, data restored; fail-closed data-key gate honored if the app has a data-encrypting key.
|
||||
3. **Negative — path traversal (CTRL-001):** craft a hostile `.fab` and confirm it is **rejected at parse**, not written.
|
||||
- Build a minimal bundle whose `manifest.json` has `"app_name":"../evil"` (and/or an `hdd_subdirs` / `volume_names` entry with `../`). Place it under a registered `exports/` dir on the host:
|
||||
`$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller sh -c 'ls /mnt/felhom-usb/exports/'"` to find the dir.
|
||||
`ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller sh -c 'ls /mnt/felhom-usb/exports/'"` to find the dir.
|
||||
- Attempt import of the hostile bundle.
|
||||
- **Good (the fix):** import **fails immediately** with a manifest/validation error; **no directory is created outside the stacks dir** (verify: `$SSH felhom-pve "pct exec 9201 -- docker exec felhom-controller ls -la /opt/docker/evil /etc/evil 2>/dev/null"` → nothing). **PRE-FIX (must NOT occur):** a dir/file written outside `/opt/docker/stacks/`.
|
||||
- **Good (the fix):** import **fails immediately** with a manifest/validation error; **no directory is created outside the stacks dir** (verify: `ssh felhom-pve "pct exec 9201 -- docker exec felhom-controller ls -la /opt/docker/evil /etc/evil 2>/dev/null"` → nothing). **PRE-FIX (must NOT occur):** a dir/file written outside `/opt/docker/stacks/`.
|
||||
- **Watch for:** the error message clarity (does the UI explain why it was rejected?).
|
||||
|
||||
---
|
||||
@@ -105,7 +107,7 @@ Re-confirm the two refusals proven on 2026-06-13. **Do NOT send a matching confi
|
||||
- **Good:** `formatted:false`, `needs_confirmation:true`, **HTTP 409**; no mkfs.
|
||||
2. **Refusal B — wrong durable_id:** same call with `"confirmed":true,"durable_id":"byid:wwn-0xDEADBEEF-DOES-NOT-EXIST"`.
|
||||
- **Good:** `formatted:false`, refused **409**; a non-matching confirmation does not authorize a wipe.
|
||||
3. **Data-safety assertion:** `$SSH felhom-pve "findmnt /mnt/felhom-usb -o TARGET,SOURCE,FSTYPE; pct exec 9201 -- docker exec felhom-controller sh -c 'df -h /mnt/felhom-usb'"` → still mounted, used space unchanged.
|
||||
3. **Data-safety assertion:** `ssh felhom-pve "findmnt /mnt/felhom-usb -o TARGET,SOURCE,FSTYPE; pct exec 9201 -- docker exec felhom-controller sh -c 'df -h /mnt/felhom-usb'"` → still mounted, used space unchanged.
|
||||
4. **Happy-path destructive wipe** = **[HUMAN]** — never wipe a real/customer drive to test; covered by the agent unit test `retarget-mismatch-refused`. Only on a genuinely disposable blank device, supervised. **[DESTRUCTIVE — operator confirm]**
|
||||
|
||||
---
|
||||
|
||||
@@ -4,8 +4,13 @@ bin/
|
||||
*.dll
|
||||
*.so
|
||||
*.dylib
|
||||
controller
|
||||
controller.exe
|
||||
# ANCHORED (leading slash) on purpose: a bare `controller` also matches the DIRECTORY
|
||||
# cmd/controller/, so ripgrep silently skipped main.go and new files there needed `git add -f`.
|
||||
# Both directions produce inert-seam mistakes — a grep for a setter finds no caller and reads as
|
||||
# "this is unused", and a genuinely-new file never gets committed. Only the built binary at the
|
||||
# module root should be ignored here.
|
||||
/controller
|
||||
/controller.exe
|
||||
|
||||
# Test artifacts
|
||||
coverage.out
|
||||
|
||||
+813
-52
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,495 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-166 §10 seam discipline — the recovery and the backfill are seams, and a seam that is never
|
||||
// called is the defect class this project has shipped four times: a correct component, green unit
|
||||
// tests that inject it directly, and no production caller.
|
||||
//
|
||||
// These walk main.go's AST. NOT strings.Contains — the sibling bootrecon test records the reason at
|
||||
// first hand: a commented-out call still satisfies a substring match, so the text version passed the
|
||||
// very red-proof it existed to fail. Comments are not code.
|
||||
|
||||
// mainBody returns func main()'s body from main.go, parsed.
|
||||
func mainBody(t *testing.T) *ast.BlockStmt {
|
||||
t.Helper()
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
for _, decl := range f.Decls {
|
||||
if fn, ok := decl.(*ast.FuncDecl); ok && fn.Name.Name == "main" && fn.Body != nil {
|
||||
return fn.Body
|
||||
}
|
||||
}
|
||||
t.Fatal("func main() not found in main.go")
|
||||
return nil
|
||||
}
|
||||
|
||||
// callsInMain returns, in source order, the names of every call in func main() whose function
|
||||
// expression is `x.Sel(...)` or `Sel(...)` — enough to identify the wiring calls by name.
|
||||
func callsInMain(t *testing.T, body *ast.BlockStmt) []string {
|
||||
t.Helper()
|
||||
var names []string
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
switch fun := call.Fun.(type) {
|
||||
case *ast.SelectorExpr:
|
||||
names = append(names, fun.Sel.Name)
|
||||
case *ast.Ident:
|
||||
names = append(names, fun.Name)
|
||||
}
|
||||
return true
|
||||
})
|
||||
return names
|
||||
}
|
||||
|
||||
func indexOfCall(names []string, want string) int {
|
||||
for i, n := range names {
|
||||
if n == want {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// TestMainWiresAppStopRecovery is the Group-I seam test. Comment out the `appStopGuard.Recover()`
|
||||
// line in main.go and this fails, where every behavioural test in internal/backup still passes.
|
||||
func TestMainWiresAppStopRecovery(t *testing.T) {
|
||||
names := callsInMain(t, mainBody(t))
|
||||
|
||||
if indexOfCall(names, "NewAppStopGuard") < 0 {
|
||||
t.Fatal("func main() no longer builds the R-166 app-stop guard — nothing writes or reads the marker")
|
||||
}
|
||||
if indexOfCall(names, "SetStarter") < 0 {
|
||||
t.Fatal("func main() no longer calls SetStarter on the app-stop guard — Recover would find the " +
|
||||
"marker and be unable to start anything, leaving every interrupted app down")
|
||||
}
|
||||
if indexOfCall(names, "Recover") < 0 {
|
||||
t.Fatal("func main() no longer calls Recover() on the app-stop guard — apps left stopped by an " +
|
||||
"interrupted backup stay down forever (the R-166 defect, un-fixed)")
|
||||
}
|
||||
if indexOfCall(names, "SetAppStopGuard") < 0 {
|
||||
t.Fatal("func main() no longer hands the recovered guard to the backup manager — the manager " +
|
||||
"would build a SECOND guard over the same file, i.e. one file with two owners")
|
||||
}
|
||||
if indexOfCall(names, "SetStopGuard") < 0 {
|
||||
t.Fatal("func main() no longer wires the exporter's stop guard — the .fab export path would be " +
|
||||
"the one uncovered stop-and-restart site, which is how a reader concludes the class is handled")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMainWiresDesiredStateBackfill pins the Part-1.5 call.
|
||||
func TestMainWiresDesiredStateBackfill(t *testing.T) {
|
||||
if indexOfCall(callsInMain(t, mainBody(t)), "BackfillDesiredState") < 0 {
|
||||
t.Fatal("func main() no longer calls BackfillDesiredState — every existing app would stay on " +
|
||||
"legacy inference until someone pressed a button on it")
|
||||
}
|
||||
}
|
||||
|
||||
// TestAppStopRecoveryPrecedesTheBootReconciler is §8.4's ORDERING requirement, and it is the reason
|
||||
// the recovery returns its result instead of pushing it through a notifier seam.
|
||||
//
|
||||
// The recovery must COMPLETE — not merely be reached — before `go runBootReconcile(...)` is
|
||||
// launched. If the boot reconciler ran first it would see an app the marker already explains, list
|
||||
// it as an unexplained boot orphan, and one fault would be reported as two.
|
||||
func TestAppStopRecoveryPrecedesTheBootReconciler(t *testing.T) {
|
||||
names := callsInMain(t, mainBody(t))
|
||||
|
||||
recover := indexOfCall(names, "Recover")
|
||||
bootrecon := indexOfCall(names, "runBootReconcile")
|
||||
backfill := indexOfCall(names, "BackfillDesiredState")
|
||||
|
||||
if recover < 0 || bootrecon < 0 || backfill < 0 {
|
||||
t.Fatalf("missing a call: Recover=%d runBootReconcile=%d BackfillDesiredState=%d", recover, bootrecon, backfill)
|
||||
}
|
||||
if recover >= bootrecon {
|
||||
t.Fatal("the app-stop Recover no longer runs BEFORE the boot reconciler is launched — an app " +
|
||||
"the marker explains would also be reported as an unexplained boot orphan (§8.4)")
|
||||
}
|
||||
if backfill >= bootrecon {
|
||||
t.Fatal("the desired-state backfill no longer runs BEFORE the boot reconciler — the reconciler " +
|
||||
"would decide from intent the backfill had not yet written")
|
||||
}
|
||||
if recover >= backfill {
|
||||
t.Fatal("the backfill no longer runs AFTER the app-stop recovery — an app the recovery just " +
|
||||
"restarted would still read as down and be left unrecorded")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMainReportsTheInterruptedOperation pins §2.4: the recovery's outcome reaches the operator.
|
||||
//
|
||||
// The reporting call is deliberately far from the recovery (the notifier does not exist yet at
|
||||
// recovery time), which is exactly the distance across which a wiring gets dropped.
|
||||
func TestMainReportsTheInterruptedOperation(t *testing.T) {
|
||||
body := mainBody(t)
|
||||
names := callsInMain(t, body)
|
||||
|
||||
if indexOfCall(names, "NotifyBackupFailed") < 0 {
|
||||
t.Fatal("func main() no longer reports an interrupted app-data operation to the operator — the " +
|
||||
"controller died mid-backup and nobody is told (§2.4)")
|
||||
}
|
||||
// It must be guarded, not unconditional: a box with nothing to recover must not email an operator
|
||||
// on every single boot.
|
||||
//
|
||||
// R-174 STRENGTHENED THIS. `!= nil` alone is no longer sufficient, because Recover now returns a
|
||||
// non-nil result for a recovery that merely REFUSED starts (an absent data drive) — the drive
|
||||
// gate working as designed. `NotifyBackupFailed` sends `backup_failed`, which is customer-enabled
|
||||
// by default (settings.DefaultEnabledEvents), so a nil-only guard would email the customer
|
||||
// "A biztonsági mentés sikertelen!" about an app nothing is wrong with. The guard must consult
|
||||
// Alarming().
|
||||
guardedByNil, guardedByAlarming := false, false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
ifst, ok := n.(*ast.IfStmt)
|
||||
if !ok || ifst.Cond == nil {
|
||||
return true
|
||||
}
|
||||
carries := false
|
||||
for _, name := range callsInMain(t, ifst.Body) {
|
||||
if name == "NotifyBackupFailed" {
|
||||
carries = true
|
||||
}
|
||||
}
|
||||
if !carries {
|
||||
return true
|
||||
}
|
||||
// Walk the whole condition: it may be `a != nil && a.Alarming()`.
|
||||
ast.Inspect(ifst.Cond, func(c ast.Node) bool {
|
||||
switch e := c.(type) {
|
||||
case *ast.BinaryExpr:
|
||||
if x, ok := e.X.(*ast.Ident); ok && x.Name == "appStopRecovery" && e.Op == token.NEQ {
|
||||
guardedByNil = true
|
||||
}
|
||||
case *ast.CallExpr:
|
||||
if sel, ok := e.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "Alarming" {
|
||||
if x, ok := sel.X.(*ast.Ident); ok && x.Name == "appStopRecovery" {
|
||||
guardedByAlarming = true
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
return true
|
||||
})
|
||||
if !guardedByNil {
|
||||
t.Fatal("the interrupted-operation alert is not guarded by `appStopRecovery != nil` — every " +
|
||||
"healthy boot would page the operator about a backup that was never interrupted")
|
||||
}
|
||||
if !guardedByAlarming {
|
||||
t.Fatal("the interrupted-operation alert is not guarded by appStopRecovery.Alarming() — a " +
|
||||
"recovery that only REFUSED starts (drive absent) would be reported through " +
|
||||
"NotifyBackupFailed, a customer-enabled event type, telling the customer their backup " +
|
||||
"failed when the drive gate was simply doing its job (R-174)")
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-171 seam: the boot drive gate must be WIRED in production -------------------------------
|
||||
|
||||
// TestMainWiresBootDriveGate is the Group-H seam test. An unwired drive gate is not a crash — it is
|
||||
// SILENTLY the pre-v0.190.0 behaviour, which started apps onto absent drives (observed live,
|
||||
// audits/DIAG-bootrecon-drive-absent-2026-08-02.md). Every behavioural test in internal/bootrecon
|
||||
// still passes with the wiring gone, which is exactly the hole this walks the AST to close.
|
||||
//
|
||||
// AST, not strings.Contains: a commented-out call still contains the string — the distinction that
|
||||
// made a previous version of this project's own seam test pass its red-proof (2026-07-21).
|
||||
func TestMainWiresBootDriveGate(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
|
||||
// (a) the settings handle the gate reads is assigned somewhere in main().
|
||||
assigned := false
|
||||
for _, name := range assignedIdentsIn(mainBody(t)) {
|
||||
if name == "bootDriveSettings" {
|
||||
assigned = true
|
||||
}
|
||||
}
|
||||
if !assigned {
|
||||
t.Fatal("func main() no longer assigns bootDriveSettings — the boot drive gate would read a " +
|
||||
"nil settings handle and could not see a disconnected drive")
|
||||
}
|
||||
|
||||
// (b) SetDriveGate is actually called where the reconciler is constructed.
|
||||
called := false
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
if sel, ok := call.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "SetDriveGate" {
|
||||
called = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !called {
|
||||
t.Fatal("main.go no longer calls SetDriveGate on the boot reconciler — the sweep would start " +
|
||||
"apps whose data drive is absent (R-171, a regression observed live on 2026-08-02)")
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-174 seam: the app-stop guard's starter must be GATED in production -----------------------
|
||||
|
||||
// TestMainWiresGatedAppStopStarter pins Part 0's production wiring. `SetStarter(stackMgr)` — the raw
|
||||
// manager, which is what shipped in v0.189.0 — compiles, passes every behavioural test in
|
||||
// internal/backup (they inject their own gating starter), and silently starts apps onto absent
|
||||
// drives at boot. The ONLY thing that distinguishes the fixed wiring from the broken one is the
|
||||
// argument at the call site, so that is what this reads.
|
||||
//
|
||||
// AST, not strings.Contains: a commented-out call still contains the string.
|
||||
func TestMainWiresGatedAppStopStarter(t *testing.T) {
|
||||
body := mainBody(t)
|
||||
|
||||
var arg ast.Expr
|
||||
found := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || sel.Sel.Name != "SetStarter" || len(call.Args) != 1 {
|
||||
return true
|
||||
}
|
||||
// Only the app-stop guard's SetStarter, not some other type's.
|
||||
if x, ok := sel.X.(*ast.Ident); !ok || x.Name != "appStopGuard" {
|
||||
return true
|
||||
}
|
||||
arg, found = call.Args[0], true
|
||||
return false
|
||||
})
|
||||
if !found {
|
||||
t.Fatal("func main() no longer calls appStopGuard.SetStarter — Recover would find the marker " +
|
||||
"and be unable to start anything")
|
||||
}
|
||||
|
||||
// The argument must be a gatedAppStopStarter composite literal. A bare identifier (`stackMgr`)
|
||||
// is precisely the v0.189.0 defect.
|
||||
lit, ok := arg.(*ast.CompositeLit)
|
||||
if !ok {
|
||||
t.Fatalf("appStopGuard.SetStarter is wired with %T, not a gatedAppStopStarter literal — an "+
|
||||
"un-gated starter restarts apps onto MISSING drives at boot (R-174, the R-171 defect one "+
|
||||
"path over)", arg)
|
||||
}
|
||||
id, ok := lit.Type.(*ast.Ident)
|
||||
if !ok || id.Name != "gatedAppStopStarter" {
|
||||
t.Fatalf("appStopGuard.SetStarter is wired with a %v literal, want gatedAppStopStarter", lit.Type)
|
||||
}
|
||||
|
||||
// And that gate must be a driveStartGate — the SAME predicate the boot sweep uses, so the two
|
||||
// cannot disagree about whether an app's drive is available.
|
||||
gated := false
|
||||
for _, el := range lit.Elts {
|
||||
kv, ok := el.(*ast.KeyValueExpr)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
k, ok := kv.Key.(*ast.Ident)
|
||||
if !ok || k.Name != "gate" {
|
||||
continue
|
||||
}
|
||||
if gl, ok := kv.Value.(*ast.CompositeLit); ok {
|
||||
if gid, ok := gl.Type.(*ast.Ident); ok && gid.Name == "driveStartGate" {
|
||||
gated = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !gated {
|
||||
t.Fatal("the app-stop starter's gate is not a driveStartGate — the crash recovery and the " +
|
||||
"boot sweep would answer \"may this app start?\" from two different implementations, " +
|
||||
"which is the drift the extraction exists to prevent")
|
||||
}
|
||||
}
|
||||
|
||||
// TestBootDriveGateAndAppStopShareTheDrivePredicate pins the OTHER half of the same claim: the boot
|
||||
// sweep must keep delegating to driveStartGate rather than growing its own copy of the drive checks.
|
||||
//
|
||||
// This is the "a comment asserting an invariant needs a test pinning it" rule. The claim — that the
|
||||
// two gates cannot disagree — is true only while both call the same code.
|
||||
func TestBootDriveGateAndAppStopShareTheDrivePredicate(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
|
||||
var mayStart *ast.FuncDecl
|
||||
for _, decl := range f.Decls {
|
||||
fn, ok := decl.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "MayStart" || fn.Recv == nil || len(fn.Recv.List) != 1 {
|
||||
continue
|
||||
}
|
||||
if id, ok := fn.Recv.List[0].Type.(*ast.Ident); ok && id.Name == "bootDriveGate" {
|
||||
mayStart = fn
|
||||
}
|
||||
}
|
||||
if mayStart == nil {
|
||||
t.Fatal("bootDriveGate.MayStart not found in main.go")
|
||||
}
|
||||
|
||||
// It must call through to the shared predicate.
|
||||
delegates := false
|
||||
ast.Inspect(mayStart.Body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || sel.Sel.Name != "MayStart" {
|
||||
return true
|
||||
}
|
||||
if x, ok := sel.X.(*ast.SelectorExpr); ok && x.Sel.Name == "drive" {
|
||||
delegates = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !delegates {
|
||||
t.Fatal("bootDriveGate.MayStart no longer delegates to the shared driveStartGate — the boot " +
|
||||
"sweep and the app-stop crash recovery would each carry their own drive logic, and the " +
|
||||
"two can then disagree about whether an app may start (R-174)")
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-158 / R-167 seams: both new alerts must be WIRED in production ---------------------------
|
||||
|
||||
// TestMainWiresTheUnitCaptureAlert pins Part 1's seam. `SetUnitNotify` is nil-safe by design, so an
|
||||
// unwired seam is not a crash — it is SILENTLY the pre-v0.191.0 behaviour, in which a per-app Tier-1
|
||||
// capture failure is a `[WARN]` line and reaches no hub channel at all. Every behavioural test in
|
||||
// internal/backup injects its own callback and passes with the production wiring gone, which is
|
||||
// exactly the hole this closes. THIS PROJECT'S COUNT OF "BUILT BUT NEVER WIRED" REACHES FIVE WITH
|
||||
// R-158 — the defect being fixed here IS an instance of it.
|
||||
func TestMainWiresTheUnitCaptureAlert(t *testing.T) {
|
||||
names := callsInMain(t, mainBody(t))
|
||||
|
||||
if indexOfCall(names, "SetUnitNotify") < 0 {
|
||||
t.Fatal("func main() no longer calls backupMgr.SetUnitNotify — a per-app recovery-unit " +
|
||||
"capture failure would reach no hub channel, which is R-158 un-fixed (the seam built " +
|
||||
"and left disconnected, for the fifth time in this project)")
|
||||
}
|
||||
if indexOfCall(names, "NotifyRecoveryUnitCaptureFailed") < 0 {
|
||||
t.Fatal("main.go no longer calls NotifyRecoveryUnitCaptureFailed — the seam is wired to " +
|
||||
"something that pushes no event, which looks identical to a working alert from inside " +
|
||||
"internal/backup")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMainWiresTheFillWatcher pins Part 2's seam. Three separate things can be dropped and each one
|
||||
// silently reverts the customer to "nothing warns before a disk fills": the watcher can go
|
||||
// unconstructed, its notify can go unwired (the Watcher is nil-safe), or it can never be scheduled.
|
||||
func TestMainWiresTheFillWatcher(t *testing.T) {
|
||||
body := mainBody(t)
|
||||
names := callsInMain(t, body)
|
||||
|
||||
if indexOfCall(names, "New") < 0 || !assignsIdent(body, "fillWatcher") {
|
||||
t.Fatal("func main() no longer constructs the fill watcher — nothing warns the customer " +
|
||||
"before a filesystem fills (R-167, decision D-c's customer half)")
|
||||
}
|
||||
if indexOfCall(names, "SetNotify") < 0 {
|
||||
t.Fatal("func main() no longer calls SetNotify on the fill watcher — the Watcher is nil-safe, " +
|
||||
"so it would run the checks, update its state, log, and tell the CUSTOMER nothing")
|
||||
}
|
||||
|
||||
// It must actually be scheduled: a watcher nobody calls is a watcher that never fires.
|
||||
scheduled := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok || len(call.Args) == 0 {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || (sel.Sel.Name != "Daily" && sel.Sel.Name != "Every") {
|
||||
return true
|
||||
}
|
||||
lit, ok := call.Args[0].(*ast.BasicLit)
|
||||
if ok && strings.Contains(lit.Value, "fill-watch") {
|
||||
scheduled = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !scheduled {
|
||||
t.Fatal("the fill watcher is never registered on the scheduler — it would be constructed, " +
|
||||
"wired, and never run, which is indistinguishable from a filesystem that never fills")
|
||||
}
|
||||
|
||||
// It must ALSO run once at startup. Neither `Every` nor `Daily` fires on registration (both wait
|
||||
// for their first tick), so a schedule-only wiring means a box that BOOTS with a filesystem
|
||||
// already over the line stays silent for up to 24 hours — a real fault visible only after a
|
||||
// deadline elapses, which is the R-100 shape. The hub's own checkers leave already-breached keys
|
||||
// unseeded at init for exactly this reason.
|
||||
if indexOfCall(names, "After") < 0 {
|
||||
t.Fatal("nothing delays a startup fill check — see fillWatchStartupDelay")
|
||||
}
|
||||
startupRun := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
g, ok := n.(*ast.GoStmt)
|
||||
if !ok || g.Call == nil {
|
||||
return true
|
||||
}
|
||||
lit, ok := g.Call.Fun.(*ast.FuncLit)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
var sawDelay, sawCheck bool
|
||||
ast.Inspect(lit.Body, func(m ast.Node) bool {
|
||||
if id, ok := m.(*ast.Ident); ok && id.Name == "fillWatchStartupDelay" {
|
||||
sawDelay = true
|
||||
}
|
||||
if call, ok := m.(*ast.CallExpr); ok {
|
||||
if sel, ok := call.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "Check" {
|
||||
if x, ok := sel.X.(*ast.Ident); ok && x.Name == "fillWatcher" {
|
||||
sawCheck = true
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
if sawDelay && sawCheck {
|
||||
startupRun = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !startupRun {
|
||||
t.Fatal("the fill watcher never runs at STARTUP — Daily/Every both wait for their first " +
|
||||
"tick, so a box that boots with a full disk would not warn for up to 24 hours (the " +
|
||||
"R-100 shape: a real fault visible only after a deadline elapses)")
|
||||
}
|
||||
}
|
||||
|
||||
// assignsIdent reports whether a block assigns to the named identifier.
|
||||
func assignsIdent(body *ast.BlockStmt, want string) bool {
|
||||
for _, n := range assignedIdentsIn(body) {
|
||||
if n == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// assignedIdentsIn returns the names assigned to in a block (plain `=` and `:=`).
|
||||
func assignedIdentsIn(body *ast.BlockStmt) []string {
|
||||
var names []string
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
as, ok := n.(*ast.AssignStmt)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
for _, lhs := range as.Lhs {
|
||||
if id, ok := lhs.(*ast.Ident); ok {
|
||||
names = append(names, id.Name)
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
return names
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"io"
|
||||
"log"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/bootrecon"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// §9 rule 6 — the seam-discipline test. Two inert-seam defects shipped in the two days before this
|
||||
// task (controller v0.154.0 and agent v0.91.0), both the same shape: the component was correct, its
|
||||
// unit tests injected the seam directly, and the PRODUCTION CALLER was never made. Everything was
|
||||
// green and the feature did nothing. So R-52 gets its wiring asserted from package main, not only
|
||||
// from internal/bootrecon.
|
||||
|
||||
// TestRunBootReconcile_InvokesTheSweep pins the function main() actually calls: after the settle
|
||||
// window it runs the sweep exactly once, with the manager it was handed.
|
||||
func TestRunBootReconcile_InvokesTheSweep(t *testing.T) {
|
||||
orig := bootReconcileFn
|
||||
t.Cleanup(func() { bootReconcileFn = orig })
|
||||
origSettle := bootReconcileSettle
|
||||
t.Cleanup(func() { bootReconcileSettle = origSettle })
|
||||
bootReconcileSettle = time.Millisecond
|
||||
|
||||
calls := 0
|
||||
var gotMgr bootrecon.StackProvider
|
||||
bootReconcileFn = func(_ context.Context, mgr bootrecon.StackProvider, _ *log.Logger) bootrecon.Result {
|
||||
calls++
|
||||
gotMgr = mgr
|
||||
return bootrecon.Result{}
|
||||
}
|
||||
|
||||
fake := &wiringStacks{}
|
||||
runBootReconcile(context.Background(), fake, log.New(io.Discard, "", 0))
|
||||
|
||||
if calls != 1 {
|
||||
t.Fatalf("the boot sweep ran %d times, want exactly 1 (start-once, never a loop)", calls)
|
||||
}
|
||||
if gotMgr != bootrecon.StackProvider(fake) {
|
||||
t.Fatalf("the sweep was handed %v, want the stack manager main() owns", gotMgr)
|
||||
}
|
||||
}
|
||||
|
||||
// A controller shutting down during its own settle window must not start anything.
|
||||
func TestRunBootReconcile_CancelledDuringSettleDoesNothing(t *testing.T) {
|
||||
orig := bootReconcileFn
|
||||
t.Cleanup(func() { bootReconcileFn = orig })
|
||||
|
||||
calls := 0
|
||||
bootReconcileFn = func(context.Context, bootrecon.StackProvider, *log.Logger) bootrecon.Result {
|
||||
calls++
|
||||
return bootrecon.Result{}
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
runBootReconcile(ctx, &wiringStacks{}, log.New(io.Discard, "", 0))
|
||||
|
||||
if calls != 0 {
|
||||
t.Fatalf("the sweep ran %d times on a cancelled context, want 0", calls)
|
||||
}
|
||||
}
|
||||
|
||||
// The call site itself. A function-variable test can only prove the function is correct — it cannot
|
||||
// prove main() calls it, which is exactly the hole both inert-seam defects fell through. This walks
|
||||
// main.go's AST for a `go runBootReconcile(...)` inside func main(); delete or comment out that line
|
||||
// and this fails, where every behavioural test above would still pass.
|
||||
//
|
||||
// It is an AST walk and not a strings.Contains for a reason found while red-proofing it: a
|
||||
// commented-out call still satisfies a substring match, so the text version passed the very
|
||||
// red-proof it existed to fail. Comments are not code.
|
||||
func TestMainWiresBootReconcile(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse main.go: %v", err)
|
||||
}
|
||||
|
||||
found := false
|
||||
for _, decl := range f.Decls {
|
||||
fn, ok := decl.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "main" || fn.Body == nil {
|
||||
continue
|
||||
}
|
||||
ast.Inspect(fn.Body, func(n ast.Node) bool {
|
||||
gostmt, ok := n.(*ast.GoStmt)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
if ident, ok := gostmt.Call.Fun.(*ast.Ident); ok && ident.Name == "runBootReconcile" {
|
||||
found = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("func main() no longer starts the R-52 boot reconciliation with `go runBootReconcile(...)` " +
|
||||
"— the sweep is inert (the v0.154.0 / v0.91.0 defect class: a correct component nobody calls)")
|
||||
}
|
||||
}
|
||||
|
||||
// The settle window must stay inside the dead-app boot grace, or a successful recovery would alert.
|
||||
func TestBootReconcileFitsInsideTheBootGrace(t *testing.T) {
|
||||
worst := bootReconcileSettle + time.Duration(bootrecon.DefaultAttempts-1)*bootrecon.DefaultRetryDelay
|
||||
if worst >= deadAppBootGrace {
|
||||
t.Fatalf("worst-case sweep %s does not fit inside the %s boot grace — a successful "+
|
||||
"recovery would fire app_start_failed", worst, deadAppBootGrace)
|
||||
}
|
||||
}
|
||||
|
||||
type wiringStacks struct{}
|
||||
|
||||
func (w *wiringStacks) GetStacks() []stacks.Stack { return nil }
|
||||
func (w *wiringStacks) StartStack(string) error { return nil }
|
||||
func (w *wiringStacks) RefreshStatus() error { return nil }
|
||||
@@ -0,0 +1,366 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"log"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/bootrecon"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// R-157 mechanism A — the sweep that looked once.
|
||||
//
|
||||
// TIMING IS NOT TESTED BY SLEEPING (§10). The window's constants are package vars, so each test
|
||||
// shrinks them to sub-millisecond values: the CONTRACT under test is "how many samples, and what
|
||||
// ends the window", not "how long a second is". A test that waited real seconds would be slow,
|
||||
// flaky, and would still not prove the contract.
|
||||
|
||||
// windowStacks is a StackProvider whose fleet CHANGES over successive GetStacks() calls — which is
|
||||
// the whole point: the pre-v0.190.0 sweep sampled once and could not see a late settler.
|
||||
type windowStacks struct {
|
||||
// frames is the fleet as seen on each successive GetStacks() call; the last frame repeats.
|
||||
frames [][]stacks.Stack
|
||||
calls int
|
||||
starts map[string]int
|
||||
onStart func(*windowStacks, string)
|
||||
refreshes int
|
||||
refreshErr error
|
||||
// cycle makes the fleet NEVER settle: frames repeat forever instead of the last one sticking.
|
||||
// Required by the budget test — with frames that eventually stop changing, the window terminates
|
||||
// by SETTLING even with the budget removed, so the red-proof would not reach the hang it exists
|
||||
// to demonstrate.
|
||||
cycle bool
|
||||
}
|
||||
|
||||
func (w *windowStacks) GetStacks() []stacks.Stack {
|
||||
i := w.calls
|
||||
w.calls++
|
||||
if i >= len(w.frames) {
|
||||
if w.cycle {
|
||||
i = i % len(w.frames)
|
||||
} else {
|
||||
i = len(w.frames) - 1
|
||||
}
|
||||
}
|
||||
return w.frames[i]
|
||||
}
|
||||
|
||||
func (w *windowStacks) RefreshStatus() error {
|
||||
w.refreshes++
|
||||
if w.refreshErr != nil {
|
||||
return w.refreshErr
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (w *windowStacks) StartStack(name string) error {
|
||||
if w.starts == nil {
|
||||
w.starts = map[string]int{}
|
||||
}
|
||||
w.starts[name]++
|
||||
if w.onStart != nil {
|
||||
w.onStart(w, name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// shrinkWindow makes the window fast and deterministic, and restores the shipped values after.
|
||||
func shrinkWindow(t *testing.T, sample time.Duration, stableFor int, budget time.Duration) {
|
||||
t.Helper()
|
||||
os, ost, ob, osettle := bootReconcileSample, bootReconcileStableFor, bootReconcileBudget, bootReconcileSettle
|
||||
t.Cleanup(func() {
|
||||
bootReconcileSample, bootReconcileStableFor, bootReconcileBudget, bootReconcileSettle = os, ost, ob, osettle
|
||||
})
|
||||
bootReconcileSample, bootReconcileStableFor, bootReconcileBudget = sample, stableFor, budget
|
||||
bootReconcileSettle = time.Millisecond
|
||||
}
|
||||
|
||||
// captureSweep replaces the sweep with a recorder and returns the fleet it was handed.
|
||||
func captureSweep(t *testing.T) *[][]stacks.Stack {
|
||||
t.Helper()
|
||||
orig := bootReconcileFn
|
||||
t.Cleanup(func() { bootReconcileFn = orig })
|
||||
var seen [][]stacks.Stack
|
||||
bootReconcileFn = func(_ context.Context, mgr bootrecon.StackProvider, _ *log.Logger) bootrecon.Result {
|
||||
seen = append(seen, mgr.GetStacks())
|
||||
return bootrecon.Result{}
|
||||
}
|
||||
return &seen
|
||||
}
|
||||
|
||||
func upStack(name string) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name, Deployed: true, State: stacks.StateRunning,
|
||||
Containers: []stacks.ContainerInfo{{Name: name, State: stacks.StateRunning}},
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
// settlingLate is the R-157-A shape: at T+5s the app is still `starting` with its containers coming
|
||||
// up, and it only comes to rest in a DOWN state later.
|
||||
func settlingLate(name string) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name, Deployed: true, State: stacks.StateStarting,
|
||||
Containers: []stacks.ContainerInfo{{Name: name, State: stacks.StateStarting}},
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
func settledDown(name string) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name, Deployed: true, State: stacks.StateExited,
|
||||
Containers: []stacks.ContainerInfo{{Name: name, State: stacks.StateExited}},
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group A / Scenario B — a late settler IS swept -----------------------------------------------
|
||||
|
||||
func TestBootWindow_LateSettlerIsSweptOnASettledFleet(t *testing.T) {
|
||||
// The fleet is still moving for the first frames and settles only later. The sweep must run
|
||||
// AFTER it settles and must be handed the SETTLED fleet — because the pre-v0.190.0 defect was a
|
||||
// candidate set derived from a fleet that had not finished moving.
|
||||
//
|
||||
// RED-PROOF: restore the single-sweep shape (delete the sampling loop so runBootReconcile calls
|
||||
// bootReconcileFn straight after the settle delay) and this test fails — the sweep is handed the
|
||||
// `starting` frame, in which the app is not a down-state candidate at all.
|
||||
// Demonstrated in REPORT.md §4.
|
||||
shrinkWindow(t, time.Millisecond, 2, 500*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
|
||||
w := &windowStacks{frames: [][]stacks.Stack{
|
||||
{settlingLate("immich")}, // T+5s: still coming up
|
||||
{settlingLate("immich")},
|
||||
{settledDown("immich")}, // settles into a down state only now
|
||||
{settledDown("immich")},
|
||||
{settledDown("immich")},
|
||||
}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("the sweep ran %d times, want exactly 1 — the window samples, it does not sweep per sample", len(*seen))
|
||||
}
|
||||
got := (*seen)[0]
|
||||
if len(got) != 1 || got[0].State != stacks.StateExited {
|
||||
t.Fatalf("the sweep was handed state=%v, want the SETTLED (exited) fleet — a candidate set "+
|
||||
"derived from a still-moving fleet is exactly the R-157 mechanism-A defect", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_SweepRunsExactlyOnceEvenOnAQuietBoot(t *testing.T) {
|
||||
shrinkWindow(t, time.Millisecond, 2, 500*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("bookstack")}}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("sweeps=%d, want exactly 1 on a quiet boot", len(*seen))
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group B / Scenario C — the window TERMINATES -------------------------------------------------
|
||||
|
||||
func TestBootWindow_BudgetEndsAForeverChangingFleet(t *testing.T) {
|
||||
// A fleet that never stops changing must not sample forever. The budget ends it, the sweep runs
|
||||
// once anyway (a churning box is exactly the box that needs it), and the log SAYS the budget
|
||||
// ended it — "settled and found nothing" and "ran out of time" are different facts.
|
||||
//
|
||||
// RED-PROOF: remove the `time.Since(started) < bootReconcileBudget` loop condition and this test
|
||||
// hangs — the unbounded-loop shape §5 bans. Demonstrated in REPORT.md §4 (observed as a timeout).
|
||||
shrinkWindow(t, time.Millisecond, 3, 30*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
var buf strings.Builder
|
||||
|
||||
// Every frame differs, so `stable` can never reach stableFor.
|
||||
frames := make([][]stacks.Stack, 0, 200)
|
||||
for i := 0; i < 200; i++ {
|
||||
s := upStack("immich")
|
||||
s.Containers = make([]stacks.ContainerInfo, i%7) // container count changes every sample
|
||||
frames = append(frames, []stacks.Stack{s})
|
||||
}
|
||||
w := &windowStacks{frames: frames, cycle: true}
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
runBootReconcile(context.Background(), w, log.New(&buf, "", 0))
|
||||
close(done)
|
||||
}()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("runBootReconcile did not terminate on a forever-changing fleet — this is the " +
|
||||
"unbounded restart-loop shape the package's own boundary forbids")
|
||||
}
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("sweeps=%d, want exactly 1 after the budget expired", len(*seen))
|
||||
}
|
||||
if out := buf.String(); !strings.Contains(out, "budget") {
|
||||
t.Fatalf("the log does not say the BUDGET ended the window, so a churning boot reads like a "+
|
||||
"quiet one:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_SettledPathSaysSettled(t *testing.T) {
|
||||
shrinkWindow(t, time.Millisecond, 2, 500*time.Millisecond)
|
||||
captureSweep(t)
|
||||
var buf strings.Builder
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("docmost")}}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(&buf, "", 0))
|
||||
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "settled") {
|
||||
t.Fatalf("a settled window must say so — otherwise it is indistinguishable from a budget "+
|
||||
"expiry:\n%s", out)
|
||||
}
|
||||
if strings.Contains(out, "budget") {
|
||||
t.Fatalf("a settled window must NOT claim the budget ended it:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_CancelledContextStopsImmediately(t *testing.T) {
|
||||
shrinkWindow(t, time.Millisecond, 3, time.Second)
|
||||
seen := captureSweep(t)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
runBootReconcile(ctx, &windowStacks{frames: [][]stacks.Stack{{upStack("x")}}}, log.New(io.Discard, "", 0))
|
||||
if len(*seen) != 0 {
|
||||
t.Fatalf("the sweep ran %d times on a cancelled context, want 0", len(*seen))
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group C / Scenario D — a customer's Stop survives the WIDENED window -------------------------
|
||||
|
||||
func TestBootWindow_CustomerStoppedAppSurvivesEveryPass(t *testing.T) {
|
||||
// THE REGRESSION THIS TASK COULD INTRODUCE. A longer window means more chances to resurrect an
|
||||
// app the customer deliberately stopped. It must survive the whole window — this drives the REAL
|
||||
// bootrecon sweep (not the captured stub), so the desired-state check is genuinely exercised.
|
||||
//
|
||||
// RED-PROOF: drop the DesiredStateStopped branch from isBootOrphan (make it fall through to the
|
||||
// running case) and this test fails with a start count of 1. Demonstrated in REPORT.md §4.
|
||||
shrinkWindow(t, time.Millisecond, 2, 200*time.Millisecond)
|
||||
|
||||
stopped := stacks.Stack{
|
||||
Name: "nextcloud", Deployed: true, State: stacks.StateStopped, Containers: nil,
|
||||
AppConfig: &stacks.AppConfig{Deployed: true, DesiredState: stacks.DesiredStateStopped},
|
||||
}
|
||||
// The fleet churns around it, so the window runs many passes before settling.
|
||||
frames := [][]stacks.Stack{
|
||||
{stopped, settlingLate("immich")},
|
||||
{stopped, settlingLate("immich")},
|
||||
{stopped, settledDown("immich")},
|
||||
{stopped, upStack("immich")},
|
||||
{stopped, upStack("immich")},
|
||||
{stopped, upStack("immich")},
|
||||
}
|
||||
w := &windowStacks{frames: frames, onStart: func(w *windowStacks, _ string) {}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if n := w.starts["nextcloud"]; n != 0 {
|
||||
t.Fatalf("the customer-stopped app was started %d time(s) by the widened window — this is the "+
|
||||
"regression a longer window makes possible and it is the worst outcome available here", n)
|
||||
}
|
||||
}
|
||||
|
||||
// --- §8.3 — a late recovery is REPORTED, never hidden ---------------------------------------------
|
||||
|
||||
func TestRecordLateRecovery_WarnsWhenTheGraceHasAlreadyExpired(t *testing.T) {
|
||||
var buf strings.Builder
|
||||
lg := log.New(&buf, "", 0)
|
||||
// started far enough back that settle + elapsed exceeds the 90 s grace
|
||||
recordLateRecovery(lg, time.Now().Add(-(deadAppBootGrace + 10*time.Second)), bootrecon.Result{Recovered: []string{"immich"}})
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "LATE RECOVERY") || !strings.Contains(out, "immich") {
|
||||
t.Fatalf("a recovery past the dead-app grace must be reported by name — otherwise a stale "+
|
||||
"alarm stands with no counter-evidence (§8.3):\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecordLateRecovery_SilentInsideTheGrace(t *testing.T) {
|
||||
var buf strings.Builder
|
||||
recordLateRecovery(log.New(&buf, "", 0), time.Now(), bootrecon.Result{Recovered: []string{"immich"}})
|
||||
if buf.Len() != 0 {
|
||||
t.Fatalf("a recovery INSIDE the grace must stay silent — that is what makes a successful "+
|
||||
"recovery invisible to the customer:\n%s", buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecordLateRecovery_SilentWhenNothingRecovered(t *testing.T) {
|
||||
var buf strings.Builder
|
||||
recordLateRecovery(log.New(&buf, "", 0), time.Now().Add(-time.Hour), bootrecon.Result{})
|
||||
if buf.Len() != 0 {
|
||||
t.Fatalf("nothing was recovered, so there is nothing late to report:\n%s", buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
// --- The window's constants must fit the grace they are justified against -------------------------
|
||||
|
||||
func TestBootWindow_CommonCaseFitsInsideTheDeadAppGrace(t *testing.T) {
|
||||
// The comment on the window constants justifies them against deadAppBootGrace. A comment
|
||||
// asserting an invariant needs a test pinning it, or it is a wish.
|
||||
common := bootReconcileSettle + bootReconcileBudget + bootrecon.DefaultRetryDelay
|
||||
if common > deadAppBootGrace {
|
||||
t.Fatalf("settle(%s) + budget(%s) + one retry(%s) = %s exceeds the %s dead-app grace — the "+
|
||||
"COMMON case must stay silent, or every slow boot alerts",
|
||||
bootReconcileSettle, bootReconcileBudget, bootrecon.DefaultRetryDelay, common, deadAppBootGrace)
|
||||
}
|
||||
if bootReconcileSample <= 0 || bootReconcileStableFor < 2 {
|
||||
t.Fatalf("sample=%s stableFor=%d — one sample cannot distinguish 'settled' from 'sampled "+
|
||||
"between two docker events'", bootReconcileSample, bootReconcileStableFor)
|
||||
}
|
||||
}
|
||||
|
||||
// --- the sample must observe REALITY, not the Manager's cache ------------------------------------
|
||||
|
||||
func TestBootWindow_EverySampleRefreshesTheStatus(t *testing.T) {
|
||||
// FOUND BY LIVE VALIDATION, not review. GetStacks() returns the Manager's in-memory map, which
|
||||
// the scheduler refreshes on its own 10 s cadence. Sampling every 5 s WITHOUT refreshing means two
|
||||
// consecutive samples can be identical because the cache did not update — so the window declares
|
||||
// "settled" on stale data and sweeps on a picture of the box from up to 10 s ago. On 9201 a
|
||||
// container removed ~5 s before the window closed was still in the sampled fleet, and the sweep
|
||||
// logged "no boot-orphaned apps" for an app that had none.
|
||||
//
|
||||
// RED-PROOF: delete the `_ = mgr.RefreshStatus()` line from sampleBootFleet and this test fails
|
||||
// with refreshes=0. Demonstrated in REPORT.md §4.
|
||||
shrinkWindow(t, time.Millisecond, 3, 500*time.Millisecond)
|
||||
captureSweep(t)
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("immich")}}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if w.refreshes < 3 {
|
||||
t.Fatalf("the window refreshed %d time(s) for %d samples — every sample must observe reality, "+
|
||||
"or 'settled' can mean 'the cache did not update'", w.refreshes, w.calls)
|
||||
}
|
||||
// calls includes ONE extra GetStacks from the captured sweep itself, which does not sample.
|
||||
if w.refreshes != w.calls-1 {
|
||||
t.Fatalf("refreshes=%d but samples=%d — each sample must refresh exactly once before reading",
|
||||
w.refreshes, w.calls-1)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBootWindow_RefreshErrorDoesNotStopTheWindow(t *testing.T) {
|
||||
// A boot window that cannot reach docker is exactly when a stale verdict is most dangerous, but
|
||||
// giving up entirely would leave the sweep un-run. Degrade, do not abort.
|
||||
shrinkWindow(t, time.Millisecond, 2, 200*time.Millisecond)
|
||||
seen := captureSweep(t)
|
||||
w := &windowStacks{frames: [][]stacks.Stack{{upStack("immich")}}, refreshErr: errRefresh{}}
|
||||
|
||||
runBootReconcile(context.Background(), w, log.New(io.Discard, "", 0))
|
||||
|
||||
if len(*seen) != 1 {
|
||||
t.Fatalf("sweeps=%d, want 1 — a refresh error must not abort the window", len(*seen))
|
||||
}
|
||||
}
|
||||
|
||||
type errRefresh struct{}
|
||||
|
||||
func (errRefresh) Error() string { return "docker unreachable" }
|
||||
@@ -0,0 +1,125 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/web"
|
||||
)
|
||||
|
||||
// v0.164.0: classifyRunStates is the single fix-3 derivation point. A deliberate user stop
|
||||
// (StateStopped) must NOT alarm — it is excluded from both the banner dead-list and the notifier
|
||||
// Down-set — while every genuine fault (StateExited / StateDegraded) keeps alerting byte-identically.
|
||||
// Invariants behind the suppression are documented at classifyRunStates (I1: compose down ⇒ zero
|
||||
// containers ⇒ StateStopped; I2: P2 census — all catalog services unless-stopped ⇒ faults never rest
|
||||
// at stopped).
|
||||
|
||||
func stack(name string, st stacks.ContainerState, deployed, deploying bool) stacks.Stack {
|
||||
return stacks.Stack{
|
||||
Name: name,
|
||||
Meta: stacks.Metadata{DisplayName: name},
|
||||
State: st,
|
||||
Deployed: deployed,
|
||||
Deploying: deploying,
|
||||
}
|
||||
}
|
||||
|
||||
func downByName(states []notify.AppRunState) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
for _, s := range states {
|
||||
m[s.Name] = s.Down
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func deadNames(dead []web.DeadApp) map[string]bool {
|
||||
m := map[string]bool{}
|
||||
for _, d := range dead {
|
||||
m[d.Name] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Group A (Scenario A) — suppression. Over a [running, stopped, exited, degraded] fixture, the dead
|
||||
// list is EXACTLY {exited, degraded} and the Down flags are {false, false, true, true}: the stopped
|
||||
// app is silent, the two faults still alarm.
|
||||
//
|
||||
// COMPANION red-proof: revert the filter to bare `stacks.IsDownState(st.State)` (drop the
|
||||
// `&& st.State != stacks.StateStopped` guard) → stopped reports Down=true and enters the dead list →
|
||||
// both the dead-set and the Down-flag assertions below fail. (Verified by hand-editing the seam.)
|
||||
func TestClassifyRunStates_StoppedIsSuppressed(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("radarr", stacks.StateRunning, true, false),
|
||||
stack("cwa", stacks.StateStopped, true, false),
|
||||
stack("immich", stacks.StateExited, true, false),
|
||||
stack("nextcloud", stacks.StateDegraded, true, false),
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
|
||||
gotDead := deadNames(dead)
|
||||
if len(gotDead) != 2 || !gotDead["immich"] || !gotDead["nextcloud"] {
|
||||
t.Fatalf("dead list must be exactly {immich(exited), nextcloud(degraded)}, got %+v", dead)
|
||||
}
|
||||
if gotDead["cwa"] {
|
||||
t.Errorf("a deliberately stopped app must NOT be in the dead list (no banner)")
|
||||
}
|
||||
if gotDead["radarr"] {
|
||||
t.Errorf("a running app must never be in the dead list")
|
||||
}
|
||||
|
||||
down := downByName(states)
|
||||
want := map[string]bool{"radarr": false, "cwa": false, "immich": true, "nextcloud": true}
|
||||
if len(down) != len(want) {
|
||||
t.Fatalf("every deployed app must have a run state, got %+v", down)
|
||||
}
|
||||
for name, w := range want {
|
||||
if down[name] != w {
|
||||
t.Errorf("Down[%s] = %v, want %v (stopped ⇒ false ⇒ no app_start_failed event)", name, down[name], w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Group B (Scenario B) — fault parity. With only exited + degraded present, BOTH surface in the dead
|
||||
// list AND both report Down=true — byte-identical to v0.163.1 for every non-stopped down state. The
|
||||
// suppression touches stopped and nothing else.
|
||||
func TestClassifyRunStates_FaultParity(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("immich", stacks.StateExited, true, false),
|
||||
stack("nextcloud", stacks.StateDegraded, true, false),
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
|
||||
gotDead := deadNames(dead)
|
||||
if len(gotDead) != 2 || !gotDead["immich"] || !gotDead["nextcloud"] {
|
||||
t.Fatalf("both faults must appear in the dead list, got %+v", dead)
|
||||
}
|
||||
down := downByName(states)
|
||||
if !down["immich"] || !down["nextcloud"] {
|
||||
t.Fatalf("both faults must report Down=true, got %+v", down)
|
||||
}
|
||||
// State strings must ride through to the banner unchanged (banner shows "(exited)"/"(degraded)").
|
||||
byName := map[string]string{}
|
||||
for _, d := range dead {
|
||||
byName[d.Name] = d.State
|
||||
}
|
||||
if byName["immich"] != string(stacks.StateExited) || byName["nextcloud"] != string(stacks.StateDegraded) {
|
||||
t.Errorf("dead-app State must carry the raw aggregate state, got %+v", byName)
|
||||
}
|
||||
}
|
||||
|
||||
// Deploying and undeployed stacks are skipped entirely (unchanged fix-3 behavior).
|
||||
func TestClassifyRunStates_SkipsDeployingAndUndeployed(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("mid", stacks.StateDeploying, true, true), // mid-deploy → skipped
|
||||
stack("gone", stacks.StateExited, false, false), // not deployed → skipped
|
||||
}
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
if len(dead) != 0 || len(states) != 0 {
|
||||
t.Fatalf("deploying and undeployed stacks must be skipped, got dead=%+v states=%+v", dead, states)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// C9-F2 — a SUSTAINED `restarting` is a crash loop and must alarm; a BRIEF one must not.
|
||||
//
|
||||
// The defect: `IsDownState` excludes `restarting` as "self-recovering", but for the catalog's
|
||||
// standard `restart: unless-stopped` Docker retries forever, so a crash loop sat in `restarting`
|
||||
// indefinitely and was counted as working. Campaign 9 watched docmost loop for nine minutes
|
||||
// (restartcount 18) while the F-OBS heartbeat printed "4 deployed app(s) evaluated, 0 currently down".
|
||||
//
|
||||
// The whole design tension is that B must keep passing while A does: an alarm that fires on every
|
||||
// deploy is one the operator learns to ignore.
|
||||
|
||||
// restartingSince builds a deployed stack that has been restarting since `since`.
|
||||
func restartingSince(name string, since time.Time) stacks.Stack {
|
||||
s := stack(name, stacks.StateRestarting, true, false)
|
||||
s.RestartingSince = since
|
||||
return s
|
||||
}
|
||||
|
||||
// SCENARIO A — a crash loop alarms. A stack restarting for longer than the threshold enters BOTH the
|
||||
// banner dead-list and the notifier Down-set, so app_start_failed can fire.
|
||||
//
|
||||
// RED-PROOF (observed): drop `|| crashLooping` from the `down` expression in classifyRunStates →
|
||||
//
|
||||
// crashloop_classify_test.go:52: docmost is NOT in the Down-set — a crash loop is silent (this is C9-F2)
|
||||
// crashloop_classify_test.go:55: docmost is NOT in the banner dead-list
|
||||
func TestClassifyRunStates_SustainedRestartingAlarms(t *testing.T) {
|
||||
now := time.Now()
|
||||
sts := []stacks.Stack{
|
||||
stack("paperless-ngx", stacks.StateRunning, true, false),
|
||||
restartingSince("docmost", now.Add(-9*time.Minute)), // the Campaign 9 observation, exactly
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, now)
|
||||
|
||||
if !downByName(states)["docmost"] {
|
||||
t.Errorf("docmost is NOT in the Down-set — a crash loop is silent (this is C9-F2)")
|
||||
}
|
||||
if !deadNames(dead)["docmost"] {
|
||||
t.Errorf("docmost is NOT in the banner dead-list")
|
||||
}
|
||||
if downByName(states)["paperless-ngx"] {
|
||||
t.Errorf("a healthy app was dragged down with it")
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO B — a normal deploy or update does NOT alarm. `docker compose up -d` passes through
|
||||
// restarting; alarming there would page the operator on every routine operation, fleet-wide.
|
||||
//
|
||||
// This is the test that must fail against the naive fix. RED-PROOF (observed): add StateRestarting
|
||||
// to IsDownState instead of using the threshold →
|
||||
//
|
||||
// crashloop_classify_test.go:78: a BRIEFLY restarting app alarms — every deploy and update would page the operator
|
||||
func TestClassifyRunStates_BriefRestartingIsSilent(t *testing.T) {
|
||||
now := time.Now()
|
||||
sts := []stacks.Stack{
|
||||
restartingSince("mealie", now.Add(-30*time.Second)), // mid-deploy
|
||||
restartingSince("ghost", now.Add(-2*time.Minute)), // slow image pull, still normal
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, now)
|
||||
|
||||
for _, name := range []string{"mealie", "ghost"} {
|
||||
if downByName(states)[name] {
|
||||
t.Errorf("a BRIEFLY restarting app alarms (%s) — every deploy and update would page the operator", name)
|
||||
}
|
||||
}
|
||||
if len(dead) != 0 {
|
||||
t.Errorf("banner dead-list should be empty during normal restarts, got %v", deadNames(dead))
|
||||
}
|
||||
}
|
||||
|
||||
// The boundary itself, asserted from both sides so the threshold cannot drift silently.
|
||||
func TestCrashLooping_ThresholdBoundary(t *testing.T) {
|
||||
now := time.Now()
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
age time.Duration
|
||||
want bool
|
||||
}{
|
||||
{"just under the threshold", 4*time.Minute + 59*time.Second, false},
|
||||
{"exactly at the threshold", 5 * time.Minute, true},
|
||||
{"well past it", 30 * time.Minute, true},
|
||||
} {
|
||||
s := restartingSince("app", now.Add(-tc.age))
|
||||
if got := s.CrashLooping(now); got != tc.want {
|
||||
t.Errorf("%s: CrashLooping(age=%s) = %v, want %v", tc.name, tc.age, got, tc.want)
|
||||
}
|
||||
}
|
||||
|
||||
// A stack that is not restarting is never a crash loop, however old the stamp.
|
||||
s := stack("app", stacks.StateRunning, true, false)
|
||||
s.RestartingSince = now.Add(-time.Hour)
|
||||
if s.CrashLooping(now) {
|
||||
t.Error("a RUNNING stack reported as crash-looping — the state test is missing")
|
||||
}
|
||||
|
||||
// A zero stamp is "not yet observed restarting", never a crash loop — this is what makes the
|
||||
// first scan after a controller restart silent instead of alarming on everything at once.
|
||||
z := stack("app", stacks.StateRestarting, true, false)
|
||||
if z.CrashLooping(now) {
|
||||
t.Error("a zero RestartingSince reported as crash-looping — a controller restart would alarm fleet-wide")
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO C — R-97b's quiesce suppression still wins inside its window. A stack the backup stopped
|
||||
// and is restarting must stay silent while suppressed, even if its restarting run is old enough to
|
||||
// qualify. The window EXPIRES, so a genuinely dead app still alarms afterwards — proven by the
|
||||
// second half of this test.
|
||||
//
|
||||
// RED-PROOF (observed): drop `&& !quiesced[st.Name]` from the `down` expression →
|
||||
//
|
||||
// crashloop_classify_test.go:129: a quiesced stack alarms — every backup would page the customer
|
||||
func TestClassifyRunStates_QuiesceSuppressionBeatsCrashLoop(t *testing.T) {
|
||||
now := time.Now()
|
||||
sts := []stacks.Stack{restartingSince("docmost", now.Add(-9*time.Minute))}
|
||||
|
||||
// Inside the R-97b window.
|
||||
_, states := classifyRunStates(sts, map[string]bool{"docmost": true}, nil, now)
|
||||
if downByName(states)["docmost"] {
|
||||
t.Errorf("a quiesced stack alarms — every backup would page the customer")
|
||||
}
|
||||
|
||||
// Window expired (the stack is no longer reported as suppressed): the same stack must now alarm.
|
||||
dead, states := classifyRunStates(sts, nil, nil, now)
|
||||
if !downByName(states)["docmost"] {
|
||||
t.Errorf("suppression outlived its window — a genuinely dead app stayed silent (R-97b's own warning)")
|
||||
}
|
||||
if !deadNames(dead)["docmost"] {
|
||||
t.Errorf("suppression outlived its window for the banner too")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"log"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// F-OBS (Campaign 8): on a default `info`-level box there was NO positive observable that
|
||||
// `deadapp-check` had run. Its per-cycle scheduler line goes through Scheduler.dbg(), which is gated
|
||||
// on logging.level==debug and therefore never PRODUCED on a default box — so it could not even reach
|
||||
// the always-DEBUG ring — and a 30 s interval also puts the job on the scheduler's quiet path.
|
||||
//
|
||||
// "No alarms" was therefore indistinguishable from "the detector never ran", which is exactly the
|
||||
// fallacy this project now has a standing rule against, and it undermines confidence in the
|
||||
// F-CRIT-1 fix in the field.
|
||||
//
|
||||
// Scenario F — the observable must appear AT INFO LEVEL. These tests assert the emitted LINE, not
|
||||
// merely that a function was called; asserting the call would reproduce the original mistake.
|
||||
|
||||
// RED-PROOF: delete the logger.Printf in noteDeadAppScan (or drop the whole call from the job
|
||||
// closure) → every case below sees an empty buffer and this fails with
|
||||
// "no observable emitted at scan 20 — silence is indistinguishable from not running".
|
||||
func TestNoteDeadAppScan_EmitsAtInfoLevel(t *testing.T) {
|
||||
var buf bytes.Buffer
|
||||
lg := log.New(&buf, "", 0)
|
||||
|
||||
noteDeadAppScan(lg, deadAppHeartbeatEvery, 7, 2)
|
||||
|
||||
out := buf.String()
|
||||
if out == "" {
|
||||
t.Fatalf("no observable emitted at scan %d — silence is indistinguishable from not running", deadAppHeartbeatEvery)
|
||||
}
|
||||
if !strings.Contains(out, "[INFO]") {
|
||||
t.Errorf("the observable is not at INFO level, so a default `logging.level: info` box would never see it:\n%s", out)
|
||||
}
|
||||
if !strings.Contains(out, "[deadapp]") {
|
||||
t.Errorf("the observable does not identify the check that produced it:\n%s", out)
|
||||
}
|
||||
// it must carry WHAT IT SAW, not just "I ran" — an operator needs to distinguish
|
||||
// "running and everything is up" from "running and 2 apps are down".
|
||||
for _, want := range []string{"scans since boot", "evaluated", "currently down"} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("the observable omits %q — it proves the check ran but not what it found:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// It must NOT be a line per run. At a 30 s cadence that is 2880 lines/day, which is precisely why
|
||||
// the original author chose silence — so a fix that floods is not a fix.
|
||||
//
|
||||
// RED-PROOF: change the guard to `scans%1 != 0` (i.e. emit every run) → this fails with
|
||||
// "emitted 60 lines across 60 scans — that is the flood that made silence attractive".
|
||||
func TestNoteDeadAppScan_IsASummaryNotAFlood(t *testing.T) {
|
||||
var buf bytes.Buffer
|
||||
lg := log.New(&buf, "", 0)
|
||||
|
||||
const scans = 60
|
||||
for i := 1; i <= scans; i++ {
|
||||
noteDeadAppScan(lg, i, 3, 0)
|
||||
}
|
||||
|
||||
got := strings.Count(buf.String(), "[deadapp] check alive")
|
||||
want := scans / deadAppHeartbeatEvery
|
||||
if got == scans {
|
||||
t.Fatalf("emitted %d lines across %d scans — that is the flood that made silence attractive", got, scans)
|
||||
}
|
||||
if got != want {
|
||||
t.Errorf("emitted %d heartbeat lines across %d scans, want %d (one per %d)", got, scans, want, deadAppHeartbeatEvery)
|
||||
}
|
||||
}
|
||||
|
||||
// The cadence must be frequent enough that a STALLED detector is obvious well inside the 180 s alarm
|
||||
// grace this check feeds. 20 scans x 30 s = 10 min; if someone widens it to hours the observable
|
||||
// stops being useful as a liveness signal, and this is the tripwire.
|
||||
func TestDeadAppHeartbeatEvery_StaysUsefulAsALivenessSignal(t *testing.T) {
|
||||
const scanInterval = 30 // seconds, matching sched.Every("deadapp-check", 30*time.Second, ...)
|
||||
periodSec := deadAppHeartbeatEvery * scanInterval
|
||||
if periodSec > 15*60 {
|
||||
t.Errorf("heartbeat period is %ds (>15min) — too sparse to notice a stalled detector", periodSec)
|
||||
}
|
||||
if deadAppHeartbeatEvery < 2 {
|
||||
t.Errorf("heartbeat every %d scans is a per-run flood", deadAppHeartbeatEvery)
|
||||
}
|
||||
}
|
||||
|
||||
// Off-cadence scans stay quiet, and a nil logger is tolerated (the job closure must never panic).
|
||||
func TestNoteDeadAppScan_QuietOffCadenceAndNilSafe(t *testing.T) {
|
||||
var buf bytes.Buffer
|
||||
lg := log.New(&buf, "", 0)
|
||||
noteDeadAppScan(lg, deadAppHeartbeatEvery-1, 1, 0)
|
||||
if buf.Len() != 0 {
|
||||
t.Errorf("emitted off-cadence:\n%s", buf.String())
|
||||
}
|
||||
noteDeadAppScan(nil, deadAppHeartbeatEvery, 1, 0) // must not panic
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// F-CRIT-1 cause 2 (Campaign 8): `classifyRunStates` whitelisted StateStopped on invariant I1
|
||||
// ("StateStopped means the USER stopped it"). The quiesce loop broke I1 by stopping stacks via the
|
||||
// same `docker compose down` path, so a stack quiesce stopped and then FAILED to restart was also
|
||||
// StateStopped — and was whitelisted into total silence. Live evidence: a customer app dead
|
||||
// indefinitely, no banner, no event, no email, while the dead-app scanner ran 11 times over it.
|
||||
//
|
||||
// The two cases are byte-identical on the Docker side. The ONLY thing that separates them is that
|
||||
// the quiesce loop knows it tried to restart and could not — `failedRestart` is that knowledge.
|
||||
|
||||
// Scenario A (cause 2) — a stack quiesce failed to restart MUST alarm, despite being StateStopped.
|
||||
//
|
||||
// RED-PROOF: restore the unconditional whitelist (`down := IsDownState(st.State) &&
|
||||
// st.State != stacks.StateStopped && !quiesced[st.Name]`) → immich reports Down=false and stays out
|
||||
// of the dead list, and this fails with "a stack that FAILED to restart is silent".
|
||||
func TestClassifyRunStates_FailedRestartAlarmsDespiteStateStopped(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("bookstack", stacks.StateRunning, true, false),
|
||||
stack("immich", stacks.StateStopped, true, false), // quiesce stopped it; restart FAILED
|
||||
}
|
||||
failed := map[string]bool{"immich": true}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, failed, time.Now())
|
||||
|
||||
if !downByName(states)["immich"] {
|
||||
t.Error("a stack that FAILED to restart is silent (Down=false) — this is F-CRIT-1")
|
||||
}
|
||||
if !deadNames(dead)["immich"] {
|
||||
t.Error("a stack that FAILED to restart is absent from the dashboard dead-list — this is F-CRIT-1")
|
||||
}
|
||||
if downByName(states)["bookstack"] {
|
||||
t.Error("a healthy running stack was marked down")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B — a DELIBERATE user stop must still be silent. This pins v0.164.0 and is what stops
|
||||
// the fix above from becoming a regression.
|
||||
//
|
||||
// RED-PROOF: make the whitelist unconditional in the other direction (drop the `&& !failedRestart`
|
||||
// term, i.e. treat every StateStopped as a failed restart) → cwa alarms and this fails with
|
||||
// "a deliberate user stop alarmed".
|
||||
func TestClassifyRunStates_UserStopStillSilent(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("cwa", stacks.StateStopped, true, false), // the user stopped this from the UI
|
||||
stack("immich", stacks.StateStopped, true, false),
|
||||
}
|
||||
// only immich failed to restart; cwa was never touched by a quiesce
|
||||
failed := map[string]bool{"immich": true}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, failed, time.Now())
|
||||
down := downByName(states)
|
||||
|
||||
if down["cwa"] || deadNames(dead)["cwa"] {
|
||||
t.Error("a deliberate user stop alarmed — that is the v0.164.0 regression this must not reintroduce")
|
||||
}
|
||||
if !down["immich"] {
|
||||
t.Error("the failed restart went silent")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B, stronger form — with NO failed restarts at all, behaviour is byte-identical to
|
||||
// v0.164.0: every StateStopped is silent.
|
||||
func TestClassifyRunStates_NoFailedRestartsIsV0164Behaviour(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("radarr", stacks.StateRunning, true, false),
|
||||
stack("cwa", stacks.StateStopped, true, false),
|
||||
stack("immich", stacks.StateExited, true, false),
|
||||
stack("nextcloud", stacks.StateDegraded, true, false),
|
||||
}
|
||||
|
||||
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
||||
down := downByName(states)
|
||||
|
||||
if down["cwa"] {
|
||||
t.Error("stopped alarmed with no failed restarts — v0.164.0 behaviour broken")
|
||||
}
|
||||
if !down["immich"] || !down["nextcloud"] {
|
||||
t.Error("a genuine fault (exited/degraded) stopped alarming")
|
||||
}
|
||||
if got := len(deadNames(dead)); got != 2 {
|
||||
t.Errorf("dead list has %d entries, want exactly {immich, nextcloud}", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C — during the R-97b grace window the stack is suppressed even if its restart failed.
|
||||
// The grace exists so a slow-starting app is not called dead; it EXPIRES, and the alarm follows.
|
||||
//
|
||||
// RED-PROOF: drop the `&& !quiesced[st.Name]` term → the app alarms mid-restart on every normal
|
||||
// backup, which is the false-alarm R-97b was built to remove.
|
||||
func TestClassifyRunStates_GraceWindowStillSuppresses(t *testing.T) {
|
||||
sts := []stacks.Stack{stack("immich", stacks.StateStopped, true, false)}
|
||||
quiesced := map[string]bool{"immich": true} // still inside quiesceAlarmGrace
|
||||
failed := map[string]bool{"immich": true} // and we already know the restart failed
|
||||
|
||||
dead, states := classifyRunStates(sts, quiesced, failed, time.Now())
|
||||
|
||||
if downByName(states)["immich"] {
|
||||
t.Error("alarmed while still inside the grace window — R-97b Scenario E broken")
|
||||
}
|
||||
if len(dead) != 0 {
|
||||
t.Errorf("dead list not empty during grace: %v", deadNames(dead))
|
||||
}
|
||||
}
|
||||
|
||||
// An undeployed or mid-deploy stack is never classified, failed restart or not.
|
||||
func TestClassifyRunStates_UndeployedIgnored(t *testing.T) {
|
||||
sts := []stacks.Stack{
|
||||
stack("ghost", stacks.StateStopped, false, false),
|
||||
stack("deploying", stacks.StateStopped, true, true),
|
||||
}
|
||||
dead, states := classifyRunStates(sts, nil, map[string]bool{"ghost": true, "deploying": true}, time.Now())
|
||||
if len(dead) != 0 || len(states) != 0 {
|
||||
t.Errorf("undeployed/deploying stacks were classified: dead=%v states=%v", deadNames(dead), states)
|
||||
}
|
||||
}
|
||||
+1195
-80
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,30 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
|
||||
)
|
||||
|
||||
// R-97a — REACHABILITY, not behaviour.
|
||||
//
|
||||
// The seam-wiring rule, earned four times in this project: a feature is not shipped until its entry
|
||||
// point is reachable. `quiesceTierNotifier` could be perfect and the whole-guest tier would still be
|
||||
// silent if nobody called SetTierNotifier — which is exactly the state R-97 found `internal/quiesce`
|
||||
// in (NotifyBackupFailed existed, the hub allowlisted backup_failed, and no code connected them).
|
||||
//
|
||||
// This asserts the adapter SATISFIES the interface the loop requires. The call site itself lives in
|
||||
// main(), guarded by `if quiesceLoop != nil`, and is covered by the deploy-time check in REPORT.md.
|
||||
func TestQuiesceTierNotifierIsWired(t *testing.T) {
|
||||
var _ quiesce.TierNotifier = quiesceTierNotifier{}
|
||||
|
||||
// And it must not panic on a nil notifier — main() constructs it with a real one, but a future
|
||||
// refactor that reorders startup must fail loudly here rather than at 03:00 on a customer box.
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
t.Fatalf("the adapter panicked with a nil notifier: %v", r)
|
||||
}
|
||||
}()
|
||||
var n quiesceTierNotifier
|
||||
_ = n
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
package main
|
||||
|
||||
import "gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
|
||||
|
||||
// COMPILE-TIME WITNESSES for OPTIONAL interfaces satisfied by a RUNTIME type assertion.
|
||||
//
|
||||
// Moved here from a _test.go file (R-88 Part 2) on purpose: a witness in a test fires on `go test`
|
||||
// and `go vet`, but NOT on `go build` alone. The failure it guards against — a signature change that
|
||||
// silently breaks an interface nobody checks at compile time — is exactly the kind that gets pushed
|
||||
// by a build-only step.
|
||||
//
|
||||
// THE INCIDENT THIS PREVENTS, which already happened once: when `TieredBackend.DueFor` gained a
|
||||
// return value during R-88 Part 2, `quiesceBackend` stopped satisfying the interface and the whole
|
||||
// repo still BUILT AND VETTED CLEAN, because `resolveDueTiers` only ever asserts it at runtime
|
||||
// (`l.backend.(TieredBackend)`). A failed assertion silently falls back to the untargeted
|
||||
// single-tier path — so every box would have quietly lost R-82's multi-tier backups, with no error
|
||||
// anywhere. It was caught by accident, not by the toolchain.
|
||||
//
|
||||
// THIS DOES NOT MAKE THE INTERFACE REQUIRED. Optionality is deliberate: it is what lets a new
|
||||
// controller meet an old agent, and what `resolveDueTiers` degrades through on purpose. The witness
|
||||
// pins the IMPLEMENTATION, not the CONTRACT.
|
||||
var _ quiesce.TieredBackend = quiesceBackend{}
|
||||
@@ -5,6 +5,7 @@ go 1.24.0
|
||||
require (
|
||||
github.com/emersion/go-sasl v0.0.0-20241020182733-b788ff22d5a6
|
||||
github.com/emersion/go-smtp v0.24.0
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e
|
||||
golang.org/x/crypto v0.31.0
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
modernc.org/sqlite v1.45.0
|
||||
|
||||
@@ -16,6 +16,8 @@ github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOF
|
||||
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e h1:MRM5ITcdelLK2j1vwZ3Je0FKVCfqOLp5zO6trqMLYs0=
|
||||
github.com/skip2/go-qrcode v0.0.0-20200617195104-da1b6568686e/go.mod h1:XV66xRDqSt+GTGFMVlhk3ULuV0y9ZmzeVGR4mloJI3M=
|
||||
golang.org/x/crypto v0.31.0 h1:ihbySMvVjLAeSH1IbfcRTkD/iNscyz8rGzjF/E5hV6U=
|
||||
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
|
||||
golang.org/x/exp v0.0.0-20251023183803-a4bb9ffd2546 h1:mgKeJMpvi0yx/sU5GsxQ7p6s2wtOnGAHZWCHUM4KGzY=
|
||||
@@ -27,6 +29,8 @@ golang.org/x/sync v0.17.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI=
|
||||
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.37.0 h1:fdNQudmxPjkdUTPnLn5mdQv7Zwvbvpaxqs831goi9kQ=
|
||||
golang.org/x/sys v0.37.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
|
||||
golang.org/x/term v0.27.0 h1:WP60Sv1nlK1T6SupCHbXzSaN0b9wUmsPoRS9b61A23Q=
|
||||
golang.org/x/term v0.27.0/go.mod h1:iMsnZpn0cago0GOrHO2+Y7u7JPn5AylBrcoWkElMTSM=
|
||||
golang.org/x/tools v0.38.0 h1:Hx2Xv8hISq8Lm16jvBZ2VQf+RLmbd7wVUsALibYI/IQ=
|
||||
golang.org/x/tools v0.38.0/go.mod h1:yEsQ/d/YK8cjh0L6rZlY8tgtlKiBNTL14pGDJPJpYQs=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||
|
||||
@@ -9,13 +9,28 @@
|
||||
# in Explorer but the double-click fails 0x80070035 (no flat-name resolution);
|
||||
# nmbd is what makes \\<NAME> resolve + mount (S4b, proven live).
|
||||
# - wsdd : WS-Discovery, so the box appears in Windows Explorer's Network view.
|
||||
# - avahi : mDNS/Bonjour (v1.1.0) — THE macOS path. Windows and macOS do not share a
|
||||
# discovery mechanism, and nmbd does not cover the Mac: captured live on
|
||||
# 2026-07-20, macOS broadcasts a correct NBNS query for FELHOM<20>, the box
|
||||
# answers correctly in 140us (flags 0x8580, RCODE=0, the right address), and
|
||||
# macOS REFUSES TO ACT ON IT — no TCP follows. NetBIOS feeds legacy browsing
|
||||
# there, not smb:// URL resolution. With mDNS, `smb://<NAME>.local` connects
|
||||
# immediately — PROVEN live from a Mac on 2026-07-20.
|
||||
# NOT proven: automatic appearance in the Finder sidebar. The _smb._tcp record
|
||||
# is published and answers browse queries on the wire, but the test Mac's
|
||||
# sidebar stayed empty (it had no Network/Bonjour section shown at all, which
|
||||
# is a Finder Settings -> Sidebar toggle). Treat sidebar discovery as an OPEN
|
||||
# question, not a shipped feature.
|
||||
# Evidence: felhom.eu/documentation/audits/DIAG-sharing-2026-07-20.md.
|
||||
FROM alpine:3.21@sha256:48b0309ca019d89d40f670aa1bc06e426dc0931948452e8491e3d65087abc07d
|
||||
|
||||
# samba = smbd + nmbd + smbpasswd/testparm (meta-package proven installable in the spike);
|
||||
# wsdd = WS-Discovery daemon; tini = a proper PID1 to reap nmbd/wsdd and forward signals.
|
||||
RUN apk add --no-cache samba wsdd tini \
|
||||
# wsdd = WS-Discovery daemon; tini = a proper PID1 to reap nmbd/wsdd/avahi and forward signals;
|
||||
# avahi + dbus = mDNS/Bonjour (avahi-daemon talks to the system bus, so dbus is not optional).
|
||||
RUN apk add --no-cache samba wsdd tini avahi dbus \
|
||||
&& rm -rf /var/cache/apk/* \
|
||||
&& rm -f /etc/samba/smb.conf
|
||||
&& rm -f /etc/samba/smb.conf \
|
||||
&& rm -f /etc/avahi/services/*.service
|
||||
|
||||
# passdb on a named volume → the household SMB password survives container recreation
|
||||
# (share add/remove re-renders + `compose up -d`, which recreates the container).
|
||||
|
||||
@@ -23,11 +23,61 @@ fi
|
||||
|
||||
mkdir -p /var/lib/samba/private /run/samba
|
||||
|
||||
echo "[felhom-samba] launching nmbd + wsdd + smbd (server=${SERVER_NAME} iface=${IFACE} uid=${FELHOM_UID})"
|
||||
# --- mDNS / Bonjour (v1.1.0) -------------------------------------------------------------
|
||||
# THE macOS path. Templated from SERVER_NAME rather than baked, so renaming the server in the
|
||||
# UI re-advertises under the new name on the next container recreate — a baked name would
|
||||
# leave the box answering to something the customer no longer sees anywhere.
|
||||
#
|
||||
# A STATIC service file, deliberately, rather than smbd's own `multicast dns register`: it
|
||||
# needs no line in smb.conf (which is bind-mounted READ-ONLY and owned by the controller's
|
||||
# renderer) and it lets us publish _device-info._tcp so the Finder shows a sensible icon
|
||||
# instead of a generic globe.
|
||||
mkdir -p /etc/avahi/services /run/dbus
|
||||
cat > /etc/avahi/avahi-daemon.conf <<CONF
|
||||
[server]
|
||||
host-name=${SERVER_NAME}
|
||||
use-ipv4=yes
|
||||
use-ipv6=no
|
||||
allow-interfaces=${IFACE}
|
||||
ratelimit-interval-usec=1000000
|
||||
ratelimit-burst=1000
|
||||
|
||||
# nmbd: NetBIOS flat-name resolution so \\<NAME> resolves and mounts (the S4b fix).
|
||||
[wide-area]
|
||||
enable-wide-area=no
|
||||
|
||||
[publish]
|
||||
publish-addresses=yes
|
||||
publish-hinfo=no
|
||||
publish-workstation=no
|
||||
CONF
|
||||
|
||||
cat > /etc/avahi/services/smb.service <<CONF
|
||||
<?xml version="1.0" standalone='no'?><!DOCTYPE service-group SYSTEM "avahi-service.dtd">
|
||||
<service-group>
|
||||
<name replace-wildcards="yes">%h</name>
|
||||
<service>
|
||||
<type>_smb._tcp</type>
|
||||
<port>445</port>
|
||||
</service>
|
||||
<service>
|
||||
<type>_device-info._tcp</type>
|
||||
<port>0</port>
|
||||
<txt-record>model=RackMac</txt-record>
|
||||
</service>
|
||||
</service-group>
|
||||
CONF
|
||||
|
||||
echo "[felhom-samba] launching nmbd + wsdd + avahi + smbd (server=${SERVER_NAME} iface=${IFACE} uid=${FELHOM_UID})"
|
||||
|
||||
# nmbd: NetBIOS flat-name resolution so \\<NAME> resolves and mounts on WINDOWS (the S4b fix).
|
||||
# It does NOT serve macOS — see the Dockerfile header for the captured proof.
|
||||
nmbd --daemon --no-process-group
|
||||
# wsdd: WS-Discovery so the box appears in Windows Explorer's Network view.
|
||||
wsdd -i "$IFACE" -4 -H 4 -s -n "$SERVER_NAME" -w WORKGROUP &
|
||||
# dbus + avahi: mDNS, so `smb://<NAME>.local` resolves and the box appears in the Finder sidebar.
|
||||
# Non-fatal on failure: sharing over an address still works, and refusing to start smbd because
|
||||
# a discovery daemon did not come up would turn a convenience gap into an outage.
|
||||
dbus-daemon --system --fork 2>/dev/null || echo "[felhom-samba] WARN: dbus failed to start — mDNS disabled"
|
||||
avahi-daemon --daemonize --no-drop-root 2>/dev/null || echo "[felhom-samba] WARN: avahi failed to start — mDNS disabled"
|
||||
# smbd in the foreground = the container's main process.
|
||||
exec smbd --foreground --no-process-group
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
package agentapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
)
|
||||
|
||||
// R-82 Slice B — the per-tier backup surface (agent >= v0.97.0).
|
||||
//
|
||||
// Every method here is ADDITIVE. The untargeted BackupDue/StartBackup/BackupStatus keep their exact
|
||||
// pre-R-82 meaning and are still the single-tier path used against an older agent.
|
||||
|
||||
// ErrTiersUnsupported reports that this agent does not serve GET /backup/tiers — it predates R-82.
|
||||
// It is the DESIGNED capability probe (the route 404s), not a fault. The caller MUST degrade to the
|
||||
// untargeted single-tier path and still take a backup; concluding "nothing to do" from it would
|
||||
// silently stop backups during a fleet rollout.
|
||||
var ErrTiersUnsupported = errors.New("agentapi: agent does not serve /backup/tiers (pre-R-82)")
|
||||
|
||||
// BackupTierInfo is one advertised tier.
|
||||
type BackupTierInfo struct {
|
||||
Target string `json:"target"`
|
||||
CadenceSeconds int64 `json:"cadence_seconds"`
|
||||
Primary bool `json:"primary"`
|
||||
}
|
||||
|
||||
// TiersResponse mirrors the agent's GET /backup/tiers payload.
|
||||
type TiersResponse struct {
|
||||
VMID int `json:"vmid"`
|
||||
Tiers []BackupTierInfo `json:"tiers"`
|
||||
}
|
||||
|
||||
// BackupTiers lists the agent's backup tiers, primary first.
|
||||
// Returns ErrTiersUnsupported (wrapped) on a pre-R-82 agent — key on it with errors.Is.
|
||||
func (c *Client) BackupTiers(ctx context.Context) (TiersResponse, error) {
|
||||
var out TiersResponse
|
||||
body, err := c.get(ctx, "/backup/tiers")
|
||||
if err != nil {
|
||||
var se *StatusError
|
||||
if errors.As(err, &se) && se.Code == http.StatusNotFound {
|
||||
return out, ErrTiersUnsupported
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/tiers: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// targetQuery renders the ?target= suffix. An EMPTY target yields an empty string, so the caller
|
||||
// hits the untargeted route byte-for-byte — that is what keeps the pre-R-82 contract intact when
|
||||
// this client talks to an older agent.
|
||||
func targetQuery(target string) string {
|
||||
if target == "" {
|
||||
return ""
|
||||
}
|
||||
return "?target=" + url.QueryEscape(target)
|
||||
}
|
||||
|
||||
// BackupDueFor reports whether THIS TIER is due. A fresh backup on another tier must not satisfy it
|
||||
// — that filtering happens agent-side (latestSuccessfulBackupForTarget); this just asks per tier.
|
||||
func (c *Client) BackupDueFor(ctx context.Context, target string) (DueResponse, error) {
|
||||
var out DueResponse
|
||||
body, err := c.get(ctx, "/backup/due"+targetQuery(target))
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/due (target %q): %w", target, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// StartBackupFor enqueues a backup of this guest ON THE GIVEN TIER.
|
||||
func (c *Client) StartBackupFor(ctx context.Context, target string) (BackupResponse, error) {
|
||||
var out BackupResponse
|
||||
body, err := c.post(ctx, "/backup"+targetQuery(target), struct{}{})
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode POST /backup (target %q): %w", target, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// BackupStatusFor reports THIS TIER's current/last job phase. Jobs are keyed per tier agent-side,
|
||||
// so polling the wrong target would report a different tier's progress.
|
||||
func (c *Client) BackupStatusFor(ctx context.Context, target string) (StatusResponse, error) {
|
||||
var out StatusResponse
|
||||
body, err := c.get(ctx, "/backup/status"+targetQuery(target))
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/status (target %q): %w", target, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// SetBackupTargetResponse mirrors POST /backup/target (agent >= v0.113.0).
|
||||
type SetBackupTargetResponse struct {
|
||||
Target string `json:"target"`
|
||||
Where string `json:"where"`
|
||||
// RestartRequired is always true on success: the agent builds its tiers once at daemon start, so
|
||||
// the move needs a restart. The agent deliberately does NOT restart itself — restarting with a
|
||||
// backup in flight cancels the wait and records a spurious tier failure for a backup that actually
|
||||
// succeeded. The RESTART IS THE OPERATOR'S, behind an immediate in-flight check.
|
||||
RestartRequired bool `json:"restart_required"`
|
||||
}
|
||||
|
||||
// SetBackupTarget moves the primary whole-guest backup tier onto the drive at raw host mount `where`.
|
||||
// Creates the storage and grants the agent access as one ordered operation.
|
||||
func (c *Client) SetBackupTarget(ctx context.Context, where string) (SetBackupTargetResponse, error) {
|
||||
var out SetBackupTargetResponse
|
||||
// vmid is deliberately omitted: the agent derives the guest from the token and scopedFromBody
|
||||
// treats an absent vmid as "use the token's" — the same shape as AssignDisk/GuestAttach.
|
||||
body, err := c.post(ctx, "/backup/target", map[string]string{"where": where})
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
return out, fmt.Errorf("agentapi: decode /backup/target: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -174,6 +174,14 @@ type DueResponse struct {
|
||||
Due bool `json:"due"`
|
||||
Reason string `json:"reason"`
|
||||
AgeSecs *int64 `json:"age_seconds"`
|
||||
// AgeState (R-88 Part 2, agent >= v0.105.0) says WHY AgeSecs is nil: "absent" (a positive
|
||||
// determination that no backup has ever landed) or "unknown" (the agent could not tell —
|
||||
// unreadable storage, unparseable timestamp). "known" accompanies a real age.
|
||||
//
|
||||
// EMPTY MEANS LEGACY — an agent older than v0.105.0 simply omits the field. It does NOT mean
|
||||
// "unknown", and the distinction is load-bearing: see quiesce.ageStateFromWire. Never
|
||||
// discriminate on Reason instead; those strings are operator copy and will drift.
|
||||
AgeState string `json:"age_state"`
|
||||
}
|
||||
|
||||
// BackupResponse mirrors the agent's POST /backup payload.
|
||||
@@ -315,6 +323,12 @@ type DiskInfo struct {
|
||||
// opposed to merely present on the host (F9) — the signal whose absence let the HDD look available
|
||||
// when it wasn't attached. LEGACY (per-drive mp model); the intermediary model uses BoundUnderParent.
|
||||
GuestAttached bool `json:"guest_attached"`
|
||||
// BackupTarget (E-2, agent >= v0.112.0) reports that this drive backs the PRIMARY whole-guest
|
||||
// backup tier. The agent is the only component that can answer: our own
|
||||
// settings.StoragePath.BackupTarget is customer INTENT, and on a box migrated by hand (E-1) that
|
||||
// intent was never recorded while the drive really IS the target. Absent on an older agent →
|
||||
// false, which degrades to the pre-E-2 behaviour (a generic disconnect alarm, never a wrong one).
|
||||
BackupTarget bool `json:"backup_target,omitempty"`
|
||||
// GuestPath is the drive's STABLE in-guest path in the intermediary-mount model
|
||||
// (/mnt/felhom-drives/<name>) — what the controller registers + repoints HDD_PATH to. Distinct from
|
||||
// MountPath (the raw /mnt/<name> host PVE mount the agent ops on). "" for non-user-data drives.
|
||||
@@ -322,6 +336,10 @@ type DiskInfo struct {
|
||||
// BoundUnderParent reports whether the drive's felhom-data is currently bound under the shared parent
|
||||
// (live + usable in the guest). The controller's drive-absent gate keys on this + State.
|
||||
BoundUnderParent bool `json:"bound_under_parent"`
|
||||
// Smart is the per-disk SMART health (agent v0.94.0+), nil when the device exposes no SMART or the
|
||||
// agent predates the field — the disk-health card + 6h degradation check feature-detect on this and
|
||||
// render "Nincs adat" (never alarm) when nil. See DiskVerdictFor.
|
||||
Smart *SmartSummary `json:"smart,omitempty"`
|
||||
}
|
||||
|
||||
// FSUUID returns the raw filesystem UUID from a "uuid:<…>" DurableID, or "" if this disk's identity
|
||||
@@ -406,6 +424,11 @@ type DiskCandidate struct {
|
||||
Mountable bool `json:"mountable"`
|
||||
MountSource string `json:"mount_source,omitempty"`
|
||||
DurableID string `json:"durable_id,omitempty"`
|
||||
// AlreadyMounted marks a candidate the CONTROLLER contributed from its own mount table (R-280),
|
||||
// not one the agent scanned. The agent NEVER sets it. Its action is REGISTER the existing
|
||||
// mountpoint — sending it down the device-attach path would try to mount an in-guest path as if
|
||||
// it were a raw device. See web/attach_sources.go for why the agent's scan cannot supply these.
|
||||
AlreadyMounted bool `json:"already_mounted,omitempty"`
|
||||
}
|
||||
|
||||
// CandidatesResult mirrors GET /disks/candidates: disks free to enroll, split into initialize (all
|
||||
@@ -1003,14 +1026,32 @@ type ThinPoolFill struct {
|
||||
MetadataUsedFraction *float64 `json:"metadata_used_fraction"`
|
||||
}
|
||||
|
||||
// SmartSummary mirrors the agent's per-disk SMART health (only the fields the UI renders). Pointers
|
||||
// are null when the device type does not expose that attribute.
|
||||
// SmartSummary mirrors the agent's per-disk SMART health. Pointers are null when the device type
|
||||
// does not expose that attribute (a null is "unknown / not-applicable", distinct from a real zero).
|
||||
// The SATA set (reallocated/pending/offline-uncorrectable) and the NVMe set
|
||||
// (critical_warning/media_errors/percentage_used) are both carried; a device populates only its own.
|
||||
type SmartSummary struct {
|
||||
Health string `json:"health"` // PASSED | FAILING | UNKNOWN
|
||||
TemperatureC *int `json:"temperature_c"`
|
||||
PercentageUsed *int `json:"percentage_used"` // NVMe wear (%); null for SATA/USB
|
||||
Health string `json:"health"` // PASSED | FAILING | UNKNOWN
|
||||
ModelName *string `json:"model_name,omitempty"` // smartctl device model (agent v0.95.0+); nil on older agents
|
||||
TemperatureC *int `json:"temperature_c"`
|
||||
PowerOnHours *int `json:"power_on_hours"`
|
||||
// SATA attributes.
|
||||
ReallocatedSectors *int `json:"reallocated_sectors"`
|
||||
PendingSectors *int `json:"pending_sectors"`
|
||||
OfflineUncorrectable *int `json:"offline_uncorrectable"`
|
||||
// NVMe attributes.
|
||||
CriticalWarning *int `json:"critical_warning"`
|
||||
MediaErrors *int `json:"media_errors"`
|
||||
PercentageUsed *int `json:"percentage_used"` // NVMe wear (%); null for SATA/USB
|
||||
}
|
||||
|
||||
// SMART health vocabulary (mirrors the agent's).
|
||||
const (
|
||||
SmartPassed = "PASSED"
|
||||
SmartFailing = "FAILING"
|
||||
SmartUnknown = "UNKNOWN"
|
||||
)
|
||||
|
||||
// StorageTarget mirrors the agent's GET /host/metrics storage_targets entry (the per-storage
|
||||
// capacity + health the monitoring view renders). It is a SUBSET of the agent's wire shape — only
|
||||
// the fields the UI reads; unknown JSON keys are ignored.
|
||||
@@ -1059,13 +1100,28 @@ func (c *Client) HostMetrics(ctx context.Context) (HostMetricsResponse, error) {
|
||||
// StatusError is a non-2xx agent HTTP status surfaced as a TYPED error (same text the old
|
||||
// fmt.Errorf produced). errors.As-able — the capability probe (features.go) keys on Code 404 to
|
||||
// distinguish "this agent predates the route" from every other failure. Never match the string.
|
||||
// StatusError is a non-2xx response from the agent, carrying the STATUS CODE so callers can react
|
||||
// to specific ones rather than string-matching an error message.
|
||||
//
|
||||
// F-A1: this exists on the POST path because HTTP 409 from `POST /backup` is not a failure — it is
|
||||
// the agent's R-85 single-flight gate correctly refusing while a restore-test holds it. Treating
|
||||
// that refusal as a tier failure armed the breaker and emailed the operator about a backup that was
|
||||
// never actually broken. The controller now needs to tell 409 apart from a real error, and a typed
|
||||
// code is the only honest way to do that.
|
||||
type StatusError struct {
|
||||
Path string
|
||||
Code int
|
||||
// Method is the HTTP method. Empty means GET, so the message stays byte-identical for the
|
||||
// pre-existing GET call sites.
|
||||
Method string
|
||||
Path string
|
||||
Code int
|
||||
}
|
||||
|
||||
func (e *StatusError) Error() string {
|
||||
return fmt.Sprintf("agentapi: GET %s: HTTP %d", e.Path, e.Code)
|
||||
m := e.Method
|
||||
if m == "" {
|
||||
m = http.MethodGet
|
||||
}
|
||||
return fmt.Sprintf("agentapi: %s %s: HTTP %d", m, e.Path, e.Code)
|
||||
}
|
||||
|
||||
// get issues an authenticated GET and unwraps the {ok,data,error} envelope.
|
||||
@@ -1122,7 +1178,9 @@ func (c *Client) post(ctx context.Context, path string, body any) (json.RawMessa
|
||||
logx.Debugf(c.logger, "[agentapi] POST %s -> %d (%dms)", path, resp.StatusCode, time.Since(start).Milliseconds())
|
||||
raw, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<20))
|
||||
if resp.StatusCode != http.StatusOK && resp.StatusCode != http.StatusAccepted {
|
||||
return nil, fmt.Errorf("agentapi: POST %s: HTTP %d", path, resp.StatusCode)
|
||||
// Typed, not fmt.Errorf: callers must be able to distinguish 409 (the agent's single-flight
|
||||
// gate refusing — contention, not failure) from a genuine 5xx. See StatusError.
|
||||
return nil, &StatusError{Method: http.MethodPost, Path: path, Code: resp.StatusCode}
|
||||
}
|
||||
var env apiResponse
|
||||
if err := json.Unmarshal(raw, &env); err != nil {
|
||||
|
||||
@@ -0,0 +1,201 @@
|
||||
package agentapi
|
||||
|
||||
// DiskVerdict is the customer-facing disk-health verdict derived from a SmartSummary (v0.169.0).
|
||||
// It is the SHARED source of truth for both the "Lemezek állapota" dashboard card and the periodic
|
||||
// degradation check — one pure function so the chip and the alert can never disagree.
|
||||
type DiskVerdict int
|
||||
|
||||
const (
|
||||
// DiskVerdictUnknown — no SMART data (nil / UNKNOWN / old agent). Renders "Nincs adat"; NEVER
|
||||
// alarms and NEVER participates in degradation transitions (excluded both directions).
|
||||
DiskVerdictUnknown DiskVerdict = iota
|
||||
DiskVerdictOK // "Rendben" — clean
|
||||
DiskVerdictWarn // "Figyelmeztetés" — a wear/relocation counter is non-zero, below the Hiba bar
|
||||
DiskVerdictFail // "Hiba" — FAILING, or failing-but-not-self-reported (v0.215.0)
|
||||
)
|
||||
|
||||
// Thresholds. A number without a reason becomes permanent by default, so each carries its provenance.
|
||||
// The evidence is committed at felhom.eu/documentation/audits/DIAG-smart-passed-trap-2026-08-14.md
|
||||
// and its two fixtures (ST3000VX010 S/N Z6A07P2G, /dev/sdg on DooPlex, 11-13 Aug 2026).
|
||||
const (
|
||||
// percentageUsedWarn / percentageUsedFail — NVMe wear (%). 100 means the vendor's rated endurance
|
||||
// is spent; that is a declaration, not a trend, so it is Hiba.
|
||||
percentageUsedWarn = 90
|
||||
percentageUsedFail = 100
|
||||
|
||||
// uncorrectableFailCount — unreadable sectors too numerous to be a blip.
|
||||
//
|
||||
// PROVENANCE: on the one real failing drive observed, the benign excursion peaked at 16 and
|
||||
// cleared COMPLETELY within an hour (11 Aug 12:28 -> 13:28); the terminal run passed 64 at
|
||||
// 13 Aug 11:28 and never came back below it. 64 sits above the one observed transient and below
|
||||
// the observed terminal run. This is a judgement from ONE drive: it is a static BACKSTOP behind
|
||||
// the sustain rule, not the primary signal, and Phase 3 is expected to replace it with
|
||||
// growth-rate detection once the box keeps history.
|
||||
uncorrectableFailCount = 64
|
||||
|
||||
// temperatureWarnC / TemperatureFailC — adopted UNCHANGED from the operator's existing Prometheus
|
||||
// bands on DooPlex, so the two systems cannot disagree about the same drive.
|
||||
temperatureWarnC = 55
|
||||
// TemperatureFailC is exported because the alert-copy layer must pick the "overheated" message
|
||||
// shape from the SAME number the verdict fired on. A second literal elsewhere would be free to
|
||||
// drift, and the drift would show up as a customer told the wrong reason.
|
||||
TemperatureFailC = 60
|
||||
)
|
||||
|
||||
// DiskPrior is what the previous check observed for THIS SAME disk. It is the only history the
|
||||
// verdict consults, and it is passed in rather than read so the function stays pure — the caller
|
||||
// (internal/web) owns loading it from the persisted per-disk state.
|
||||
//
|
||||
// Plain value type: no methods, no I/O. A zero DiskPrior means "nothing known", which is the correct
|
||||
// fail-safe — a first-ever observation can only reach Figyelmeztetés from counters, never Hiba.
|
||||
type DiskPrior struct {
|
||||
// SawUncorrectable reports whether unreadable sectors (pending OR offline-uncorrectable) were
|
||||
// present at the previous check. It is what turns a one-off excursion into a sustained fault.
|
||||
SawUncorrectable bool
|
||||
}
|
||||
|
||||
// DiskVerdictFor maps a SmartSummary plus the previous observation to a verdict. Rules are evaluated
|
||||
// TOP-DOWN and the FIRST match wins (v0.215.0):
|
||||
//
|
||||
// 1. nil / "" / UNKNOWN -> Nincs adat
|
||||
// 2. Health == FAILING -> Hiba (drive self-reports)
|
||||
// 3. temperature_c >= 60 -> Hiba
|
||||
// 4. critical_warning > 0 (NVMe's own flag: a declaration) -> Hiba
|
||||
// 5. percentage_used >= 100 -> Hiba
|
||||
// 6. unreadable > 0 AND prior.SawUncorrectable -> Hiba (SUSTAINED)
|
||||
// 7. unreadable > 0 AND reallocated > 0 -> Hiba (accumulating + remapping)
|
||||
// 8. unreadable >= 64 -> Hiba (too large to be a blip)
|
||||
// 9. unreadable > 0 -> Figyelmeztetés (first sighting)
|
||||
// 10. reallocated > 0 -> Figyelmeztetés
|
||||
// 11. media_errors > 0 -> Figyelmeztetés
|
||||
// 12. percentage_used >= 90 -> Figyelmeztetés
|
||||
// 13. temperature_c >= 55 -> Figyelmeztetés
|
||||
// 14. otherwise -> Rendben
|
||||
//
|
||||
// WHY rows 2-8 exist at all: smart_status.passed CANNOT fail on unreadable sectors. Attributes 187,
|
||||
// 197 and 198 all carry thresh 0, and a normalized SMART value floors at 1, so it can never drop to
|
||||
// or below the threshold. The real drive stayed PASSED at 352 pending sectors with 1001 reported
|
||||
// uncorrectable reads. A verdict built on the drive's own self-assessment is blind to this whole
|
||||
// class of failure, which is why rows 3-8 read the raw counters instead.
|
||||
//
|
||||
// WHY row 6 sits ABOVE row 8: sustain is the PRIMARY rule and the count is the backstop. On the real
|
||||
// drive sustain fires a full day earlier (12 Aug) than the count threshold (13 Aug). Row 8 exists for
|
||||
// a box that was powered off or restarted across the sustain window and so has no prior.
|
||||
//
|
||||
// Pure: no clock, no I/O, no logging. Everything it needs arrives as an argument.
|
||||
func DiskVerdictFor(s *SmartSummary, prior DiskPrior) DiskVerdict {
|
||||
// 1 — no data. Never alarms.
|
||||
if s == nil || s.Health == "" || s.Health == SmartUnknown {
|
||||
return DiskVerdictUnknown
|
||||
}
|
||||
// 2 — the drive admits failure.
|
||||
if s.Health == SmartFailing {
|
||||
return DiskVerdictFail
|
||||
}
|
||||
// Health == PASSED (or any non-empty non-FAILING value we treat as passing): inspect the counters,
|
||||
// because the overall verdict is structurally unable to report this class of fault.
|
||||
switch {
|
||||
case atLeast(s.TemperatureC, TemperatureFailC): // 3
|
||||
return DiskVerdictFail
|
||||
case positive(s.CriticalWarning): // 4
|
||||
return DiskVerdictFail
|
||||
case atLeast(s.PercentageUsed, percentageUsedFail): // 5
|
||||
return DiskVerdictFail
|
||||
}
|
||||
unreadable := UncorrectableSectors(s)
|
||||
switch {
|
||||
case unreadable > 0 && prior.SawUncorrectable: // 6 — sustained across two consecutive checks
|
||||
return DiskVerdictFail
|
||||
case unreadable > 0 && positive(s.ReallocatedSectors): // 7 — accumulating and remapping together
|
||||
return DiskVerdictFail
|
||||
case unreadable >= uncorrectableFailCount: // 8 — too large to be a blip
|
||||
return DiskVerdictFail
|
||||
case unreadable > 0: // 9 — first sighting, below the bar
|
||||
return DiskVerdictWarn
|
||||
case positive(s.ReallocatedSectors): // 10
|
||||
return DiskVerdictWarn
|
||||
case positive(s.MediaErrors): // 11
|
||||
return DiskVerdictWarn
|
||||
case atLeast(s.PercentageUsed, percentageUsedWarn): // 12
|
||||
return DiskVerdictWarn
|
||||
case atLeast(s.TemperatureC, temperatureWarnC): // 13
|
||||
return DiskVerdictWarn
|
||||
}
|
||||
return DiskVerdictOK // 14
|
||||
}
|
||||
|
||||
// UncorrectableSectors is the disk's unreadable-sector count: max(pending, offline_uncorrectable).
|
||||
// The two attributes track the same physical defect and on the real drive moved in lockstep, so the
|
||||
// larger is the honest figure. 0 when neither is reported (an old agent or a device without them).
|
||||
// Exported because the alert copy quotes this number and the persisted state remembers it.
|
||||
func UncorrectableSectors(s *SmartSummary) int {
|
||||
if s == nil {
|
||||
return 0
|
||||
}
|
||||
n := 0
|
||||
if s.PendingSectors != nil && *s.PendingSectors > n {
|
||||
n = *s.PendingSectors
|
||||
}
|
||||
if s.OfflineUncorrectable != nil && *s.OfflineUncorrectable > n {
|
||||
n = *s.OfflineUncorrectable
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// Label is the exact Hungarian customer copy for the verdict (shared by the card chip and the email).
|
||||
//
|
||||
// There are FOUR labels and there will not be a fifth: a predicted failure is "Hiba", the same word a
|
||||
// self-reported failure gets. A fourth word sharing a root with "Figyelmeztetés" would make the MORE
|
||||
// severe state read as the milder one (settled operator decision, v0.215.0).
|
||||
func (v DiskVerdict) Label() string {
|
||||
switch v {
|
||||
case DiskVerdictOK:
|
||||
return "Rendben"
|
||||
case DiskVerdictWarn:
|
||||
return "Figyelmeztetés"
|
||||
case DiskVerdictFail:
|
||||
return "Hiba"
|
||||
default:
|
||||
return "Nincs adat"
|
||||
}
|
||||
}
|
||||
|
||||
// DegradedAttributes returns the human-readable Hungarian names of the attribute(s) behind a
|
||||
// degraded verdict, for the alert body.
|
||||
//
|
||||
// v0.215.0: this now also names the attributes behind a Hiba REACHED FROM COUNTERS (truth-table rows
|
||||
// 3 and 6-8), not only a Figyelmeztetés — the alert message needs to say what is wrong, and those
|
||||
// rows do have a triggering counter. It returns nil ONLY for row 2 (the drive self-reports FAILING,
|
||||
// a whole-disk verdict with no single triggering counter) and, naturally, for Nincs adat / Rendben.
|
||||
func DegradedAttributes(s *SmartSummary) []string {
|
||||
if s == nil || s.Health == "" || s.Health == SmartUnknown || s.Health == SmartFailing {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
if positive(s.ReallocatedSectors) {
|
||||
out = append(out, "áthelyezett szektorok")
|
||||
}
|
||||
if positive(s.PendingSectors) {
|
||||
out = append(out, "függőben lévő szektorok")
|
||||
}
|
||||
if positive(s.OfflineUncorrectable) {
|
||||
out = append(out, "javíthatatlan szektorok")
|
||||
}
|
||||
if positive(s.CriticalWarning) {
|
||||
out = append(out, "kritikus figyelmeztetés")
|
||||
}
|
||||
if positive(s.MediaErrors) {
|
||||
out = append(out, "adathordozó-hibák")
|
||||
}
|
||||
if atLeast(s.PercentageUsed, percentageUsedWarn) {
|
||||
out = append(out, "elhasználódás")
|
||||
}
|
||||
// Newly able to trigger a verdict on its own (rows 3 and 13), so it must be nameable.
|
||||
if atLeast(s.TemperatureC, temperatureWarnC) {
|
||||
out = append(out, "hőmérséklet")
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func positive(p *int) bool { return p != nil && *p > 0 }
|
||||
func atLeast(p *int, n int) bool { return p != nil && *p >= n }
|
||||
@@ -0,0 +1,202 @@
|
||||
package agentapi
|
||||
|
||||
import "testing"
|
||||
|
||||
// The v0.215.0 severity ladder, verdict half. The event half (emission, damping, cooldown,
|
||||
// persistence) lives in internal/web — this file pins ONLY what the pure function decides.
|
||||
//
|
||||
// Every value used here is taken from the committed evidence:
|
||||
// felhom.eu/documentation/audits/fixtures/smart-ST3000VX010-failing-2026-08-14.json
|
||||
// (ST3000VX010-2E3166, S/N Z6A07P2G, /dev/sdg on DooPlex).
|
||||
|
||||
// realDrive is the failing drive AS CAPTURED on 2026-08-14: PASSED, 352 pending, 352 offline
|
||||
// uncorrectable, 0 reallocated, 40 °C. The whole point of the fixture is that Health is PASSED.
|
||||
func realDrive() *SmartSummary {
|
||||
return &SmartSummary{
|
||||
Health: SmartPassed,
|
||||
PendingSectors: ip(352),
|
||||
OfflineUncorrectable: ip(352),
|
||||
ReallocatedSectors: ip(0),
|
||||
TemperatureC: ip(40),
|
||||
}
|
||||
}
|
||||
|
||||
// Group A (verdict half) — Scenario A. The real drive on its SECOND observation reaches Hiba, and
|
||||
// the chip label is exactly "Hiba".
|
||||
//
|
||||
// Red-proof: delete truth-table row 6 (the `prior.SawUncorrectable` case) from DiskVerdictFor →
|
||||
// the drive still reaches Fail via row 8 (352 >= 64), so this test alone does NOT prove row 6.
|
||||
// TestLadder_SustainIsWhatFires below is the one that isolates it.
|
||||
func TestLadder_RealDrive_ReachesHiba(t *testing.T) {
|
||||
got := DiskVerdictFor(realDrive(), DiskPrior{SawUncorrectable: true})
|
||||
if got != DiskVerdictFail {
|
||||
t.Fatalf("real failing drive verdict = %d (%s), want Fail/Hiba", got, got.Label())
|
||||
}
|
||||
if got.Label() != "Hiba" {
|
||||
t.Errorf("label = %q, want %q", got.Label(), "Hiba")
|
||||
}
|
||||
// The trap this whole change exists for: the drive's own verdict says everything is fine.
|
||||
if realDrive().Health != SmartPassed {
|
||||
t.Fatal("fixture drift: the real drive's Health must be PASSED — that IS the defect")
|
||||
}
|
||||
}
|
||||
|
||||
// Groups B + C (verdict half) — Scenarios B and C. The SAME SmartSummary yields Figyelmeztetés on a
|
||||
// first sighting and Hiba once it is sustained. This is the pair that isolates row 6: the counters
|
||||
// are identical and only `prior` differs, so nothing else in the table can be producing the change.
|
||||
//
|
||||
// The values are the 11 August excursion (8 sectors), which cleared completely within an hour — a
|
||||
// count deliberately far below the 64 backstop so row 8 cannot mask row 6.
|
||||
//
|
||||
// Red-proof: remove the `prior.SawUncorrectable` clause from row 6 → the sustained case stays Warn.
|
||||
func TestLadder_SustainIsWhatFires(t *testing.T) {
|
||||
excursion := func() *SmartSummary {
|
||||
return &SmartSummary{Health: SmartPassed, PendingSectors: ip(8), OfflineUncorrectable: ip(8), ReallocatedSectors: ip(0)}
|
||||
}
|
||||
if got := DiskVerdictFor(excursion(), DiskPrior{}); got != DiskVerdictWarn {
|
||||
t.Errorf("first sighting of 8 sectors = %d (%s), want Warn/Figyelmeztetés — a single "+
|
||||
"excursion that clears by itself is normal and must NOT reach Hiba", got, got.Label())
|
||||
}
|
||||
if got := DiskVerdictFor(excursion(), DiskPrior{SawUncorrectable: true}); got != DiskVerdictFail {
|
||||
t.Errorf("SAME 8 sectors, now sustained = %d (%s), want Fail/Hiba", got, got.Label())
|
||||
}
|
||||
if got := DiskVerdictFor(excursion(), DiskPrior{}).Label(); got != "Figyelmeztetés" {
|
||||
t.Errorf("first-sighting label = %q, want Figyelmeztetés", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Row 8, the backstop — for a box that was powered off or restarted across the sustain window and so
|
||||
// has NO prior. 63 stays Warn, 64 reaches Hiba. The boundary is inclusive, which is what
|
||||
// `uncorrectableFailCount` claims and what the real drive did at 13 Aug 11:28 (exactly 64).
|
||||
//
|
||||
// Red-proof: change `>=` to `>` in row 8 → the "exactly 64" case reads Warn.
|
||||
func TestLadder_CountBackstopBoundary(t *testing.T) {
|
||||
cases := []struct {
|
||||
pending int
|
||||
want DiskVerdict
|
||||
}{
|
||||
{63, DiskVerdictWarn},
|
||||
{64, DiskVerdictFail},
|
||||
{352, DiskVerdictFail},
|
||||
}
|
||||
for _, c := range cases {
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(c.pending)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != c.want {
|
||||
t.Errorf("%d pending sectors, no prior = %d (%s), want %d", c.pending, got, got.Label(), c.want)
|
||||
}
|
||||
}
|
||||
// Row 7 — unreadable AND remapping together is Hiba even at a low count with no prior.
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(8), ReallocatedSectors: ip(1)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != DiskVerdictFail {
|
||||
t.Errorf("row 7 (unreadable + reallocated) = %d, want Fail", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Group I — Scenario I, heat. 61 → Hiba, 56 → Figyelmeztetés, 54 → Rendben, with all counters clean.
|
||||
//
|
||||
// Red-proof: remove rows 3 and 13 → all three read Rendben.
|
||||
func TestLadder_Temperature(t *testing.T) {
|
||||
cases := []struct {
|
||||
temp int
|
||||
want DiskVerdict
|
||||
}{
|
||||
{54, DiskVerdictOK},
|
||||
{55, DiskVerdictWarn}, // inclusive boundary
|
||||
{56, DiskVerdictWarn},
|
||||
{59, DiskVerdictWarn},
|
||||
{60, DiskVerdictFail}, // inclusive boundary
|
||||
{61, DiskVerdictFail},
|
||||
}
|
||||
for _, c := range cases {
|
||||
s := &SmartSummary{Health: SmartPassed, TemperatureC: ip(c.temp), PendingSectors: ip(0), ReallocatedSectors: ip(0)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != c.want {
|
||||
t.Errorf("%d °C = %d (%s), want %d", c.temp, got, got.Label(), c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Group J (verdict half) — Scenario J. No data never alarms, and a prior must not manufacture one:
|
||||
// a nil/UNKNOWN SmartSummary reads Nincs adat EVEN WITH SawUncorrectable set. Row 1 is first in the
|
||||
// table for exactly this reason.
|
||||
//
|
||||
// Red-proof: move row 1 below row 6 → the UNKNOWN-with-prior case reads Hiba, i.e. a disk whose
|
||||
// SMART briefly became unreadable would be reported as failing.
|
||||
func TestLadder_UnknownNeverAlarms(t *testing.T) {
|
||||
for _, s := range []*SmartSummary{nil, {Health: ""}, {Health: SmartUnknown}} {
|
||||
if got := DiskVerdictFor(s, DiskPrior{SawUncorrectable: true}); got != DiskVerdictUnknown {
|
||||
t.Errorf("no-data disk with a prior = %d (%s), want Unknown/Nincs adat", got, got.Label())
|
||||
}
|
||||
}
|
||||
if got := DiskVerdictFor(&SmartSummary{Health: SmartUnknown}, DiskPrior{}).Label(); got != "Nincs adat" {
|
||||
t.Errorf("label = %q, want Nincs adat", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The zero DiskPrior must be the SAFE default: a caller that forgets to load history can only
|
||||
// under-report (Figyelmeztetés), never over-report (Hiba) on a first sighting. This pins the
|
||||
// fail-safe direction the persisted-state loader relies on when its file is missing or corrupt.
|
||||
func TestLadder_ZeroPriorIsFailSafe(t *testing.T) {
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(8)}
|
||||
if got := DiskVerdictFor(s, DiskPrior{}); got != DiskVerdictWarn {
|
||||
t.Fatalf("zero prior must degrade to Warn, not Fail; got %d (%s)", got, got.Label())
|
||||
}
|
||||
}
|
||||
|
||||
// UncorrectableSectors is max(pending, offline) — the number the alert copy quotes and the persisted
|
||||
// state remembers. A wrong answer here puts a wrong count in a customer's email.
|
||||
func TestUncorrectableSectors(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in *SmartSummary
|
||||
want int
|
||||
}{
|
||||
{"nil summary", nil, 0},
|
||||
{"neither reported (old agent)", &SmartSummary{Health: SmartPassed}, 0},
|
||||
{"both zero", &SmartSummary{PendingSectors: ip(0), OfflineUncorrectable: ip(0)}, 0},
|
||||
{"pending only", &SmartSummary{PendingSectors: ip(8)}, 8},
|
||||
{"offline only", &SmartSummary{OfflineUncorrectable: ip(24)}, 24},
|
||||
{"pending larger", &SmartSummary{PendingSectors: ip(40), OfflineUncorrectable: ip(24)}, 40},
|
||||
{"offline larger", &SmartSummary{PendingSectors: ip(24), OfflineUncorrectable: ip(40)}, 40},
|
||||
{"the real drive", realDrive(), 352},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := UncorrectableSectors(c.in); got != c.want {
|
||||
t.Errorf("%s: UncorrectableSectors = %d, want %d", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// DegradedAttributes must NAME the counters behind a Hiba reached from counters (v0.215.0) — the
|
||||
// alert body is built from this and an empty list produces a message that says nothing is wrong.
|
||||
// It still returns nil for row 2 (drive-reported FAILING), which has no single triggering counter.
|
||||
//
|
||||
// Red-proof: restore the pre-v0.215.0 body (nil for anything at Fail) → the real-drive case returns
|
||||
// an empty list.
|
||||
func TestDegradedAttributes_NamesFailCounters(t *testing.T) {
|
||||
got := DegradedAttributes(realDrive())
|
||||
if len(got) == 0 {
|
||||
t.Fatal("a Hiba reached from counters must name its attributes, got none")
|
||||
}
|
||||
found := map[string]bool{}
|
||||
for _, a := range got {
|
||||
found[a] = true
|
||||
}
|
||||
for _, want := range []string{"függőben lévő szektorok", "javíthatatlan szektorok"} {
|
||||
if !found[want] {
|
||||
t.Errorf("missing attribute %q in %v", want, got)
|
||||
}
|
||||
}
|
||||
// Row 2 — the drive self-reports FAILING: no single triggering counter, so nil.
|
||||
if a := DegradedAttributes(&SmartSummary{Health: SmartFailing, PendingSectors: ip(5)}); a != nil {
|
||||
t.Errorf("FAILING (row 2) must return nil attributes, got %v", a)
|
||||
}
|
||||
// Nincs adat must never produce attribute names either.
|
||||
if a := DegradedAttributes(&SmartSummary{Health: SmartUnknown}); a != nil {
|
||||
t.Errorf("UNKNOWN must return nil attributes, got %v", a)
|
||||
}
|
||||
// Temperature is newly able to trigger on its own, so it must be nameable.
|
||||
hot := DegradedAttributes(&SmartSummary{Health: SmartPassed, TemperatureC: ip(61)})
|
||||
if len(hot) != 1 || hot[0] != "hőmérséklet" {
|
||||
t.Errorf("hot disk attributes = %v, want [hőmérséklet]", hot)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
package agentapi
|
||||
|
||||
import "testing"
|
||||
|
||||
func ip(v int) *int { return &v }
|
||||
|
||||
// Verdict table (Part 2, extended v0.215.0). Red-proof: change the PercentageUsed boundary from
|
||||
// `>= 90` to `> 90` in DiskVerdictFor → the "NVMe percentage_used exactly 90 → Figyelmeztetés" case
|
||||
// fails.
|
||||
//
|
||||
// v0.215.0 moved ONE pre-existing case deliberately: critical_warning>0 was Figyelmeztetés and is
|
||||
// now Hiba (truth-table row 4). It is NVMe's own critical flag — a declaration by the device, not a
|
||||
// counter that might drift back — so it belongs with the self-reported failures, not below them.
|
||||
func TestDiskVerdictFor(t *testing.T) {
|
||||
noPrior := DiskPrior{}
|
||||
cases := []struct {
|
||||
name string
|
||||
in *SmartSummary
|
||||
prior DiskPrior
|
||||
want DiskVerdict
|
||||
}{
|
||||
{"nil → unknown", nil, noPrior, DiskVerdictUnknown},
|
||||
{"empty health → unknown", &SmartSummary{Health: ""}, noPrior, DiskVerdictUnknown},
|
||||
{"UNKNOWN → unknown", &SmartSummary{Health: SmartUnknown}, noPrior, DiskVerdictUnknown},
|
||||
{"FAILING → fail", &SmartSummary{Health: SmartFailing}, noPrior, DiskVerdictFail},
|
||||
{"FAILING beats counters", &SmartSummary{Health: SmartFailing, ReallocatedSectors: ip(0)}, noPrior, DiskVerdictFail},
|
||||
{"PASSED clean → ok", &SmartSummary{Health: SmartPassed, ReallocatedSectors: ip(0), PendingSectors: ip(0), TemperatureC: ip(30)}, noPrior, DiskVerdictOK},
|
||||
{"PASSED nil counters → ok", &SmartSummary{Health: SmartPassed}, noPrior, DiskVerdictOK},
|
||||
{"reallocated>0 alone → warn", &SmartSummary{Health: SmartPassed, ReallocatedSectors: ip(1)}, noPrior, DiskVerdictWarn},
|
||||
{"pending>0 first sighting → warn", &SmartSummary{Health: SmartPassed, PendingSectors: ip(5)}, noPrior, DiskVerdictWarn},
|
||||
{"offline_unc>0 first sighting → warn", &SmartSummary{Health: SmartPassed, OfflineUncorrectable: ip(2)}, noPrior, DiskVerdictWarn},
|
||||
{"critical_warning>0 → fail (row 4)", &SmartSummary{Health: SmartPassed, CriticalWarning: ip(1)}, noPrior, DiskVerdictFail},
|
||||
{"media_errors>0 → warn", &SmartSummary{Health: SmartPassed, MediaErrors: ip(3)}, noPrior, DiskVerdictWarn},
|
||||
{"percentage_used 89 → ok", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(89)}, noPrior, DiskVerdictOK},
|
||||
{"percentage_used exactly 90 → warn", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(90)}, noPrior, DiskVerdictWarn},
|
||||
{"percentage_used 95 → warn", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(95)}, noPrior, DiskVerdictWarn},
|
||||
{"percentage_used exactly 100 → fail (row 5)", &SmartSummary{Health: SmartPassed, PercentageUsed: ip(100)}, noPrior, DiskVerdictFail},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := DiskVerdictFor(c.in, c.prior); got != c.want {
|
||||
t.Errorf("%s: DiskVerdictFor = %d, want %d", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiskVerdict_Label(t *testing.T) {
|
||||
want := map[DiskVerdict]string{
|
||||
DiskVerdictUnknown: "Nincs adat",
|
||||
DiskVerdictOK: "Rendben",
|
||||
DiskVerdictWarn: "Figyelmeztetés",
|
||||
DiskVerdictFail: "Hiba",
|
||||
}
|
||||
for v, w := range want {
|
||||
if got := v.Label(); got != w {
|
||||
t.Errorf("verdict %d Label = %q, want %q", v, got, w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A warn lists every triggering attribute at once (Scenario "multiple attributes degrade" → ONE event).
|
||||
func TestDegradedAttributes_ListsAll(t *testing.T) {
|
||||
s := &SmartSummary{Health: SmartPassed, PendingSectors: ip(5), ReallocatedSectors: ip(2), PercentageUsed: ip(91)}
|
||||
got := DegradedAttributes(s)
|
||||
if len(got) != 3 {
|
||||
t.Fatalf("want 3 attributes, got %d: %v", len(got), got)
|
||||
}
|
||||
// clean disk → none
|
||||
if a := DegradedAttributes(&SmartSummary{Health: SmartPassed}); len(a) != 0 {
|
||||
t.Errorf("clean disk should list no attributes, got %v", a)
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package agentapi
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
)
|
||||
@@ -117,3 +118,165 @@ func (c *Client) EscrowCeremonyClaim(ctx context.Context) (string, int, error) {
|
||||
}
|
||||
return out.RecoveryCode, status, nil
|
||||
}
|
||||
|
||||
// RecoverOffsiteRepoPassword asks the agent to open this host's hub-held sealed bundle with the
|
||||
// customer's recovery code and return ONLY the offsite restic repository password, plus its sha256
|
||||
// (R-199, agent >= v0.125.0).
|
||||
//
|
||||
// R CROSSES HERE, AND NOWHERE ELSE IN THIS DIRECTION. It travels in the request body over the pinned
|
||||
// local-API channel (the operator's 2026-08-04 acceptance) and is not retained by this client. The
|
||||
// shared POST helper logs path/status/duration and never bodies — do not add a body log, on either
|
||||
// the request or the response side: the request carries R and the response carries the password.
|
||||
func (c *Client) RecoverOffsiteRepoPassword(ctx context.Context, recoveryCode string) (password, sha256hex string, err error) {
|
||||
env, status, perr := c.postWithStatus(ctx, "/escrow/recover-offsite-password",
|
||||
map[string]string{"recovery_code": recoveryCode})
|
||||
if perr != nil {
|
||||
return "", "", perr
|
||||
}
|
||||
// R-224: this route's refusal keeps its STATUS as a value. `refusalError` flattens status into a
|
||||
// sentence, and a sentence is not something a caller can branch on — which is exactly how a failed
|
||||
// fetch and a wrong recovery code came to produce one customer-facing message.
|
||||
if status < 200 || status > 299 || !env.OK {
|
||||
return "", "", &RecoveryRefusal{Status: status, Reason: truncateErr(env.Error, 300)}
|
||||
}
|
||||
var out struct {
|
||||
ResticRepoPassword string `json:"restic_repo_password"`
|
||||
ResticPwSHA256 string `json:"restic_pw_sha256"`
|
||||
}
|
||||
if uerr := json.Unmarshal(env.Data, &out); uerr != nil {
|
||||
return "", "", fmt.Errorf("agentapi: decode /escrow/recover-offsite-password: %w", uerr)
|
||||
}
|
||||
if out.ResticRepoPassword == "" || out.ResticPwSHA256 == "" {
|
||||
return "", "", fmt.Errorf("agentapi: the agent returned an empty recovery result")
|
||||
}
|
||||
return out.ResticRepoPassword, out.ResticPwSHA256, nil
|
||||
}
|
||||
|
||||
// ── R-224 — CLASSIFYING A FAILED UNLOCK ─────────────────────────────────────────────────────────
|
||||
//
|
||||
// CAMPAIGN-11 measured what happens without this. On 2026-08-05, with a CORRECT current recovery
|
||||
// code: the hub firewalled off returned the customer "this code does not open your package" in
|
||||
// 0.0556 s, and this agent stopped returned the same in 0.0299 s — against ~1.0 s for a genuine
|
||||
// unseal. Neither attempted one. The failure path had exactly two branches, both of them statements
|
||||
// about the customer's code, and `rerr` was never inspected.
|
||||
//
|
||||
// The rule this type exists to enforce: **the customer is blamed only after a real attempt refused
|
||||
// their code.** Everything else — including anything we cannot classify — says something else.
|
||||
|
||||
// RecoveryRefusal is the agent's refusal of an unlock, carrying the STATUS as a value so callers
|
||||
// classify on it rather than on the sentence. The message keeps `refusalError`'s shape so operator
|
||||
// logs read as they did.
|
||||
type RecoveryRefusal struct {
|
||||
Status int
|
||||
Reason string
|
||||
}
|
||||
|
||||
func (e *RecoveryRefusal) Error() string {
|
||||
reason := e.Reason
|
||||
if reason == "" {
|
||||
reason = "(no reason in agent response)"
|
||||
}
|
||||
return fmt.Sprintf("agentapi: POST /escrow/recover-offsite-password: HTTP %d: %s", e.Status, reason)
|
||||
}
|
||||
|
||||
// RecoveryFailure is what went wrong, as far as it can be known.
|
||||
type RecoveryFailure int
|
||||
|
||||
const (
|
||||
// RecoveryUnknown — the cause could not be determined. **The safe default**, and deliberately the
|
||||
// zero value: a new status, a transport shape nobody anticipated, or an agent too old to
|
||||
// distinguish fetch from refusal all land here, and none of them may blame the customer.
|
||||
RecoveryUnknown RecoveryFailure = iota
|
||||
// RecoveryHubUnreachable — the agent answered, and it could not FETCH the sealed package: the hub
|
||||
// refused, was unreachable, or recovery is not configured on this agent. **The code was not used.**
|
||||
RecoveryHubUnreachable
|
||||
// RecoveryAskedAndRefused — the bundle was fetched and the code did not open it. The ONLY class
|
||||
// from which the customer may be told to check their typing.
|
||||
RecoveryAskedAndRefused
|
||||
// RecoveryNoBundle — the hub holds no sealed package for this host at all.
|
||||
RecoveryNoBundle
|
||||
// RecoveryBundleTooOld — the bundle opened but predates the repository-password field.
|
||||
RecoveryBundleTooOld
|
||||
// RecoveryAgentUnreachable — the machine's own in-house service never answered, so there is no
|
||||
// agent verdict at all. **The code was not used.** Distinct from RecoveryHubUnreachable because
|
||||
// it is a different fault, with different words and a different remedy.
|
||||
RecoveryAgentUnreachable
|
||||
// RecoveryCodeOpensRetained — the code was used, it WORKED, and it opened a RETAINED earlier
|
||||
// package rather than the one currently held (R-311, agent >= v0.129.0).
|
||||
//
|
||||
// **The customer is not at fault here and must not be told they might be.** This class exists
|
||||
// because until 2026-08-12 this situation and a mistype were indistinguishable: both fail closed
|
||||
// against the current package, and nothing ever tried the retained ones. The screen said as much
|
||||
// out loud — a true sentence about our own incuriosity that a customer reads as a statement about
|
||||
// their code.
|
||||
RecoveryCodeOpensRetained
|
||||
)
|
||||
|
||||
// ClassifyRecoveryFailure maps an unlock error to its class, from the VALUE and never the text.
|
||||
//
|
||||
// ⚠ `trustRefusal` is the agent-version gate and it is not optional. An agent older than v0.126.0
|
||||
// answers **400 for BOTH** a fetch failure and a wrong code, so a 400 from one cannot be read as
|
||||
// "the code was refused" — it means "one of two things, and we cannot tell which". Pass false there
|
||||
// and the 400 degrades to RecoveryUnknown, which is neutral. That degradation is the point: it is
|
||||
// safe, it is silent, and it heals itself when the agent updates.
|
||||
// ⚠ `trustRetained` is the R-311 twin of `trustRefusal` and is separate on purpose: the two gates
|
||||
// name different agent versions (v0.126.0 and v0.129.0) and a box can sit between them. Passing
|
||||
// `trustRefusal` for both would let a v0.126–128 agent's unexpected 422 be read as a verdict it
|
||||
// cannot produce.
|
||||
func ClassifyRecoveryFailure(err error, trustRefusal, trustRetained bool) RecoveryFailure {
|
||||
if err == nil {
|
||||
return RecoveryUnknown
|
||||
}
|
||||
var ref *RecoveryRefusal
|
||||
if !errors.As(err, &ref) {
|
||||
// Not a refusal at all — the request never produced an agent verdict (dial failure, TLS,
|
||||
// timeout, or the channel could not be built). The machine could not even ASK its own service,
|
||||
// which is a different sentence from "the hub was unreachable" and a different thing to fix.
|
||||
return RecoveryAgentUnreachable
|
||||
}
|
||||
switch ref.Status {
|
||||
case http.StatusBadGateway, http.StatusServiceUnavailable, http.StatusGatewayTimeout:
|
||||
// 502 is agent >= v0.126.0's "the sealed bundle could not be fetched". 503 is its
|
||||
// "recovery is not configured on this agent (no hub client)". Neither used the code.
|
||||
return RecoveryHubUnreachable
|
||||
case http.StatusNotFound:
|
||||
return RecoveryNoBundle
|
||||
case http.StatusConflict:
|
||||
return RecoveryBundleTooOld
|
||||
case http.StatusUnprocessableEntity:
|
||||
// R-311. Gated on the SAME trust flag as 400, and for the mirror-image reason: an agent that
|
||||
// predates the retained lookup cannot emit 422 at all, so a 422 from anywhere else is a shape
|
||||
// we did not design and must not be read as a statement about the customer's code.
|
||||
if trustRetained {
|
||||
return RecoveryCodeOpensRetained
|
||||
}
|
||||
return RecoveryUnknown
|
||||
case http.StatusBadRequest:
|
||||
if trustRefusal {
|
||||
return RecoveryAskedAndRefused
|
||||
}
|
||||
return RecoveryUnknown
|
||||
default:
|
||||
return RecoveryUnknown
|
||||
}
|
||||
}
|
||||
|
||||
// String names the class for the operator log. The customer never sees these words.
|
||||
func (f RecoveryFailure) String() string {
|
||||
switch f {
|
||||
case RecoveryHubUnreachable:
|
||||
return "hub-unreachable"
|
||||
case RecoveryAgentUnreachable:
|
||||
return "agent-unreachable"
|
||||
case RecoveryAskedAndRefused:
|
||||
return "asked-and-refused"
|
||||
case RecoveryNoBundle:
|
||||
return "no-bundle"
|
||||
case RecoveryBundleTooOld:
|
||||
return "bundle-too-old"
|
||||
case RecoveryCodeOpensRetained:
|
||||
return "code-opens-retained"
|
||||
default:
|
||||
return "unknown"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,6 +29,60 @@ const FeatureNetstorageVerify Feature = "netstorage_verify"
|
||||
// (GET/POST /guest/memory) shipped together, so GET /guest/memory IS the capability signal.
|
||||
const FeatureGuestMemoryResize Feature = "guest_memory_resize"
|
||||
|
||||
// FeatureBackupAgeState is R-88 Part 2 (agent v0.105.0): GET /backup/due carries `age_state`,
|
||||
// distinguishing "never backed up" (absent) from "could not tell" (unknown). There is no route
|
||||
// probe for it — the signal is a FIELD on an existing route, so the version floor is the gate and
|
||||
// an empty field means legacy.
|
||||
const FeatureBackupAgeState Feature = "backup_age_state"
|
||||
|
||||
// FeatureOffsiteKeyRecovery is the customer-facing off-site key recovery (agent v0.125.0, R-199
|
||||
// links 7–8): POST /escrow/recover-offsite-password fetches this host's sealed bundle, unseals it
|
||||
// with R and returns the single repository-password field.
|
||||
//
|
||||
// ⚠ THIS GATE FAILS CLOSED, and it is the ONLY feature in this table that does. Read §7.1 of the
|
||||
// R-216 fix before "correcting" it back to the package default.
|
||||
//
|
||||
// The package default is fail-OPEN: SupportUnknown proceeds, because for every other coupled feature
|
||||
// a wrong "unsupported" would block something harmless while a down agent already speaks through the
|
||||
// normal error paths. **That default is what produced R-216.** Measured live on 2026-08-05
|
||||
// (CAMPAIGN-11 Phase 1): an agent 0.120.0 answered the recovery route with 404, the unlock attempt
|
||||
// went ahead anyway, and the customer was told — in Hungarian, on the one screen whose whole purpose
|
||||
// is to be believed about backups — that their perfectly correct recovery code was not accepted and
|
||||
// they should check their typing. A correct code, refused in 0.134 s, blamed on the customer.
|
||||
//
|
||||
// So here: anything other than SupportYes means the screen says THE MACHINE cannot ask yet. The
|
||||
// unlock is never attempted when it cannot complete, because the failure of an attempt that could
|
||||
// never have worked is attributed to the code.
|
||||
const FeatureOffsiteKeyRecovery Feature = "offsite_key_recovery"
|
||||
|
||||
// FeatureRecoveryFailureClass is agent v0.126.0's SPLIT of a failed unlock into distinguishable
|
||||
// statuses (R-224): 502 the sealed bundle could not be FETCHED · 400 it was fetched and the code was
|
||||
// refused · 404 no bundle · 409 the bundle predates the repository-password field.
|
||||
//
|
||||
// ⚠ WHAT THIS GATE ACTUALLY GUARDS is the meaning of **400**, and nothing else. An agent older than
|
||||
// v0.126.0 answers 400 for BOTH a fetch failure and a wrong code — one status, one sentence, two
|
||||
// situations — so on such an agent a 400 cannot be read as "the code was refused". It means "one of
|
||||
// two things and we cannot tell which", which is `RecoveryUnknown`, which is neutral.
|
||||
//
|
||||
// So this gate does not block anything and has no fail-closed behaviour to get wrong: the unlock is
|
||||
// attempted either way (FeatureOffsiteKeyRecovery already decides THAT). It only decides whether the
|
||||
// customer may be told to check their typing. Unknown → they may not. **That is the safe direction,
|
||||
// and it heals itself the moment the agent updates.**
|
||||
const FeatureRecoveryFailureClass Feature = "recovery_failure_class"
|
||||
|
||||
// FeatureRetainedRecoveryClass is agent v0.129.0's FIFTH status on a failed unlock (R-311): 422, the
|
||||
// code is correct and opens a RETAINED earlier package rather than the current one.
|
||||
//
|
||||
// ⚠ WHAT THIS GATE GUARDS is whether the screen may say WHICH of the two causes it is. Before
|
||||
// v0.129.0 nothing ever tried the retained packages, so a correct-but-earlier code and a mistype were
|
||||
// genuinely indistinguishable and the screen said so. That sentence was HONEST then and becomes a
|
||||
// falsehood the moment the agent can tell them apart — so the gate decides which of two true
|
||||
// sentences to print, never whether to attempt the unlock.
|
||||
//
|
||||
// Unknown → the older, hedged sentence. That is the safe direction: it claims less, it was correct
|
||||
// for two months, and it heals itself when the agent updates.
|
||||
const FeatureRetainedRecoveryClass Feature = "retained_recovery_class"
|
||||
|
||||
// SupportState is a probe verdict. The zero value is SupportUnknown (fail-open: unknown never
|
||||
// refuses — the existing agent-error paths speak honestly when the agent is down).
|
||||
type SupportState int
|
||||
@@ -86,12 +140,36 @@ var featureProbes = map[Feature]func(ctx context.Context, p SupportProber) error
|
||||
_, err := gm.GuestMemory(ctx)
|
||||
return err
|
||||
},
|
||||
// The recovery route is a POST that performs work and consumes a recovery code — it cannot be
|
||||
// probed. Like the memory prober's negative case this returns a sentinel that classifies to
|
||||
// SupportUnknown, so the decision falls to the VERSION path above.
|
||||
//
|
||||
// The row must exist even though it cannot probe: SupportsWithSource looks up featureProbes
|
||||
// FIRST and returns "unregistered"/SupportUnknown on a table gap, before the version path runs.
|
||||
// A featureMinAgent row without a featureProbes row is therefore never consulted at all.
|
||||
FeatureOffsiteKeyRecovery: func(ctx context.Context, p SupportProber) error {
|
||||
return errNoRecoveryProbe
|
||||
},
|
||||
// Same POST route, same reason it cannot be probed — the decision falls to the VERSION path.
|
||||
FeatureRecoveryFailureClass: func(ctx context.Context, p SupportProber) error {
|
||||
return errNoRecoveryProbe
|
||||
},
|
||||
// R-311, same route and same reason. The row must exist or SupportsWithSource returns
|
||||
// "unregistered"/SupportUnknown on the table gap and the version row is never consulted.
|
||||
FeatureRetainedRecoveryClass: func(ctx context.Context, p SupportProber) error {
|
||||
return errNoRecoveryProbe
|
||||
},
|
||||
}
|
||||
|
||||
// errNoMemoryProbe classifies to SupportUnknown (not a *StatusError 404), so a prober that cannot be
|
||||
// asked never reads as "unsupported".
|
||||
var errNoMemoryProbe = errors.New("agentapi: prober does not support the guest-memory probe")
|
||||
|
||||
// errNoRecoveryProbe classifies to SupportUnknown: the off-site key recovery route is a POST that
|
||||
// consumes a recovery code and so cannot be probed, leaving the VERSION path to decide. Its caller
|
||||
// fails CLOSED on Unknown — see FeatureOffsiteKeyRecovery.
|
||||
var errNoRecoveryProbe = errors.New("agentapi: the offsite key recovery route cannot be probed")
|
||||
|
||||
// featureMinAgent maps each coupled feature to the MINIMUM agent version that carries its coupled
|
||||
// semantics (the CHANGELOG `MinAgent:` header value). Used by Supports when the agent's version is
|
||||
// KNOWN (the v0.82.0 X-Felhom-Agent-Version channel) — a direct comparison, no probe traffic. A
|
||||
@@ -99,8 +177,28 @@ var errNoMemoryProbe = errors.New("agentapi: prober does not support the guest-m
|
||||
var featureMinAgent = map[Feature]string{
|
||||
FeatureNetstorageVerify: "0.81.0",
|
||||
FeatureGuestMemoryResize: "0.90.0",
|
||||
// R-88 Part 2: /backup/due carries age_state, distinguishing "never backed up" from "cannot tell".
|
||||
FeatureBackupAgeState: "0.105.0",
|
||||
// R-199 links 7–8: POST /escrow/recover-offsite-password. R-216 — this row is the whole reason a
|
||||
// correct recovery code can no longer be reported as wrong on an agent that cannot answer.
|
||||
FeatureOffsiteKeyRecovery: "0.125.0",
|
||||
|
||||
// R-224 — the four-way status split of a failed unlock.
|
||||
FeatureRecoveryFailureClass: "0.126.0",
|
||||
|
||||
// R-311 — the FIFTH status: 422, "your code is correct, it opens an EARLIER package". Before
|
||||
// v0.129.0 the agent never looked at retained packages, so this situation was indistinguishable
|
||||
// from a mistype and arrived as 400. An older agent therefore cannot produce a 422 at all, and the
|
||||
// screen must keep saying it cannot tell the two apart — which was true, and is what this gate
|
||||
// preserves for boxes that have not updated yet.
|
||||
FeatureRetainedRecoveryClass: "0.129.0",
|
||||
}
|
||||
|
||||
// MinAgentFor returns the declared minimum agent version for a feature ("" when the feature has no
|
||||
// row). Read-only accessor over featureMinAgent so a refusal can NAME the version it needs instead of
|
||||
// hard-coding the number a second time at the call site.
|
||||
func MinAgentFor(f Feature) string { return featureMinAgent[f] }
|
||||
|
||||
// AgentVersionReporter is optionally implemented by a SupportProber (*Client is one): it reports
|
||||
// the last strictly-validated agent version seen on its traffic ("" = unknown → probe fallback).
|
||||
type AgentVersionReporter interface {
|
||||
@@ -222,3 +320,9 @@ func classifySupportErr(err error) SupportState {
|
||||
}
|
||||
return SupportUnknown
|
||||
}
|
||||
|
||||
// AgentVersionReporter witness. Asserted at SupportsWithSource (`p.(AgentVersionReporter)`); a failed
|
||||
// assertion falls back from the version gate to the live probe. That degrade is benign — both paths
|
||||
// decide correctly — but *Client is the production prober and losing the version path would silently
|
||||
// turn every MinAgent floor into a probe round-trip, which is a behaviour change nobody would see.
|
||||
var _ AgentVersionReporter = (*Client)(nil)
|
||||
|
||||
@@ -9,8 +9,8 @@ import (
|
||||
)
|
||||
|
||||
func netStub(t *testing.T) (*httptest.Server, string, *struct {
|
||||
addBody AddNetStorageRequest
|
||||
removed string
|
||||
addBody AddNetStorageRequest
|
||||
removed string
|
||||
}) {
|
||||
captured := &struct {
|
||||
addBody AddNetStorageRequest
|
||||
|
||||
@@ -27,6 +27,7 @@ func (p *snapshotsStubProvider) GetStackComposePath(name string) (string, bool)
|
||||
func (p *snapshotsStubProvider) ListDeployedStacks() []backup.StackSummary { return nil }
|
||||
func (p *snapshotsStubProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *snapshotsStubProvider) GetStackHDDPath(string) string { return p.hdd }
|
||||
func (p *snapshotsStubProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
func (p *snapshotsStubProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *snapshotsStubProvider) StopStack(string) error { return nil }
|
||||
func (p *snapshotsStubProvider) StartStack(string) error { return nil }
|
||||
@@ -38,9 +39,10 @@ func (p *snapshotsStubProvider) GetStackClassifiedBinds(string) ([]backup.Classi
|
||||
return nil, false
|
||||
}
|
||||
func (p *snapshotsStubProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *snapshotsStubProvider) RecreateStackFromUnit(string, string, map[string]string) error {
|
||||
func (p *snapshotsStubProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *snapshotsStubProvider) StartStackServices(string, []string) error { return nil }
|
||||
|
||||
// newSnapshotsRouter wires a Router with a real backup.Manager over a tempdir drive.
|
||||
func newSnapshotsRouter(t *testing.T) (*Router, string) {
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// R-166 Part 1.3 — THE CUSTOMER-INTENT POINT.
|
||||
//
|
||||
// `stackMgr` is a concrete *stacks.Manager, so actionStack cannot be driven with a fake without
|
||||
// Docker. The two properties that actually carry the correctness are therefore pinned the only way
|
||||
// they can be: the mapping is a pure function with its own table test, and the ORDER (§8.2) is
|
||||
// asserted structurally over actionStack's AST. Both fail if someone reverses the write and the act,
|
||||
// which is the mistake that would undo a customer's Stop at the next boot.
|
||||
|
||||
func TestDesiredStateForAction_MapsEveryAction(t *testing.T) {
|
||||
cases := []struct {
|
||||
action string
|
||||
want string
|
||||
ok bool
|
||||
}{
|
||||
{"start", stacks.DesiredStateRunning, true},
|
||||
// restart and update both END in `compose up -d`, so a customer who presses either is asking
|
||||
// for the app to be up afterwards.
|
||||
{"restart", stacks.DesiredStateRunning, true},
|
||||
{"update", stacks.DesiredStateRunning, true},
|
||||
{"stop", stacks.DesiredStateStopped, true},
|
||||
// Anything unrecognised records NOTHING rather than guessing — a future action must not
|
||||
// silently acquire an intent it was never meant to carry.
|
||||
{"", "", false},
|
||||
{"delete", "", false},
|
||||
{"pause", "", false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
got, ok := desiredStateForAction(tc.action)
|
||||
if got != tc.want || ok != tc.ok {
|
||||
t.Fatalf("desiredStateForAction(%q) = (%q, %v), want (%q, %v)", tc.action, got, ok, tc.want, tc.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDesiredStateForAction_NeverRecordsStoppedForANonStop(t *testing.T) {
|
||||
// The asymmetry that matters: writing "stopped" for anything other than a Stop would permanently
|
||||
// disable auto-recovery for an app nobody stopped.
|
||||
for _, a := range []string{"start", "restart", "update", "deploy", "delete", ""} {
|
||||
if got, _ := desiredStateForAction(a); got == stacks.DesiredStateStopped {
|
||||
t.Fatalf("action %q maps to desired_state=stopped", a)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestActionStack_RecordsIntentBeforeActing is §8.2, asserted structurally.
|
||||
//
|
||||
// If the SetDesiredState call moved BELOW the action switch, a stop could remove every container
|
||||
// while app.yaml still recorded `running` — and the boot reconciler would then start an app the
|
||||
// customer had just deliberately stopped. That is the single worst outcome available in Part 1, and
|
||||
// no behavioural test in this package can reach it without a Docker daemon.
|
||||
func TestActionStack_RecordsIntentBeforeActing(t *testing.T) {
|
||||
body := funcBody(t, "actionStack")
|
||||
|
||||
setPos, switchPos := -1, -1
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
switch node := n.(type) {
|
||||
case *ast.CallExpr:
|
||||
if sel, ok := node.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "SetDesiredState" && setPos < 0 {
|
||||
setPos = int(node.Pos())
|
||||
}
|
||||
case *ast.SwitchStmt:
|
||||
// The action switch is the one whose tag is the `action` identifier.
|
||||
if id, ok := node.Tag.(*ast.Ident); ok && id.Name == "action" && switchPos < 0 {
|
||||
switchPos = int(node.Pos())
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if setPos < 0 {
|
||||
t.Fatal("actionStack no longer calls SetDesiredState — the customer's start/stop decision is " +
|
||||
"recorded nowhere, which is the R-166 defect un-fixed")
|
||||
}
|
||||
if switchPos < 0 {
|
||||
t.Fatal("actionStack no longer has a `switch action` — this test needs updating")
|
||||
}
|
||||
if setPos >= switchPos {
|
||||
t.Fatal("actionStack records the desired state AFTER performing the action (§8.2 violated): a " +
|
||||
"stop whose intent write fails or lands late leaves zero containers with `running` " +
|
||||
"recorded, and the boot reconciler would restart an app the customer just stopped")
|
||||
}
|
||||
}
|
||||
|
||||
// TestActionStack_RefusesTheActionWhenIntentCannotBeRecorded pins the other half of §8.2: a failed
|
||||
// write REFUSES the act. Proceeding anyway would perform a stop that nothing records — exactly the
|
||||
// ambiguity this release removes.
|
||||
func TestActionStack_RefusesTheActionWhenIntentCannotBeRecorded(t *testing.T) {
|
||||
body := funcBody(t, "actionStack")
|
||||
|
||||
refuses := false
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
ifst, ok := n.(*ast.IfStmt)
|
||||
if !ok || ifst.Init == nil {
|
||||
return true
|
||||
}
|
||||
// Look for `if derr := ...SetDesiredState(...); derr != nil { ... return }`
|
||||
assign, ok := ifst.Init.(*ast.AssignStmt)
|
||||
if !ok || len(assign.Rhs) != 1 {
|
||||
return true
|
||||
}
|
||||
call, ok := assign.Rhs[0].(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok || sel.Sel.Name != "SetDesiredState" {
|
||||
return true
|
||||
}
|
||||
for _, stmt := range ifst.Body.List {
|
||||
if _, isReturn := stmt.(*ast.ReturnStmt); isReturn {
|
||||
refuses = true
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if !refuses {
|
||||
t.Fatal("actionStack does not RETURN when SetDesiredState fails — it would go on to stop or " +
|
||||
"start an app whose intent could not be recorded (§8.2)")
|
||||
}
|
||||
}
|
||||
|
||||
// funcBody parses router.go and returns the named method's body.
|
||||
func funcBody(t *testing.T, name string) *ast.BlockStmt {
|
||||
t.Helper()
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("parse router.go: %v", err)
|
||||
}
|
||||
for _, decl := range f.Decls {
|
||||
if fn, ok := decl.(*ast.FuncDecl); ok && fn.Name.Name == name && fn.Body != nil {
|
||||
return fn.Body
|
||||
}
|
||||
}
|
||||
t.Fatalf("func %s not found in router.go", name)
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestDeployStackWiresTheLifecycleGate — the seam-discipline test (§9 rule 6).
|
||||
//
|
||||
// The lifecycle predicate is unit-tested in internal/stacks, and a test there passes whether or not
|
||||
// deployStack ever calls it. Three inert-seam defects shipped fully-green in three days (controller
|
||||
// v0.154.0, agent v0.91.0, agent v0.92.0's missing sudoers grant), all this exact shape: correct
|
||||
// component, absent caller. So the CALLER is asserted here, from source.
|
||||
//
|
||||
// It walks the AST rather than doing strings.Contains on the file, because a commented-out call
|
||||
// still contains the string — the lesson recorded in PROMPT-TEMPLATE §10.
|
||||
//
|
||||
// It also asserts ORDER: the gate must precede the DeployStack call, or it is not fail-closed.
|
||||
func TestDeployStackWiresTheLifecycleGate(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0) // comments dropped: a commented call is not a call
|
||||
if err != nil {
|
||||
t.Fatalf("parse router.go: %v", err)
|
||||
}
|
||||
|
||||
var fn *ast.FuncDecl
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
if d, ok := n.(*ast.FuncDecl); ok && d.Name.Name == "deployStack" {
|
||||
fn = d
|
||||
return false
|
||||
}
|
||||
return true
|
||||
})
|
||||
if fn == nil {
|
||||
t.Fatal("deployStack not found in router.go — did it move? the gate's wiring is now unasserted")
|
||||
}
|
||||
|
||||
canInstallPos, deployPos := -1, -1
|
||||
ast.Inspect(fn, func(n ast.Node) bool {
|
||||
call, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
sel, ok := call.Fun.(*ast.SelectorExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
off := fset.Position(call.Pos()).Offset
|
||||
switch sel.Sel.Name {
|
||||
case "CanInstall":
|
||||
if canInstallPos == -1 {
|
||||
canInstallPos = off
|
||||
}
|
||||
case "DeployStack":
|
||||
if deployPos == -1 {
|
||||
deployPos = off
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if canInstallPos == -1 {
|
||||
t.Fatal("deployStack never calls Meta.CanInstall() — the lifecycle gate is INERT: " +
|
||||
"a withdrawn app is hidden from the catalog page but still installable by direct POST")
|
||||
}
|
||||
if deployPos == -1 {
|
||||
t.Fatal("deployStack no longer calls DeployStack — this test's ordering assertion is meaningless")
|
||||
}
|
||||
if canInstallPos > deployPos {
|
||||
t.Fatalf("the lifecycle gate (offset %d) runs AFTER DeployStack (offset %d) — a gate that "+
|
||||
"fires after the mutation is not fail-closed", canInstallPos, deployPos)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLifecycleRefusalMessageIsCustomerFacingHungarian: the refusal text reaches the customer via
|
||||
// showAlert(), so it must be the sentence the spec ruled, not a Go error string.
|
||||
func TestLifecycleRefusalMessageIsCustomerFacingHungarian(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
const want = "Ez az alkalmazás jelenleg nem telepíthető."
|
||||
found := false
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
if lit, ok := n.(*ast.BasicLit); ok && lit.Kind == token.STRING && strings.Contains(lit.Value, want) {
|
||||
found = true
|
||||
}
|
||||
return true
|
||||
})
|
||||
if !found {
|
||||
t.Fatalf("the ruled refusal message %q is not present in router.go", want)
|
||||
}
|
||||
}
|
||||
@@ -410,6 +410,21 @@ func (r *Router) deployStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
return
|
||||
}
|
||||
|
||||
// Lifecycle gate: an app withdrawn from the catalog (`lifecycle: hidden` / `abandoned`) is not
|
||||
// installable. FAIL-CLOSED and server-side on purpose — the catalog page already omits these, so
|
||||
// anything reaching here is a stale link, a bookmarked deploy form, or a direct POST, and a gate
|
||||
// that only hides the button is not a gate. Deliberately BEFORE every mutation.
|
||||
//
|
||||
// This does NOT touch an already-deployed instance: it is on the deploy path only, and the
|
||||
// manager refuses a redeploy of an existing stack through its own "already deployed" check.
|
||||
if st, ok := r.stackMgr.GetStack(name); ok && !st.Meta.CanInstall() {
|
||||
r.logger.Printf("[WARN] [api] Deploy refused for %s: lifecycle=%s (not offered for new installs)",
|
||||
name, st.Meta.EffectiveLifecycle())
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false,
|
||||
Error: "Ez az alkalmazás jelenleg nem telepíthető."})
|
||||
return
|
||||
}
|
||||
|
||||
// Prevention layer (storage-split): refuse a deploy when the Docker-data volume is at/under its
|
||||
// reserved buffer, so customer apps can't fill the volume the infra containers (controller,
|
||||
// traefik, cloudflared, filebrowser) depend on. Fail-OPEN on a measurement error — the buffer is
|
||||
@@ -433,6 +448,20 @@ func (r *Router) deployStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
return
|
||||
}
|
||||
|
||||
// R-108: an app's data namespace may NOT live on network storage — its backups would land at
|
||||
// `<share>/backups/primary/<stack>/`, inside the share-ROOT bind FileBrowser serves with
|
||||
// download:true (and that bind cannot be narrowed — see settings.RefuseAsAppNamespace).
|
||||
//
|
||||
// THIS is the boundary, not the deploy dropdown. The dropdown is a UI list; this endpoint accepts
|
||||
// whatever HDD_PATH a caller supplies and `DeployStack` validates only that it EXISTS on the
|
||||
// filesystem (os.Stat, internal/stacks/deploy.go). A filter on the list alone would have left the
|
||||
// surface wide open — the R-108 row's "no IsNetwork() filter on the dropdown" understates it.
|
||||
if refuse, why := r.sett.RefuseAsAppNamespace(body.Values["HDD_PATH"]); refuse {
|
||||
r.logger.Printf("[WARN] [api] Deploy refused for %s: HDD_PATH is not usable as an app namespace (R-108)", name)
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false, Error: why})
|
||||
return
|
||||
}
|
||||
|
||||
deployReq := stacks.DeployRequest{
|
||||
StackName: name,
|
||||
Values: body.Values,
|
||||
@@ -505,6 +534,24 @@ func (r *Router) startGatedByMissingDrive(name string) (bool, string) {
|
||||
return false, ""
|
||||
}
|
||||
|
||||
// desiredStateForAction maps a stack action to the customer intent it expresses, or (_, false) for
|
||||
// an action that expresses none. Pure, so the §8.1/§1.3 mapping is testable without a Manager.
|
||||
//
|
||||
// `restart` and `update` both mean running: a customer who updates or restarts an app is asking for
|
||||
// it to be up afterwards, and both end in `compose up -d`. Anything not listed here — an unknown
|
||||
// action string — records nothing rather than guessing, so a future action cannot silently acquire
|
||||
// an intent it was never meant to carry.
|
||||
func desiredStateForAction(action string) (string, bool) {
|
||||
switch action {
|
||||
case "start", "restart", "update":
|
||||
return stacks.DesiredStateRunning, true
|
||||
case "stop":
|
||||
return stacks.DesiredStateStopped, true
|
||||
default:
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
|
||||
func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
r.logger.Printf("[INFO] [api] %s requested for stack: %s", action, name)
|
||||
r.dbg("actionStack: action=%s name=%s", action, name)
|
||||
@@ -550,6 +597,26 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
}
|
||||
}
|
||||
|
||||
// R-166: THE CUSTOMER-INTENT POINT. This switch is where a human's decision about whether their
|
||||
// app should be running enters the system, and until v0.189.0 that decision was recorded nowhere
|
||||
// — so the box had to infer it from container counts, and inferred wrong for a power cut and for
|
||||
// an interrupted backup alike.
|
||||
//
|
||||
// Written BEFORE the act (§8.2) and a failed write REFUSES the act: performing a stop whose
|
||||
// intent could not be recorded would recreate exactly the ambiguity this closes. Both gates that
|
||||
// can legitimately refuse an action (protected-stack, drive-absent, memory) have already run
|
||||
// above, so nothing is recorded for an action that was never going to happen.
|
||||
if desired, ok := desiredStateForAction(action); ok {
|
||||
if derr := r.stackMgr.SetDesiredState(name, desired); derr != nil {
|
||||
r.logger.Printf("[ERROR] [api] %s for %s refused: could not record desired state: %v", action, name, derr)
|
||||
writeJSON(w, http.StatusInternalServerError, apiResponse{
|
||||
OK: false,
|
||||
Error: "A művelet nem hajtható végre: az alkalmazás beállításai nem menthetők.",
|
||||
})
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
var err error
|
||||
switch action {
|
||||
case "start":
|
||||
@@ -943,6 +1010,7 @@ func (r *Router) triggerBackup(w http.ResponseWriter, _ *http.Request) {
|
||||
}
|
||||
|
||||
r.logger.Println("[INFO] [api] Manual app-data backup (DB dump) triggered")
|
||||
r.backupMgr.MarkManualRun() // R-182: operator-triggered — its digest must not be collapsed into the nightly one
|
||||
go r.backupMgr.RunDBDumps(context.Background())
|
||||
|
||||
writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Mentés elindítva"})
|
||||
|
||||
@@ -19,15 +19,19 @@ type StackDataProvider interface {
|
||||
GetStackComposePath(name string) (composePath string, ok bool)
|
||||
ListDeployedStacks() []StackSummary
|
||||
GetStackHDDMounts(name string) []string
|
||||
GetStackHDDPath(name string) string // raw HDD_PATH from app.yaml (empty if no HDD)
|
||||
GetStackHDDPath(name string) string // raw HDD_PATH from app.yaml (empty if no HDD)
|
||||
// GetImportRoot returns the CANONICAL drop-zone root (R-75): <system namespace root>/userdata/import.
|
||||
// It is app-INDEPENDENT and lives on the SYSTEM drive, so ${IMPORT_PATH} binds cannot be resolved
|
||||
// from GetStackHDDPath. Empty when unresolvable — structuralGuard refuses such binds loudly.
|
||||
GetImportRoot() string
|
||||
GetDockerVolumes(name string) []string // full Docker volume names (project-prefixed)
|
||||
StopStack(name string) error
|
||||
StartStack(name string) error
|
||||
RefreshAndIsRunning(name string) bool
|
||||
// GetStackRecoveryInfo returns the data needed to capture a SECRET-FREE recovery unit
|
||||
// (Phase 2): the stack dir, pinned image tags, the non-secret env, and the NAMES of the
|
||||
// secret/data-key env vars (values are NEVER returned — they are recovered at restore time
|
||||
// from the guest's own app.yaml, live or via the PBS whole-guest snapshot). ok=false if the
|
||||
// GetStackRecoveryInfo returns the data needed to capture a recovery unit: the stack dir,
|
||||
// pinned image tags, the non-secret env, the NAMES of the secret/data-key env vars, and (D5)
|
||||
// the decrypted VALUES of the portable class. A WITHHELD secret's value is never returned —
|
||||
// it is recovered at restore time from the guest's app.yaml, or regenerated. ok=false if the
|
||||
// stack is unknown.
|
||||
GetStackRecoveryInfo(name string) (RecoveryInfo, bool)
|
||||
|
||||
@@ -39,10 +43,18 @@ type StackDataProvider interface {
|
||||
// fail-closed gate decides what to do. The unit is never the source of secrets.
|
||||
RecoverStackSecrets(name string, names []string) map[string]string
|
||||
|
||||
// RecreateStackFromUnit restores an app's definition from the unit's compose dir into the stack
|
||||
// dir, writes app.yaml from fullEnv (encrypting secret fields), and (re-)deploys it via
|
||||
// `docker compose up -d`, which re-pulls the pinned image. Secrets are NEVER regenerated.
|
||||
RecreateStackFromUnit(name, composeSrcDir string, fullEnv map[string]string) error
|
||||
// RecreateStackDefinitionFromUnit restores an app's DEFINITION from the unit's compose dir into
|
||||
// the stack dir and writes app.yaml from fullEnv (encrypting secret fields). Secrets are NEVER
|
||||
// regenerated. It starts NOTHING: the caller owns the bring-up order, because a DB-bearing app
|
||||
// must have its database service started alone for the dump replay (R-47). It was
|
||||
// `RecreateStackFromUnit` until v0.153.0 and ended in a full `docker compose up -d` — that full
|
||||
// start before the replay IS the H4 race.
|
||||
RecreateStackDefinitionFromUnit(name, composeSrcDir string, fullEnv map[string]string) error
|
||||
|
||||
// StartStackServices brings up ONLY the named compose services, leaving the rest of the stack
|
||||
// down — the DB-only window in which a dump is replayed without the application racing it.
|
||||
// Implementations must REFUSE an empty list (an argument-less `up -d` is a full start).
|
||||
StartStackServices(name string, services []string) error
|
||||
|
||||
// GetStackClassifiedBinds returns the app's backup-classified compose binds + whether it carries a
|
||||
// (valid) backup block (Task 2, referential coupling). INERT — no tier consumes it yet; wired now
|
||||
@@ -50,17 +62,27 @@ type StackDataProvider interface {
|
||||
GetStackClassifiedBinds(name string) ([]ClassifiedBind, bool)
|
||||
}
|
||||
|
||||
// RecoveryInfo carries everything needed to write a secret-free recovery unit for a stack.
|
||||
// It deliberately holds NO secret values — only the names of secret/data-key env vars, so the
|
||||
// manifest can record what must be recovered from elsewhere (guest app.yaml / PBS) without the
|
||||
// unit ever storing a secret or a data-encrypting key.
|
||||
// RecoveryInfo carries everything needed to write a recovery unit for a stack.
|
||||
//
|
||||
// D5: it now carries the VALUES of the PORTABLE secret class (stacks.PortableSecretEnvVars — every
|
||||
// `type: secret` field bar the nonPortableSecrets register), because a Tier-1/2 restore that depends
|
||||
// on the guest for a data-encrypting key or a DB password is not independent of the guest at all: the
|
||||
// data sits safely on the drive and cannot be read back. The EXCLUDED class (`type: password` admin
|
||||
// logins) is still name-only and never leaves the guest.
|
||||
type RecoveryInfo struct {
|
||||
StackDir string // dir holding docker-compose.yml + .felhom.yml + app.yaml
|
||||
DisplayName string // app display name
|
||||
ImagePins []string // pinned image tags from compose `image:` lines (re-pulled on restore)
|
||||
NonSecretEnv map[string]string // env with all secret/password/data-key values removed (plaintext only)
|
||||
SecretEnvVars []string // NAMES of stripped secret/password fields (recovered from guest/PBS)
|
||||
NonSecretEnv map[string]string // env with ALL secret/password values removed (plaintext only)
|
||||
SecretEnvVars []string // NAMES of every secret/password field
|
||||
DataKeyEnvVars []string // NAMES of data-encrypting-key fields (fail-closed gate on restore)
|
||||
// PortableSecretEnvVars are the NAMES of the secrets that travel in the unit (D5), and
|
||||
// PortableSecrets their DECRYPTED values. A name present here but absent from PortableSecrets was
|
||||
// unset/empty in the guest's app.yaml — the restore's fail-closed gate decides what that means.
|
||||
// Never logged, never in the manifest's value space: the values reach disk only inside the unit's
|
||||
// 0600 app.yaml.
|
||||
PortableSecretEnvVars []string
|
||||
PortableSecrets map[string]string
|
||||
}
|
||||
|
||||
// ParseComposeImages extracts the pinned image references (`image: repo:tag`) from a
|
||||
|
||||
@@ -59,6 +59,8 @@ const (
|
||||
reasonEscape = "path escapes the drive root"
|
||||
reasonBareRoot = "bare drive-root bind would capture the backups tree"
|
||||
reasonReserved = "path inside the reserved backups zone"
|
||||
// reasonNoImportRoot: a ${IMPORT_PATH} bind with no resolvable system namespace root (R-75).
|
||||
reasonNoImportRoot = "canonical import root unresolvable (system_data_path unconfigured)"
|
||||
)
|
||||
|
||||
// ComputeCaptureSet resolves an app's classified binds into the tier-filtered absolute capture set.
|
||||
@@ -71,16 +73,17 @@ const (
|
||||
// so the engines' no-block branch stays byte-identical to today (the SQ5 cost-regression guard).
|
||||
//
|
||||
// Resolution: RootHDD → path.Join(hddPath, relPath); RootUserdata → path.Join(hddPath, "userdata",
|
||||
// relPath). Guards run AFTER the tier filter, so Skipped means exactly "would have been captured by
|
||||
// this tier, refused for structural safety".
|
||||
func ComputeCaptureSet(binds []ClassifiedBind, hasClassification bool, tier CaptureTier, hddPath string) CaptureSet {
|
||||
// relPath); RootImport → path.Join(importRoot, relPath) — the SYSTEM drive, never hddPath (R-75).
|
||||
// Guards run AFTER the tier filter, so Skipped means exactly "would have been captured by this tier,
|
||||
// refused for structural safety".
|
||||
func ComputeCaptureSet(binds []ClassifiedBind, hasClassification bool, tier CaptureTier, hddPath, importRoot string) CaptureSet {
|
||||
if !hasClassification {
|
||||
return CaptureSet{HasClassification: false}
|
||||
}
|
||||
cs := CaptureSet{HasClassification: true}
|
||||
|
||||
// Stages 1–3: tier filter → structural guards → equal-Abs collapse (shared with ComputeFabBuckets).
|
||||
uniq, skipped := resolveGuardCollapse(binds, hddPath, func(c BindClass) bool { return tierKeeps(tier, c) })
|
||||
uniq, skipped := resolveGuardCollapse(binds, hddPath, importRoot, func(c BindClass) bool { return tierKeeps(tier, c) })
|
||||
cs.Skipped = skipped
|
||||
|
||||
// Stage 4: containment dedup — drop any path whose ancestor is already present (keep the ancestor).
|
||||
@@ -107,18 +110,18 @@ func ComputeCaptureSet(binds []ClassifiedBind, hasClassification bool, tier Capt
|
||||
// collapse (mandatory > optional > excluded; ties by smaller Root/RelPath). It does NOT apply
|
||||
// containment dedup — the caller decides (ComputeCaptureSet does; ComputeFabBuckets must not, so a
|
||||
// mandatory child inside an excluded parent stays independently addressable).
|
||||
func resolveGuardCollapse(binds []ClassifiedBind, hddPath string, keep func(BindClass) bool) (uniq []CapturePath, skipped []SkippedPath) {
|
||||
func resolveGuardCollapse(binds []ClassifiedBind, hddPath, importRoot string, keep func(BindClass) bool) (uniq []CapturePath, skipped []SkippedPath) {
|
||||
var resolved []CapturePath
|
||||
for _, b := range binds {
|
||||
if !keep(b.Class) {
|
||||
continue
|
||||
}
|
||||
if reason, bad := structuralGuard(b.Root, b.RelPath); bad {
|
||||
if reason, bad := structuralGuard(b.Root, b.RelPath, importRoot); bad {
|
||||
skipped = append(skipped, SkippedPath{Root: b.Root, RelPath: b.RelPath, Class: b.Class, Reason: reason})
|
||||
continue
|
||||
}
|
||||
resolved = append(resolved, CapturePath{
|
||||
Abs: resolveAbs(hddPath, b.Root, b.RelPath), Root: b.Root, RelPath: b.RelPath, Class: b.Class,
|
||||
Abs: resolveAbs(hddPath, importRoot, b.Root, b.RelPath), Root: b.Root, RelPath: b.RelPath, Class: b.Class,
|
||||
})
|
||||
}
|
||||
byAbs := make(map[string]CapturePath, len(resolved))
|
||||
@@ -153,12 +156,12 @@ type FabBuckets struct {
|
||||
// selection UI + plan. Same resolution + structural guards + equal-Abs collapse as ComputeCaptureSet
|
||||
// (via resolveGuardCollapse), bucketed by class, no cross-bucket containment dedup. Each bucket is
|
||||
// Abs-sorted (deterministic).
|
||||
func ComputeFabBuckets(binds []ClassifiedBind, hasClassification bool, hddPath string) FabBuckets {
|
||||
func ComputeFabBuckets(binds []ClassifiedBind, hasClassification bool, hddPath, importRoot string) FabBuckets {
|
||||
if !hasClassification {
|
||||
return FabBuckets{HasClassification: false}
|
||||
}
|
||||
fb := FabBuckets{HasClassification: true}
|
||||
uniq, skipped := resolveGuardCollapse(binds, hddPath, func(BindClass) bool { return true })
|
||||
uniq, skipped := resolveGuardCollapse(binds, hddPath, importRoot, func(BindClass) bool { return true })
|
||||
fb.Skipped = skipped
|
||||
for _, cp := range uniq {
|
||||
switch cp.Class {
|
||||
@@ -201,10 +204,19 @@ func tierKeeps(tier CaptureTier, class BindClass) bool {
|
||||
// there (ParseComposeClassifiableBinds path.Cleans; ValidateBackupSpec vets only SPEC entries), so an
|
||||
// unlisted writable "${HDD_PATH}/../x" bind reaches here classed mandatory — this guard is
|
||||
// load-bearing security, not defence-in-depth.
|
||||
func structuralGuard(root BindRoot, relPath string) (reason string, bad bool) {
|
||||
func structuralGuard(root BindRoot, relPath, importRoot string) (reason string, bad bool) {
|
||||
if relPathEscapes(relPath) {
|
||||
return reasonEscape, true
|
||||
}
|
||||
// RootImport (R-75) resolves against the SYSTEM drive, not hddPath. If that root is unresolvable
|
||||
// (system_data_path unconfigured) the bind cannot be placed at all — refuse it LOUDLY into Skipped
|
||||
// rather than let resolveAbs join onto "" and produce a relative, wrong-drive path. The other two
|
||||
// roots cannot hit this: hddPath is checked by their own callers.
|
||||
if root == RootImport && importRoot == "" {
|
||||
return reasonNoImportRoot, true
|
||||
}
|
||||
// A bare ${IMPORT_PATH} bind is allowed: it resolves to <sysNS>/userdata/import, which nests no
|
||||
// backups/ tree (backups live at <sysNS>/backups, a sibling of userdata).
|
||||
if root == RootHDD {
|
||||
if relPath == "" {
|
||||
return reasonBareRoot, true // bare ${HDD_PATH} would nest <hddPath>/backups into the capture
|
||||
@@ -233,11 +245,21 @@ func relPathEscapes(relPath string) bool {
|
||||
}
|
||||
|
||||
// resolveAbs maps a guarded (root, relPath) to its in-container absolute path via slash algebra.
|
||||
func resolveAbs(hddPath string, root BindRoot, relPath string) string {
|
||||
if root == RootUserdata {
|
||||
//
|
||||
// RootImport is the one root that does NOT resolve against hddPath: the canonical drop-zone lives on
|
||||
// the SYSTEM drive (R-75), so importRoot is supplied separately by the caller. Resolving it against
|
||||
// hddPath would silently name a directory on the WRONG DRIVE — a .fab opt-in would then capture (or
|
||||
// on restore, write) somewhere that merely looks plausible. An empty importRoot is the unresolvable
|
||||
// case and is refused upstream by structuralGuard, never silently joined.
|
||||
func resolveAbs(hddPath, importRoot string, root BindRoot, relPath string) string {
|
||||
switch root {
|
||||
case RootUserdata:
|
||||
return path.Join(hddPath, "userdata", relPath)
|
||||
case RootImport:
|
||||
return path.Join(importRoot, relPath)
|
||||
default:
|
||||
return path.Join(hddPath, relPath)
|
||||
}
|
||||
return path.Join(hddPath, relPath)
|
||||
}
|
||||
|
||||
// strongerCapture picks the winner of an equal-Abs collision: mandatory beats optional; on equal
|
||||
|
||||
@@ -38,12 +38,12 @@ func TestComputeCaptureSet_PerTierSplit(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/photos", ReadOnly: true}, Class: ClassOptional, Origin: OriginExplicit},
|
||||
}
|
||||
|
||||
off := ComputeCaptureSet(binds, true, TierOffsite, drv)
|
||||
off := ComputeCaptureSet(binds, true, TierOffsite, drv, "")
|
||||
if got, want := absList(off), []string{hdd("appdata/immich")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("offsite Paths = %v, want %v (mandatory only — the :ro optional must NOT ship offsite)", got, want)
|
||||
}
|
||||
|
||||
sec := ComputeCaptureSet(binds, true, TierSecondary, drv)
|
||||
sec := ComputeCaptureSet(binds, true, TierSecondary, drv, "")
|
||||
want := []string{hdd("appdata/immich"), udat("media/photos")}
|
||||
if got := absList(sec); !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("secondary Paths = %v, want %v (sorted)", got, want)
|
||||
@@ -72,7 +72,7 @@ func TestComputeCaptureSet_LegacyInert(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/sonarr"}, Origin: OriginLegacy},
|
||||
}
|
||||
for _, tier := range []CaptureTier{TierOffsite, TierSecondary} {
|
||||
cs := ComputeCaptureSet(binds, false, tier, drv)
|
||||
cs := ComputeCaptureSet(binds, false, tier, drv, "")
|
||||
if cs.HasClassification {
|
||||
t.Errorf("%s: HasClassification=true for a legacy app", tier)
|
||||
}
|
||||
@@ -94,7 +94,7 @@ func TestComputeCaptureSet_ExcludedInvisible(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "import/paperless"}, Class: ClassExcluded, Origin: OriginExplicit},
|
||||
}
|
||||
for _, tier := range []CaptureTier{TierOffsite, TierSecondary} {
|
||||
cs := ComputeCaptureSet(binds, true, tier, drv)
|
||||
cs := ComputeCaptureSet(binds, true, tier, drv, "")
|
||||
if got, want := absList(cs), []string{hdd("appdata/paperless/media")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("%s Paths = %v, want %v (excluded filtered)", tier, got, want)
|
||||
}
|
||||
@@ -108,12 +108,12 @@ func TestComputeCaptureSet_ExcludedInvisible(t *testing.T) {
|
||||
|
||||
func TestComputeCaptureSet_StructuralGuards(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "../evil"}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d1 traversal
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: ""}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d2 bare hdd root
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "../evil"}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d1 traversal
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: ""}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d2 bare hdd root
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "backups/primary/x"}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d3 reserved zone
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: ""}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d4 bare userdata — ALLOWED
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: ""}, Class: ClassMandatory, Origin: OriginDefaultWritable}, // d4 bare userdata — ALLOWED
|
||||
}
|
||||
cs := ComputeCaptureSet(binds, true, TierOffsite, drv)
|
||||
cs := ComputeCaptureSet(binds, true, TierOffsite, drv, "")
|
||||
|
||||
// Paths: ONLY d4's userdata root — no escaped root, no backups/ anywhere.
|
||||
if got, want := absList(cs), []string{udat("")}; !reflect.DeepEqual(got, want) {
|
||||
@@ -156,7 +156,7 @@ func TestComputeCaptureSet_LegitDotDotName(t *testing.T) {
|
||||
binds := []ClassifiedBind{
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/a..b"}, Class: ClassMandatory, Origin: OriginExplicit},
|
||||
}
|
||||
cs := ComputeCaptureSet(binds, true, TierOffsite, drv)
|
||||
cs := ComputeCaptureSet(binds, true, TierOffsite, drv, "")
|
||||
if got, want := absList(cs), []string{hdd("appdata/a..b")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("Paths = %v, want %v (a..b is a legit name, not traversal)", got, want)
|
||||
}
|
||||
@@ -174,7 +174,7 @@ func TestComputeCaptureSet_ContainmentAndCollision(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "userdata/media"}, Class: ClassOptional, Origin: OriginExplicit}, // Abs collides with next
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media"}, Class: ClassMandatory, Origin: OriginExplicit}, // same Abs, mandatory
|
||||
}
|
||||
cs := ComputeCaptureSet(binds, true, TierSecondary, drv)
|
||||
cs := ComputeCaptureSet(binds, true, TierSecondary, drv, "")
|
||||
|
||||
want := []string{hdd("appdata/paperless"), udat("media")}
|
||||
if got := absList(cs); !reflect.DeepEqual(got, want) {
|
||||
@@ -186,7 +186,7 @@ func TestComputeCaptureSet_ContainmentAndCollision(t *testing.T) {
|
||||
}
|
||||
|
||||
// determinism: recompute and compare full struct
|
||||
cs2 := ComputeCaptureSet(binds, true, TierSecondary, drv)
|
||||
cs2 := ComputeCaptureSet(binds, true, TierSecondary, drv, "")
|
||||
if !reflect.DeepEqual(cs, cs2) {
|
||||
t.Error("ComputeCaptureSet is non-deterministic across runs")
|
||||
}
|
||||
|
||||
@@ -34,12 +34,23 @@ type BindRoot string
|
||||
const (
|
||||
RootUserdata BindRoot = "userdata" // relative to ${USERDATA_PATH}
|
||||
RootHDD BindRoot = "hdd" // relative to ${HDD_PATH}
|
||||
// RootImport is relative to ${IMPORT_PATH} — the CANONICAL drop-zone root (R-75). Unlike the
|
||||
// other two it does NOT resolve against the app's own drive: it lives on the system drive's
|
||||
// namespace, so every app's ingest folder is in one place. Resolvers therefore need the import
|
||||
// root passed in separately; they cannot derive it from hddPath.
|
||||
RootImport BindRoot = "import"
|
||||
)
|
||||
|
||||
// BackupSpec is the .felhom.yml `backup:` block. Paths are forward-slash, relative, path.Clean'd.
|
||||
type BackupSpec struct {
|
||||
Userdata []BindSpec `yaml:"userdata,omitempty" json:"userdata,omitempty"`
|
||||
HDD []BindSpec `yaml:"hdd,omitempty" json:"hdd,omitempty"`
|
||||
// Import classifies ${IMPORT_PATH}-relative binds (R-75). An app whose ingest bind moved from
|
||||
// ${USERDATA_PATH}/import/<app> to ${IMPORT_PATH}/<app> MUST move its backup entry here in the
|
||||
// same change: ValidateBackupSpec rejects an entry matching no compose bind, and the rejection is
|
||||
// WHOLE-BLOCK, so a stale `userdata: import/<app>` would discard the app's OTHER classifications
|
||||
// (e.g. an hdd appdata path classed mandatory) and silently degrade it to legacy.
|
||||
Import []BindSpec `yaml:"import,omitempty" json:"import,omitempty"`
|
||||
}
|
||||
|
||||
// BindSpec is one classified entry in a BackupSpec.
|
||||
@@ -86,6 +97,42 @@ func validClass(c BindClass) bool {
|
||||
}
|
||||
}
|
||||
|
||||
// ValidateRelPath is THE path-safety refusal set for every ${VAR}-relative catalog path — the
|
||||
// `backup:` block and `data_paths:` both run through it, so there is exactly ONE definition of what
|
||||
// a safe relative path is. Refuses: empty, backslash, absolute, non-path.Clean'd, and any leading
|
||||
// ".." escape. It deliberately does NOT check "matches a compose bind" — that rule needs the bind
|
||||
// list and differs per caller (whole-block reject for backup:, per-entry for data_paths:).
|
||||
func ValidateRelPath(root BindRoot, p string) error {
|
||||
where := fmt.Sprintf("%s[%q]", root, p)
|
||||
if p == "" {
|
||||
return fmt.Errorf("%s: empty path", where)
|
||||
}
|
||||
if strings.ContainsRune(p, '\\') {
|
||||
return fmt.Errorf("%s: backslash in path (paths are forward-slash relative)", where)
|
||||
}
|
||||
if path.IsAbs(p) {
|
||||
return fmt.Errorf("%s: absolute path (must be relative to the %s root)", where, root)
|
||||
}
|
||||
if p != path.Clean(p) {
|
||||
return fmt.Errorf("%s: non-clean path (want %q)", where, path.Clean(p))
|
||||
}
|
||||
// path.Clean has run — ".." can only survive as a leading "../" segment.
|
||||
if p == ".." || strings.HasPrefix(p, "../") {
|
||||
return fmt.Errorf("%s: path escapes the root (..)", where)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ValidRoot reports whether r is one of the three known bind roots.
|
||||
func ValidRoot(r BindRoot) bool {
|
||||
switch r {
|
||||
case RootUserdata, RootHDD, RootImport:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// ValidateBackupSpec checks a parsed backup block against the app's actual compose binds and returns
|
||||
// the FIRST defect (whole-block semantics — the caller rejects the ENTIRE block on any error, so the
|
||||
// app degrades to legacy rather than partially classifying). A nil spec is vacuously valid (legacy).
|
||||
@@ -113,21 +160,8 @@ func ValidateBackupSpec(spec *BackupSpec, binds []ComposeBind) error {
|
||||
if !validClass(e.Class) {
|
||||
return fmt.Errorf("%s: invalid class %q (want mandatory|optional|excluded)", where, e.Class)
|
||||
}
|
||||
if e.Path == "" {
|
||||
return fmt.Errorf("%s: empty path", where)
|
||||
}
|
||||
if strings.ContainsRune(e.Path, '\\') {
|
||||
return fmt.Errorf("%s: backslash in path (paths are forward-slash relative)", where)
|
||||
}
|
||||
if path.IsAbs(e.Path) {
|
||||
return fmt.Errorf("%s: absolute path (must be relative to the %s root)", where, root)
|
||||
}
|
||||
if e.Path != path.Clean(e.Path) {
|
||||
return fmt.Errorf("%s: non-clean path (want %q)", where, path.Clean(e.Path))
|
||||
}
|
||||
// path.Clean has run — ".." can only survive as a leading "../" segment.
|
||||
if e.Path == ".." || strings.HasPrefix(e.Path, "../") {
|
||||
return fmt.Errorf("%s: path escapes the root (..)", where)
|
||||
if err := ValidateRelPath(root, e.Path); err != nil {
|
||||
return err
|
||||
}
|
||||
key := string(root) + "\x00" + e.Path
|
||||
if seen[key] {
|
||||
@@ -143,7 +177,10 @@ func ValidateBackupSpec(spec *BackupSpec, binds []ComposeBind) error {
|
||||
if err := check(RootUserdata, spec.Userdata); err != nil {
|
||||
return err
|
||||
}
|
||||
return check(RootHDD, spec.HDD)
|
||||
if err := check(RootHDD, spec.HDD); err != nil {
|
||||
return err
|
||||
}
|
||||
return check(RootImport, spec.Import)
|
||||
}
|
||||
|
||||
// ClassifyBinds resolves every compose bind to a class + origin, applying the two-level default. The
|
||||
@@ -181,6 +218,7 @@ func ClassifyBinds(spec *BackupSpec, binds []ComposeBind) (classified []Classifi
|
||||
}
|
||||
add(RootUserdata, spec.Userdata)
|
||||
add(RootHDD, spec.HDD)
|
||||
add(RootImport, spec.Import)
|
||||
|
||||
for _, b := range binds {
|
||||
cb := ClassifiedBind{ComposeBind: b}
|
||||
|
||||
@@ -51,6 +51,20 @@ type DumpValidation struct {
|
||||
Error string
|
||||
FileSize int64
|
||||
ModTime time.Time
|
||||
// R-44 (v0.148.0) content sniff — a WARN-LEVEL signal, never a gate.
|
||||
//
|
||||
// Structural validity says nothing about whether a dump holds the customer's data. The immich
|
||||
// dump of 2026-07-19 was 52MB, had a valid header and 60+ CREATE TABLEs, and contained zero
|
||||
// users and zero assets: its whole bulk was the geodata reference tables immich ships. Size and
|
||||
// table count are therefore both useless as emptiness heuristics — but an accounts table with
|
||||
// no rows is a strong, cheap, app-agnostic hint that a dump predates the customer entirely.
|
||||
//
|
||||
// Deliberately NOT a refusal: plenty of legitimate apps have no users table (UserTableFound
|
||||
// false → inconclusive → silent), and a false positive that blocked a restore would be far
|
||||
// worse than the skew it guards against. The restore confirm shows it as one extra line.
|
||||
UserTableFound bool
|
||||
UserRows int
|
||||
LooksEmpty bool // UserTableFound && UserRows == 0
|
||||
}
|
||||
|
||||
// DumpFileInfo holds info about a dump file on disk.
|
||||
@@ -101,12 +115,10 @@ func DiscoverDatabases(ctx context.Context, logger *log.Logger, debug bool, know
|
||||
|
||||
id, name, image := parts[0], parts[1], strings.ToLower(parts[2])
|
||||
|
||||
var dbType DBType
|
||||
if strings.Contains(image, "postgres") {
|
||||
dbType = DBTypePostgres
|
||||
} else if strings.Contains(image, "mariadb") || strings.Contains(image, "mysql") {
|
||||
dbType = DBTypeMariaDB
|
||||
} else {
|
||||
// R-47: the same predicate that DBServiceNames applies to compose `image:` values, so a dump
|
||||
// that exists is always attributable to a startable service (see dbservices.go).
|
||||
dbType, isDB := dbTypeForImage(image)
|
||||
if !isDB {
|
||||
if debug {
|
||||
logger.Printf("[DEBUG] DiscoverDatabases: skipping container %s (image=%s, not a database)", name, image)
|
||||
}
|
||||
@@ -363,6 +375,9 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
lineNum := 0
|
||||
headerFound := false
|
||||
tableCount := 0
|
||||
// R-44 sniff state. inUserCopy tracks a postgres `COPY … FROM stdin;` block for an accounts
|
||||
// table; rows are counted until the `\.` terminator.
|
||||
inUserCopy := false
|
||||
for {
|
||||
lineBytes, isPrefix, err := reader.ReadLine()
|
||||
if err != nil {
|
||||
@@ -374,7 +389,12 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
break // EOF
|
||||
}
|
||||
if isPrefix {
|
||||
// Line exceeds buffer — skip remainder (COPY data, large INSERTs)
|
||||
// Line exceeds buffer — skip remainder (COPY data, large INSERTs).
|
||||
// A long line inside a user COPY block is still a ROW: count it before discarding it,
|
||||
// or a table whose rows happen to be wide would sniff as empty and raise a false alarm.
|
||||
if inUserCopy {
|
||||
v.UserRows++
|
||||
}
|
||||
for isPrefix && err == nil {
|
||||
_, isPrefix, err = reader.ReadLine()
|
||||
}
|
||||
@@ -384,6 +404,23 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
line := string(lineBytes)
|
||||
lineNum++
|
||||
|
||||
// R-44 content sniff (warn-level; see DumpValidation).
|
||||
if inUserCopy {
|
||||
if line == `\.` {
|
||||
inUserCopy = false
|
||||
} else {
|
||||
v.UserRows++
|
||||
}
|
||||
} else if isUserCopyStart(line, dbType) {
|
||||
inUserCopy = true
|
||||
v.UserTableFound = true
|
||||
} else if dbType == DBTypeMariaDB && isUserInsert(line) {
|
||||
// mysqldump writes multi-row `INSERT INTO \`users\` VALUES (…),(…);` — the row count is
|
||||
// not worth parsing out of it, and presence alone answers the only question asked here.
|
||||
v.UserTableFound = true
|
||||
v.UserRows++
|
||||
}
|
||||
|
||||
// Header check — scan first 10 lines for expected dump header
|
||||
// MariaDB 11.4+ prepends a sandbox comment before the header line
|
||||
if lineNum <= 10 && !headerFound {
|
||||
@@ -427,10 +464,65 @@ func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
||||
return v
|
||||
}
|
||||
|
||||
v.LooksEmpty = v.UserTableFound && v.UserRows == 0
|
||||
if v.LooksEmpty {
|
||||
log.Printf("[WARN] [backup] ValidateDump: %s is structurally valid (%d tables) but its accounts table has NO rows — the dump may predate the customer's data", filePath, tableCount)
|
||||
}
|
||||
|
||||
v.Valid = true
|
||||
return v
|
||||
}
|
||||
|
||||
// userTableNames are the table names treated as "the accounts table" by the R-44 sniff. Kept
|
||||
// deliberately short: a wider net (anything containing "user") would match join/audit tables like
|
||||
// `user_metadata` or `album_user`, which are legitimately empty on a healthy single-user install
|
||||
// and would produce exactly the false alarm this signal must not raise.
|
||||
var userTableNames = []string{"user", "users", "account", "accounts"}
|
||||
|
||||
// isUserCopyStart reports whether a line opens a postgres `COPY <accounts-table> … FROM stdin;`
|
||||
// block. pg_dump writes the table qualified and optionally quoted — `COPY public."user" (…)`,
|
||||
// `COPY public.users (…)` — so both forms are matched.
|
||||
func isUserCopyStart(line string, dbType DBType) bool {
|
||||
if dbType != DBTypePostgres || !strings.HasPrefix(line, "COPY ") {
|
||||
return false
|
||||
}
|
||||
rest := strings.TrimPrefix(line, "COPY ")
|
||||
sp := strings.IndexByte(rest, ' ')
|
||||
if sp < 0 {
|
||||
return false
|
||||
}
|
||||
return matchesUserTable(rest[:sp])
|
||||
}
|
||||
|
||||
// isUserInsert reports whether a line is a mysqldump INSERT into an accounts table.
|
||||
func isUserInsert(line string) bool {
|
||||
const pfx = "INSERT INTO "
|
||||
if !strings.HasPrefix(line, pfx) {
|
||||
return false
|
||||
}
|
||||
rest := strings.TrimPrefix(line, pfx)
|
||||
sp := strings.IndexByte(rest, ' ')
|
||||
if sp < 0 {
|
||||
return false
|
||||
}
|
||||
return matchesUserTable(rest[:sp])
|
||||
}
|
||||
|
||||
// matchesUserTable strips schema qualification and quoting from a dumped table reference and
|
||||
// reports whether the bare name is an accounts table.
|
||||
func matchesUserTable(ref string) bool {
|
||||
if dot := strings.LastIndexByte(ref, '.'); dot >= 0 {
|
||||
ref = ref[dot+1:]
|
||||
}
|
||||
ref = strings.Trim(ref, "\"`")
|
||||
for _, n := range userTableNames {
|
||||
if strings.EqualFold(ref, n) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ListDumpFiles returns info about SQL dump files on disk.
|
||||
//
|
||||
// M18: ValidateDump scans the dump line-by-line; on a customer with hundreds-of-MB dumps that is wasted
|
||||
@@ -672,7 +764,8 @@ func getMariaDBPassword(ctx context.Context, containerID string) string {
|
||||
// - else known[containerName] → containerName (the container name IS the stack — don't strip, e.g. my-cache).
|
||||
// - else longest known prefix → handles <stack>_postgres / <stack>-1 / compose-suffixed names.
|
||||
// - else → candidate (fall back to today's suffix-strip; preserves behaviour when
|
||||
// the stack list is empty/unavailable, so nothing regresses).
|
||||
// the stack list is empty/unavailable, so nothing regresses).
|
||||
//
|
||||
// A nil/empty `known` map = the legacy fast path (pure suffix-strip).
|
||||
func deriveStackName(containerName string, known map[string]bool) string {
|
||||
candidate := suffixStripStackName(containerName)
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-44 content-sniff tests.
|
||||
//
|
||||
// The dump that triggered this work (DIAG-immich-restore-2026-07-19) was 52MB, had a valid
|
||||
// PostgreSQL header and 60+ CREATE TABLE statements, and contained zero users and zero assets —
|
||||
// its entire bulk was immich's shipped geodata reference tables. Both of the signals the product
|
||||
// already had (file size, table count) called it healthy. These tests pin the one signal that
|
||||
// would have caught it, and the boundaries that keep it from crying wolf.
|
||||
|
||||
func writeDump(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "d.sql")
|
||||
if err := os.WriteFile(p, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
const pgHead = `-- PostgreSQL database dump
|
||||
-- Dumped from database version 16.10
|
||||
SET statement_timeout = 0;
|
||||
SET client_encoding = 'UTF8';
|
||||
CREATE TABLE public.asset (id uuid NOT NULL);
|
||||
CREATE TABLE public."user" (id uuid NOT NULL, email text);
|
||||
`
|
||||
|
||||
// TestSniffFlagsEmptyAccountsTable is the 2026-07-19 shape: structurally perfect, no customer.
|
||||
func TestSniffFlagsEmptyAccountsTable(t *testing.T) {
|
||||
body := pgHead + "COPY public.\"user\" (id, email) FROM stdin;\n\\.\n" +
|
||||
"COPY public.asset (id) FROM stdin;\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if !v.Valid {
|
||||
t.Fatalf("the dump is structurally valid; sniff must not change that: %s", v.Error)
|
||||
}
|
||||
if !v.UserTableFound {
|
||||
t.Fatal("the accounts table COPY block was not recognised")
|
||||
}
|
||||
if v.UserRows != 0 {
|
||||
t.Fatalf("UserRows = %d, want 0", v.UserRows)
|
||||
}
|
||||
if !v.LooksEmpty {
|
||||
t.Fatal("a valid dump with zero account rows MUST raise the warn signal — this is the whole point of R-44")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffQuietOnPopulatedDump — the common case must stay silent, or the warning becomes noise
|
||||
// and gets ignored precisely when it matters.
|
||||
func TestSniffQuietOnPopulatedDump(t *testing.T) {
|
||||
body := pgHead + "COPY public.\"user\" (id, email) FROM stdin;\n" +
|
||||
"a\tone@example.invalid\nb\ttwo@example.invalid\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserRows != 2 {
|
||||
t.Fatalf("UserRows = %d, want 2", v.UserRows)
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("a dump with account rows must not be flagged")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffInconclusiveWithoutAccountsTable — plenty of legitimate apps have no users table. No
|
||||
// table, no claim: a false positive here would warn on every restore of such an app forever.
|
||||
func TestSniffInconclusiveWithoutAccountsTable(t *testing.T) {
|
||||
body := "-- PostgreSQL database dump\nCREATE TABLE public.thing (id int);\n" +
|
||||
"COPY public.thing (id) FROM stdin;\n\\.\n" + strings.Repeat("-- pad\n", 20)
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserTableFound {
|
||||
t.Fatal("no accounts table exists — none must be reported")
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("an app without an accounts table must be INCONCLUSIVE, never flagged empty")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffIgnoresJoinAndAuditTables is the false-alarm guard that shaped the name list, and it is
|
||||
// written as the case that DISCRIMINATES: an app with NO accounts table but with `user_metadata` /
|
||||
// `album_user` / `user_audit` — all legitimately empty on a healthy box. Exact-matching leaves this
|
||||
// inconclusive (silent, correct). A substring match on "user" would treat a join table as the
|
||||
// accounts table, find zero rows, and shout "your backup looks empty" on every single restore of a
|
||||
// perfectly healthy app — which is how a warning signal becomes noise and then gets ignored.
|
||||
func TestSniffIgnoresJoinAndAuditTables(t *testing.T) {
|
||||
body := "-- PostgreSQL database dump\nCREATE TABLE public.album (id int);\n" +
|
||||
"COPY public.user_metadata (id) FROM stdin;\n\\.\n" +
|
||||
"COPY public.album_user (id) FROM stdin;\n\\.\n" +
|
||||
"COPY public.user_audit (id) FROM stdin;\n\\.\n" +
|
||||
"COPY public.album (id) FROM stdin;\n1\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserTableFound {
|
||||
t.Fatal("a join/audit table must never be mistaken for the accounts table")
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("empty join/audit tables must not trigger the warning — this app has no accounts table at all")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffCountsOnlyTheAccountsTable pins the counting boundary separately: with a real accounts
|
||||
// table present, rows from neighbouring user-ish tables must not inflate it.
|
||||
func TestSniffCountsOnlyTheAccountsTable(t *testing.T) {
|
||||
body := pgHead +
|
||||
"COPY public.user_metadata (id) FROM stdin;\nm1\nm2\nm3\n\\.\n" +
|
||||
"COPY public.\"user\" (id, email) FROM stdin;\na\tone@example.invalid\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserRows != 1 {
|
||||
t.Fatalf("only the real accounts table may be counted; UserRows = %d, want 1", v.UserRows)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffCountsWideRows — a row wider than the read buffer is skipped by the structural scan, but
|
||||
// it is still a row. Counting it wrong would flag a populated table as empty (immich asset rows are
|
||||
// genuinely long, which is what makes this reachable).
|
||||
func TestSniffCountsWideRows(t *testing.T) {
|
||||
wide := strings.Repeat("x", 300*1024)
|
||||
body := pgHead + "COPY public.\"user\" (id, email) FROM stdin;\n" + wide + "\n\\.\n"
|
||||
v := ValidateDump(writeDump(t, body), DBTypePostgres)
|
||||
if v.UserRows != 1 {
|
||||
t.Fatalf("a buffer-exceeding row must still count; UserRows = %d, want 1", v.UserRows)
|
||||
}
|
||||
if v.LooksEmpty {
|
||||
t.Fatal("a table whose single row is very wide must not sniff as empty")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSniffMariaDBInsertForm — mysqldump writes multi-row INSERTs, not COPY blocks.
|
||||
func TestSniffMariaDBInsertForm(t *testing.T) {
|
||||
head := "-- MariaDB dump 10.19\nCREATE TABLE `users` (id int);\n" + strings.Repeat("-- pad\n", 20)
|
||||
empty := ValidateDump(writeDump(t, head), DBTypeMariaDB)
|
||||
if empty.UserTableFound {
|
||||
t.Fatal("a CREATE TABLE alone is not an accounts-table row source")
|
||||
}
|
||||
full := ValidateDump(writeDump(t, head+"INSERT INTO `users` VALUES (1),(2);\n"), DBTypeMariaDB)
|
||||
if !full.UserTableFound || full.LooksEmpty {
|
||||
t.Fatalf("a populated mariadb dump must not be flagged: %+v", full)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// R-47 — naming the database SERVICE, not just the running container.
|
||||
//
|
||||
// A dump replay must never race the application's own schema management. Proven live on 2026-07-19
|
||||
// (DIAG-immich-restore-round2-2026-07-19, H4): the reconstitution started the whole stack before
|
||||
// replaying, immich-server rebuilt `clip_index` two seconds before the dump's own CREATE INDEX, and
|
||||
// the replay aborted `already exists` under ON_ERROR_STOP=1 — leaving a half-applied schema that the
|
||||
// app itself then reported as drift. The fix is to bring up ONLY the database service(s) for the
|
||||
// replay, which requires knowing their compose SERVICE names (docker `up -d <svc>` takes service
|
||||
// names, not container names).
|
||||
//
|
||||
// The symmetry that makes this safe: a `.sql` dump can only exist because DiscoverDatabases matched
|
||||
// the running container's image string, and the compose `image:` value IS that image string. So the
|
||||
// same predicate — dbTypeForImage — decides both "is there a dump" and "which service holds it".
|
||||
|
||||
// dbTypeForImage maps a container/compose image reference to the database engine the backup code
|
||||
// supports, or ok=false for anything else (redis/valkey/app images — never started in the DB-only
|
||||
// phase). Extracted from DiscoverDatabases so the discovery heuristic and the compose heuristic can
|
||||
// never drift apart; behaviour is byte-equivalent to the inline form it replaced.
|
||||
func dbTypeForImage(image string) (DBType, bool) {
|
||||
img := strings.ToLower(image)
|
||||
switch {
|
||||
case strings.Contains(img, "postgres"):
|
||||
return DBTypePostgres, true
|
||||
case strings.Contains(img, "mariadb"), strings.Contains(img, "mysql"):
|
||||
return DBTypeMariaDB, true
|
||||
}
|
||||
return DBType(""), false
|
||||
}
|
||||
|
||||
// composeServicesDoc is the minimal view of a compose file needed here: the `services:` MAP and each
|
||||
// service's `image:`. Deliberately a real YAML parse and not a line scan — a top-level `volumes:`
|
||||
// block (immich's `immich_ml_cache:`) has exactly the shape a naive scan misreads as a service, and
|
||||
// starting a phantom service, or missing the real one, both land in the wrong branch.
|
||||
type composeServicesDoc struct {
|
||||
Services map[string]struct {
|
||||
Image string `yaml:"image"`
|
||||
} `yaml:"services"`
|
||||
}
|
||||
|
||||
// DBServiceNames returns the sorted compose SERVICE names in composePath whose `image:` identifies a
|
||||
// supported database engine — the exact argument list for `docker compose up -d <svc>...`.
|
||||
//
|
||||
// A file with no (or an empty) `services:` key returns (nil, nil): an app with no identifiable DB
|
||||
// service is a legitimate, common case and the caller decides what it means. An unreadable or
|
||||
// unparseable file returns an error, because "cannot tell" must never silently read as "no database"
|
||||
// — the callers turn that into a refusal when a dump exists.
|
||||
//
|
||||
// Image values are matched literally. Catalog templates pin their images literally (enforced since
|
||||
// Campaign 7), so an interpolated `${...}` image simply does not match and lands in the caller's
|
||||
// fail-closed branch by design, rather than being guessed at.
|
||||
func DBServiceNames(composePath string) ([]string, error) {
|
||||
data, err := os.ReadFile(composePath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("reading compose file: %w", err)
|
||||
}
|
||||
var doc composeServicesDoc
|
||||
if err := yaml.Unmarshal(data, &doc); err != nil {
|
||||
return nil, fmt.Errorf("parsing compose file %s: %w", composePath, err)
|
||||
}
|
||||
var names []string
|
||||
for name, svc := range doc.Services {
|
||||
if _, ok := dbTypeForImage(svc.Image); ok {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names, nil
|
||||
}
|
||||
@@ -0,0 +1,195 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-47 (v0.153.0) — the DB-service resolver.
|
||||
//
|
||||
// These exist because a dump replay that starts the WHOLE stack races the application's own schema
|
||||
// management: proven live on 2026-07-19 (DIAG-immich-restore-round2-2026-07-19, H4) when
|
||||
// immich-server rebuilt `clip_index` two seconds before the dump's CREATE INDEX and the replay
|
||||
// aborted `already exists`. Closing that window means bringing up ONLY the database service, which
|
||||
// means naming it correctly — every case below is a way of naming it wrongly.
|
||||
|
||||
// writeCompose drops a compose file in a temp dir and returns its path.
|
||||
func writeCompose(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "docker-compose.yml")
|
||||
if err := os.WriteFile(p, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// TestDBTypeForImage pins the shared heuristic. It is the SAME predicate DiscoverDatabases applies to
|
||||
// a running container's image, which is what makes "a dump exists ⇒ a service can be named" hold:
|
||||
// the compose `image:` value IS the container's image string. The table reproduces the inline form
|
||||
// this function replaced, byte for byte, including the redis/valkey negatives that must never be
|
||||
// started in the DB-only window.
|
||||
func TestDBTypeForImage(t *testing.T) {
|
||||
cases := []struct {
|
||||
image string
|
||||
want DBType
|
||||
ok bool
|
||||
}{
|
||||
{"docker.io/library/postgres:16-alpine", DBTypePostgres, true},
|
||||
// immich's real pin — a vector-extended postgres whose REPO segment carries the substring.
|
||||
{"ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0", DBTypePostgres, true},
|
||||
{"postgres", DBTypePostgres, true},
|
||||
{"POSTGRES:16", DBTypePostgres, true}, // the discovery path lowercases; so does this
|
||||
{"mariadb:11", DBTypeMariaDB, true},
|
||||
{"mysql:8.4", DBTypeMariaDB, true},
|
||||
{"docker.io/library/MySQL:8", DBTypeMariaDB, true},
|
||||
{"redis:7-alpine", "", false},
|
||||
{"valkey/valkey:8", "", false},
|
||||
{"ghcr.io/immich-app/immich-server:v1.119.0", "", false},
|
||||
{"", "", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, ok := dbTypeForImage(c.image)
|
||||
if ok != c.ok || (ok && got != c.want) {
|
||||
t.Errorf("dbTypeForImage(%q) = (%q, %v), want (%q, %v)", c.image, got, ok, c.want, c.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDBServiceNames(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
body string
|
||||
want []string
|
||||
}{
|
||||
{
|
||||
name: "postgres service is named",
|
||||
body: "services:\n app:\n image: ghcr.io/x/app:1\n database:\n image: postgres:16\n",
|
||||
want: []string{"database"},
|
||||
},
|
||||
{
|
||||
name: "mariadb service is named",
|
||||
body: "services:\n db:\n image: mariadb:11\n web:\n image: nextcloud:30\n",
|
||||
want: []string{"db"},
|
||||
},
|
||||
{
|
||||
name: "mysql service is named",
|
||||
body: "services:\n mysql:\n image: mysql:8.4\n",
|
||||
want: []string{"mysql"},
|
||||
},
|
||||
{
|
||||
name: "redis-only app has no database service",
|
||||
body: "services:\n app:\n image: ghcr.io/x/app:1\n redis:\n image: redis:7-alpine\n",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "multiple databases are returned SORTED (one up -d carries them all)",
|
||||
body: "services:\n zdb:\n image: postgres:16\n adb:\n image: mariadb:11\n app:\n image: x:1\n",
|
||||
want: []string{"adb", "zdb"},
|
||||
},
|
||||
{
|
||||
name: "no services key at all",
|
||||
body: "volumes:\n data:\n",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "empty services map",
|
||||
body: "services:\n",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "an interpolated image is not guessed at",
|
||||
body: "services:\n db:\n image: ${DB_IMAGE}\n",
|
||||
want: nil,
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got, err := DBServiceNames(writeCompose(t, c.body))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(got, c.want) {
|
||||
t.Errorf("DBServiceNames = %v, want %v", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBServiceNames_TopLevelKeysAreNotServices is the decoy test, and the reason this is a YAML
|
||||
// parse rather than a line scan. immich's real compose carries a top-level `volumes:` block whose
|
||||
// entry (`immich_ml_cache:`) sits at exactly the indentation a service name does, and a top-level
|
||||
// `networks:` block does the same. A scanner that collected "indented keys followed by image-ish
|
||||
// lines" would either invent a service that `docker compose up -d` cannot start, or — worse — match
|
||||
// the wrong one and leave the real database down while the app came up around the replay.
|
||||
func TestDBServiceNames_TopLevelKeysAreNotServices(t *testing.T) {
|
||||
// The service/volume/network names and the image pins are the catalog's real immich template.
|
||||
// `immich_postgres_data` is the trap made concrete: a top-level VOLUME key whose name contains
|
||||
// "postgres" and which no `up -d` could ever start.
|
||||
body := `services:
|
||||
immich-server:
|
||||
image: ghcr.io/immich-app/immich-server:v3.0.3
|
||||
immich-machine-learning:
|
||||
image: ghcr.io/immich-app/immich-machine-learning:v3.0.3
|
||||
immich-postgres:
|
||||
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0
|
||||
immich-redis:
|
||||
image: redis:7-alpine
|
||||
volumes:
|
||||
immich_ml_cache:
|
||||
immich_postgres_data:
|
||||
immich_redis_data:
|
||||
networks:
|
||||
traefik-public:
|
||||
external: true
|
||||
immich-internal:
|
||||
`
|
||||
got, err := DBServiceNames(writeCompose(t, body))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(got, []string{"immich-postgres"}) {
|
||||
t.Fatalf("DBServiceNames = %v, want [immich-postgres] — a top-level volume/network key was mistaken for a service", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBServiceNames_UnreadableAndUnparseableError proves the fail-closed direction: "cannot tell"
|
||||
// must surface as an ERROR, never as the empty (= "this app has no database") answer. The callers
|
||||
// turn an empty result into a refusal only when a dump exists; if a read failure silently produced
|
||||
// the same empty slice for an app with no dump, a genuinely broken compose would flow on unnoticed.
|
||||
func TestDBServiceNames_UnreadableAndUnparseableError(t *testing.T) {
|
||||
if _, err := DBServiceNames(filepath.Join(t.TempDir(), "nope.yml")); err == nil {
|
||||
t.Fatal("a missing compose file must be an error, not an empty service list")
|
||||
}
|
||||
// Valid YAML scalar where a map is required, plus outright broken YAML.
|
||||
if _, err := DBServiceNames(writeCompose(t, "services: [1, 2, 3\n broken")); err == nil {
|
||||
t.Fatal("an unparseable compose file must be an error, not an empty service list")
|
||||
}
|
||||
}
|
||||
|
||||
// TestDiscoverAndComposeAgreeOnTheSameImages is the SYMMETRY guard: whatever image string makes
|
||||
// DiscoverDatabases produce a dump must also make DBServiceNames name a service. They now share one
|
||||
// predicate; this asserts the property that sharing is FOR, so a future edit to either side that
|
||||
// breaks it fails here rather than in a customer's restore.
|
||||
func TestDiscoverAndComposeAgreeOnTheSameImages(t *testing.T) {
|
||||
images := []string{"postgres:16", "mariadb:11", "mysql:8.4", "redis:7", "ghcr.io/x/app:1"}
|
||||
var body strings.Builder
|
||||
body.WriteString("services:\n")
|
||||
var wantDB []string
|
||||
for i, img := range images {
|
||||
svc := string(rune('a' + i))
|
||||
body.WriteString(" " + svc + ":\n image: " + img + "\n")
|
||||
if _, ok := dbTypeForImage(img); ok {
|
||||
wantDB = append(wantDB, svc)
|
||||
}
|
||||
}
|
||||
got, err := DBServiceNames(writeCompose(t, body.String()))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(got, wantDB) {
|
||||
t.Fatalf("compose resolver named %v but the discovery predicate says %v — the two sides have drifted", got, wantDB)
|
||||
}
|
||||
}
|
||||
@@ -21,7 +21,7 @@ func TestComputeFabBuckets_Classified(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/movies"}, Class: ClassExcluded},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/app"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv)
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
if !fb.HasClassification {
|
||||
t.Fatal("HasClassification must be true")
|
||||
}
|
||||
@@ -39,7 +39,7 @@ func TestComputeFabBuckets_Classified(t *testing.T) {
|
||||
// legacy (no block) → empty buckets (the full-root capture stays out of the classified plan).
|
||||
func TestComputeFabBuckets_LegacyEmpty(t *testing.T) {
|
||||
binds := []ClassifiedBind{{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/tv"}, Origin: OriginLegacy}}
|
||||
fb := ComputeFabBuckets(binds, false, drv)
|
||||
fb := ComputeFabBuckets(binds, false, drv, "")
|
||||
if fb.HasClassification || fb.Mandatory != nil || fb.Optional != nil || fb.Excluded != nil {
|
||||
t.Errorf("legacy app must yield empty buckets, got %+v", fb)
|
||||
}
|
||||
@@ -52,7 +52,7 @@ func TestComputeFabBuckets_GuardsAllClasses(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "../evil"}, Class: ClassExcluded},
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "appdata/ok"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv)
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
for _, b := range [][]CapturePath{fb.Mandatory, fb.Optional, fb.Excluded} {
|
||||
for _, p := range b {
|
||||
if p.RelPath == "../evil" {
|
||||
@@ -71,7 +71,7 @@ func TestComputeFabBuckets_NoCrossBucketContainment(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media"}, Class: ClassExcluded},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media/books"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv)
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
if got, want := bucketAbs(fb.Mandatory), []string{udat("media/books")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("mandatory child must survive independently: Mandatory = %v, want %v", got, want)
|
||||
}
|
||||
@@ -86,7 +86,7 @@ func TestComputeFabBuckets_EqualAbsMandatoryWins(t *testing.T) {
|
||||
{ComposeBind: ComposeBind{Root: RootHDD, RelPath: "userdata/media"}, Class: ClassOptional},
|
||||
{ComposeBind: ComposeBind{Root: RootUserdata, RelPath: "media"}, Class: ClassMandatory},
|
||||
}
|
||||
fb := ComputeFabBuckets(binds, true, drv)
|
||||
fb := ComputeFabBuckets(binds, true, drv, "")
|
||||
if got, want := bucketAbs(fb.Mandatory), []string{udat("media")}; !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("collapsed path must land in Mandatory, got Mandatory=%v", got)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
package appbackup
|
||||
|
||||
import "testing"
|
||||
|
||||
// R-203 — the ONE drive-kind rule. Table-driven over BOTH drive kinds on purpose: this defect
|
||||
// survived because it is invisible on the kind that already worked, so a test that only covers the
|
||||
// enrolled drive proves nothing about the fix.
|
||||
|
||||
func TestNamespaceRootFor_BothDriveKinds(t *testing.T) {
|
||||
const sys = "/mnt/sys_drive"
|
||||
cases := []struct {
|
||||
name, drive, want string
|
||||
}{
|
||||
// Scenario B — the enrolled drive must be BYTE-IDENTICAL to pre-R-203 behaviour. The
|
||||
// in-guest mount already IS the namespace root; appending felhom-data here would recreate
|
||||
// the .../felhom-data/felhom-data/... double-nest NamespaceRoot's comment exists to prevent.
|
||||
{"enrolled usb", "/mnt/felhom-usb", "/mnt/felhom-usb"},
|
||||
{"enrolled hdd", "/mnt/felhom-drives/hdd_1", "/mnt/felhom-drives/hdd_1"},
|
||||
{"enrolled nvme", "/mnt/felhom-drives/nvme-1tb", "/mnt/felhom-drives/nvme-1tb"},
|
||||
// Scenario A — the system-data fallback gains the segment. This is the case that was wrong.
|
||||
{"system drive", "/mnt/sys_drive", "/mnt/sys_drive/felhom-data"},
|
||||
// A trailing slash is the same drive. Before R-203 the backup package's copy of this rule
|
||||
// compared WITHOUT Clean while the stacks package's copy compared WITH it — so a config value
|
||||
// with a trailing slash would have flipped the mode in one package and not the other.
|
||||
{"system drive, trailing slash", "/mnt/sys_drive/", "/mnt/sys_drive/felhom-data"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := NamespaceRootFor(tc.drive, sys); got != tc.want {
|
||||
t.Fatalf("NamespaceRootFor(%q, %q) = %q, want %q", tc.drive, sys, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The rule must survive a trailing slash on the SYSTEM path too — it comes from config.
|
||||
func TestIsEnrolledDrive_CleansBothSides(t *testing.T) {
|
||||
if IsEnrolledDrive("/mnt/sys_drive", "/mnt/sys_drive/") {
|
||||
t.Error("a trailing slash on the system path must not make the system drive look enrolled")
|
||||
}
|
||||
if IsEnrolledDrive("/mnt/sys_drive/", "/mnt/sys_drive") {
|
||||
t.Error("a trailing slash on the drive path must not make the system drive look enrolled")
|
||||
}
|
||||
if !IsEnrolledDrive("/mnt/felhom-usb", "/mnt/sys_drive") {
|
||||
t.Error("an enrolled drive must report enrolled")
|
||||
}
|
||||
}
|
||||
|
||||
// The consequence the whole item is about: the directory an app binds and the directory the capture
|
||||
// set looks in must be the SAME on both drive kinds.
|
||||
//
|
||||
// RED-PROOF: replace `UserdataDir(NamespaceRootFor(drive, sys))` with `UserdataDir(drive)` — the
|
||||
// pre-R-203 call — and the system-drive row FAILS with the two paths differing by exactly
|
||||
// `/felhom-data`. That is production behaviour up to v0.196.0.
|
||||
func TestAppBindAndCaptureRootAgree(t *testing.T) {
|
||||
const sys = "/mnt/sys_drive"
|
||||
for _, drive := range []string{"/mnt/felhom-usb", "/mnt/felhom-drives/hdd_1", "/mnt/sys_drive"} {
|
||||
nsRoot := NamespaceRootFor(drive, sys)
|
||||
appBind := UserdataDir(nsRoot) // what the deploy sets as ${USERDATA_PATH}
|
||||
captureRoot := UserdataDir(nsRoot) // what the capture set resolves RootUserdata against
|
||||
if appBind != captureRoot {
|
||||
t.Fatalf("drive %q: the app binds %q while the backup captures %q", drive, appBind, captureRoot)
|
||||
}
|
||||
// And it must be the canonical location — the one EnsureUserdataSkeleton creates.
|
||||
if drive == sys && appBind != "/mnt/sys_drive/felhom-data/userdata" {
|
||||
t.Fatalf("system drive resolved to %q, want the canonical /mnt/sys_drive/felhom-data/userdata", appBind)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -32,6 +32,35 @@ func NamespaceRoot(drivePath string, inGuestDrive bool) string {
|
||||
return filepath.Join(drivePath, FelhomDataDir)
|
||||
}
|
||||
|
||||
// IsEnrolledDrive reports whether a drive path is an ENROLLED user-data drive (Model A: its in-guest
|
||||
// mount already IS the namespace root) rather than the system-data fallback. It is the ONE comparison
|
||||
// that decides which NamespaceRoot mode applies, and it lives here so no package re-derives it.
|
||||
//
|
||||
// Both sides are Clean'd: `/mnt/sys_drive/` and `/mnt/sys_drive` are the same drive, and a trailing
|
||||
// slash arriving from config must not silently flip the mode.
|
||||
func IsEnrolledDrive(drivePath, systemDataPath string) bool {
|
||||
return filepath.Clean(drivePath) != filepath.Clean(systemDataPath)
|
||||
}
|
||||
|
||||
// NamespaceRootFor is the resolver every caller should use when it holds a bare DRIVE path and the
|
||||
// system-data path — i.e. everywhere outside the backup package, which already had this rule.
|
||||
//
|
||||
// R-203: FIVE call sites passed a bare drive path straight to UserdataDir (and its siblings), which
|
||||
// take a NAMESPACE ROOT. On an enrolled drive the two coincide, so nothing showed; on the system-data
|
||||
// fallback they differ by exactly the felhom-data segment, and the app then bound a directory the
|
||||
// backup never looked at. The run still reported ok. Measured live on demo-hp 2026-08-04:
|
||||
// the app wrote to /mnt/sys_drive/userdata/media/books while the off-site capture set looked for
|
||||
// /mnt/sys_drive/felhom-data/userdata/media/books.
|
||||
//
|
||||
// THE CONTRACT, restated because four callers got it wrong and a fifth will: UserdataDir,
|
||||
// PrimaryBackupPath, RecoveryUnitPath and AppDataDir all take a NAMESPACE ROOT. If you are holding
|
||||
// something that came out of HDD_PATH or a StoragePath, it is a DRIVE path — put it through here
|
||||
// first. `UserdataDir(bareDrivePath)` still compiles and is still wrong; TestNoBareDrivePathToUserdataDir
|
||||
// is the guard that keeps the count from growing.
|
||||
func NamespaceRootFor(drivePath, systemDataPath string) string {
|
||||
return NamespaceRoot(drivePath, IsEnrolledDrive(drivePath, systemDataPath))
|
||||
}
|
||||
|
||||
// PrimaryBackupPath returns the root primary backup directory under a felhom-data namespace root.
|
||||
func PrimaryBackupPath(nsRoot string) string {
|
||||
return filepath.Join(nsRoot, "backups", "primary")
|
||||
@@ -40,9 +69,10 @@ func PrimaryBackupPath(nsRoot string) string {
|
||||
// RecoveryUnitPath returns the per-app self-contained recovery-unit ROOT under a namespace root.
|
||||
// It is the existing per-app backup dir (`backups/primary/<stack>/`) — the legacy name is kept so the
|
||||
// db-dumps/ and volume-dumps/ already written there need no migration; the unit gains compose/ and
|
||||
// manifest.json as siblings, making the whole dir a complete, recreatable unit (Phase 2). The unit is
|
||||
// secret-free: secrets/data-keys are recovered from the guest's own app.yaml (live or via PBS), never
|
||||
// stored here. See backup.recoveryUnit / restore for the capture + restore flow.
|
||||
// manifest.json as siblings, making the whole dir a complete, recreatable unit (Phase 2). Since D5 the
|
||||
// unit's compose/app.yaml CARRIES the portable secret class (data keys, DB passwords, internal signing
|
||||
// secrets) at mode 0600, so a Tier-1/2 restore needs the drive and nothing else; internet-reachable
|
||||
// admin logins are still withheld. See backup.recoveryUnit / restore for the capture + restore flow.
|
||||
func RecoveryUnitPath(nsRoot, stackName string) string {
|
||||
return filepath.Join(nsRoot, "backups", "primary", stackName)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-75 Scenario C — DETERMINISM. This is the P6 gate.
|
||||
//
|
||||
// The spike measured the naive map-order derivation producing 20 DISTINCT outputs from 20 identical
|
||||
// runs. fbNeedsRecreate force-recreates FileBrowser on ANY byte difference in the generated config,
|
||||
// and SyncFileBrowserMounts has ~14 call sites — so a non-deterministic skeleton is a fleet-wide
|
||||
// FileBrowser restart loop, the v0.151-class bug. 20 identical generations or this fails.
|
||||
func TestScenarioC_SkeletonDeterminism(t *testing.T) {
|
||||
// Deliberately UNSORTED input, with duplicates and a deep path, so the function has real work to
|
||||
// normalise. A sort applied only to the input would not save a map-ordered implementation.
|
||||
derived := []string{
|
||||
"media/podcasts", "roms", "media/books", "downloads", "media",
|
||||
"media/photos", "media/books", "a/b/c/d",
|
||||
}
|
||||
const n = 20
|
||||
first := BuildUserdataSkeleton(derived)
|
||||
for i := 1; i < n; i++ {
|
||||
got := BuildUserdataSkeleton(derived)
|
||||
if !slices.Equal(got, first) {
|
||||
t.Fatalf("generation %d/%d differs — a non-deterministic skeleton force-recreates FileBrowser on every sync pass\n first: %v\n got: %v",
|
||||
i+1, n, first, got)
|
||||
}
|
||||
}
|
||||
if !slices.IsSorted(first) {
|
||||
t.Errorf("skeleton must be sorted, got %v", first)
|
||||
}
|
||||
// Ancestor expansion: a deep derived path implies its whole chain.
|
||||
for _, want := range []string{"a", "a/b", "a/b/c", "a/b/c/d"} {
|
||||
if !slices.Contains(first, want) {
|
||||
t.Errorf("ancestor chain incomplete: %q missing from %v", want, first)
|
||||
}
|
||||
}
|
||||
// Dedup: "media/books" appeared twice in the input and "media" both derived and as an ancestor.
|
||||
for _, d := range []string{"media", "media/books"} {
|
||||
if c := countOf(first, d); c != 1 {
|
||||
t.Errorf("%q appears %d times, want exactly 1", d, c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func countOf(xs []string, want string) int {
|
||||
n := 0
|
||||
for _, x := range xs {
|
||||
if x == want {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// R-75 Scenario D — ZERO REMOVALS, proven by construction.
|
||||
//
|
||||
// The derived set drops `documents` (implied by no catalog app) and, after the R-75 move, the two
|
||||
// import/* entries. The carry-list is what keeps them. This asserts the merged set is a strict
|
||||
// SUPERSET of the historical hardcoded skeleton for any derived input — including the empty one, the
|
||||
// fresh-box case where the catalog has not synced yet.
|
||||
func TestScenarioD_SkeletonNeverDropsACarriedDir(t *testing.T) {
|
||||
for _, derived := range [][]string{
|
||||
nil, // fresh box, catalog not yet synced
|
||||
{"media/podcasts"}, // the one genuinely new entry
|
||||
{"roms", "downloads", "media/photos"}, // a partial catalog
|
||||
} {
|
||||
got := BuildUserdataSkeleton(derived)
|
||||
for _, carried := range UserdataSkeletonCarry() {
|
||||
if !slices.Contains(got, carried) {
|
||||
t.Errorf("derived=%v: carried dir %q was DROPPED — zero-removals violated", derived, carried)
|
||||
}
|
||||
}
|
||||
}
|
||||
// And the new entry really is added when the catalog implies it.
|
||||
if !slices.Contains(BuildUserdataSkeleton([]string{"media/podcasts"}), "media/podcasts") {
|
||||
t.Error("media/podcasts must be added when the catalog implies it")
|
||||
}
|
||||
// `documents` is the specific entry the spike flagged: in the carry-list, in no catalog app.
|
||||
if !slices.Contains(BuildUserdataSkeleton([]string{"media/podcasts"}), "documents") {
|
||||
t.Error("`documents` must survive — it exists on both demo boxes and may hold customer files")
|
||||
}
|
||||
}
|
||||
|
||||
// A traversal or absolute entry reaching the skeleton would make EnsureUserdataSkeleton create a
|
||||
// directory outside the userdata root. The derived set comes from a compose parser, so this is a
|
||||
// guard on untrusted-ish catalog input, not defence in depth.
|
||||
func TestSkeletonRefusesEscapes(t *testing.T) {
|
||||
got := BuildUserdataSkeleton([]string{"../escape", "..", "", "/abs/path", "ok/dir"})
|
||||
for _, bad := range []string{"../escape", "..", "", "/abs/path"} {
|
||||
if slices.Contains(got, bad) {
|
||||
t.Errorf("escape entry %q must not reach the skeleton: %v", bad, got)
|
||||
}
|
||||
}
|
||||
for _, d := range got {
|
||||
if filepath.IsAbs(d) || d == ".." || len(d) > 3 && d[:3] == "../" {
|
||||
t.Errorf("unsafe skeleton entry %q", d)
|
||||
}
|
||||
}
|
||||
if !slices.Contains(got, "ok/dir") {
|
||||
t.Error("a legitimate entry alongside bad ones must still be kept")
|
||||
}
|
||||
// "/abs/path" is not dropped outright — it is normalised to a relative path and kept, which is
|
||||
// safe (it lands under the userdata root). Pin that so the behaviour is a decision, not a guess.
|
||||
if !slices.Contains(got, "abs/path") {
|
||||
t.Errorf("an absolute entry should be normalised to relative, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// EnsureUserdataSkeleton creates every dir it is given and NOTHING ELSE, and never removes.
|
||||
func TestEnsureUserdataSkeletonCreatesOnly(t *testing.T) {
|
||||
ns := t.TempDir()
|
||||
// A pre-existing customer dir that no catalog app implies and the carry-list does not contain.
|
||||
stray := filepath.Join(UserdataDir(ns), "sajat-mappa")
|
||||
if err := os.MkdirAll(stray, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
dirs := BuildUserdataSkeleton([]string{"media/podcasts"})
|
||||
if err := EnsureUserdataSkeleton(ns, dirs); err != nil {
|
||||
// chown to gid 1000 fails for a non-root test user; the dirs are still created.
|
||||
t.Logf("EnsureUserdataSkeleton returned %v (expected when not running as root)", err)
|
||||
}
|
||||
for _, d := range dirs {
|
||||
if fi, err := os.Stat(filepath.Join(UserdataDir(ns), d)); err != nil || !fi.IsDir() {
|
||||
t.Errorf("skeleton dir %q not created: %v", d, err)
|
||||
}
|
||||
}
|
||||
if _, err := os.Stat(stray); err != nil {
|
||||
t.Errorf("a pre-existing customer dir was removed — zero-removals violated: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// R-75: a DATA drive must never get a per-drive drop-zone from the skeleton. Carrying the old
|
||||
// `import/*` entries would have the skeleton re-create a dead lookalike on every drive forever —
|
||||
// one that is also never backed up, since import paths are class: excluded.
|
||||
//
|
||||
// This is NOT a zero-removals violation: nothing deletes the dirs a box already has (see
|
||||
// TestEnsureUserdataSkeletonCreatesOnly). They stop being maintained and stop appearing on fresh boxes.
|
||||
func TestSkeletonNeverCreatesAPerDriveDropZone(t *testing.T) {
|
||||
// The catalog no longer implies any ${USERDATA_PATH}/import path — the binds moved to
|
||||
// ${IMPORT_PATH} — so the only way one could appear is via the carry-list.
|
||||
for _, derived := range [][]string{nil, {"media/podcasts", "roms"}} {
|
||||
for _, d := range BuildUserdataSkeleton(derived) {
|
||||
if d == "import" || strings.HasPrefix(d, "import/") {
|
||||
t.Errorf("derived=%v: skeleton created a per-drive drop-zone %q — the canonical root is on the SYSTEM drive", derived, d)
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, c := range UserdataSkeletonCarry() {
|
||||
if c == "import" || strings.HasPrefix(c, "import/") {
|
||||
t.Errorf("the carry-list still holds %q", c)
|
||||
}
|
||||
}
|
||||
// A catalog app that genuinely declares a ${USERDATA_PATH}/import/... bind would still be
|
||||
// honoured — the rule is "don't carry them", not "filter them out".
|
||||
if !slices.Contains(BuildUserdataSkeleton([]string{"import/valami"}), "import/valami") {
|
||||
t.Error("a genuinely derived userdata import path must still be created")
|
||||
}
|
||||
}
|
||||
@@ -2,7 +2,10 @@ package appbackup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Customer-facing userdata layout + the shared-storage ownership convention (v0.66.0).
|
||||
@@ -28,19 +31,91 @@ func UserdataDir(nsRoot string) string {
|
||||
return filepath.Join(nsRoot, "userdata")
|
||||
}
|
||||
|
||||
// UserdataSkeleton is the standard subtree created on every storage path (relative to UserdataDir).
|
||||
// ImportDirName is the single import (drop-zone) subtree name under a userdata root.
|
||||
const ImportDirName = "import"
|
||||
|
||||
// ImportDir returns the CANONICAL drop-zone root under a namespace root (R-75).
|
||||
//
|
||||
// Unlike every other userdata dir, this one is drive-INDEPENDENT: the caller resolves it against the
|
||||
// SYSTEM drive's namespace root, never against the app's own HDD_PATH, so a multi-drive box has
|
||||
// exactly ONE import tree. That is the whole point. Each drop-zone app has exactly one ingest bind,
|
||||
// so a per-drive import/ would put a folder that LOOKS like a drop-zone on every drive while only
|
||||
// one of them does anything — and because import paths are `class: excluded`, files stranded in a
|
||||
// dead one are never backed up either.
|
||||
//
|
||||
// It deliberately stays INSIDE the userdata tree, so the 2775/setgid/GID-1000 convention, the
|
||||
// FileBrowser mount and the ownership rules all apply to it unchanged.
|
||||
func ImportDir(nsRoot string) string {
|
||||
return filepath.Join(UserdataDir(nsRoot), ImportDirName)
|
||||
}
|
||||
|
||||
// UserdataSkeletonCarry is the explicit NON-DERIVED carry-list: every entry the v0.171.0 hardcoded
|
||||
// skeleton created, retained verbatim and forever.
|
||||
//
|
||||
// It exists so the catalog-derived skeleton (R-75) can only ever ADD. That makes the zero-removals
|
||||
// invariant true BY CONSTRUCTION rather than by review, and it is not hypothetical:
|
||||
//
|
||||
// - `documents` is implied by NO catalog app (SPIKE P0(a)) yet exists on both demo boxes and is
|
||||
// customer-visible — it may hold customer files. Derivation alone would drop it.
|
||||
//
|
||||
// It doubles as the fresh-box floor: on a box whose catalog has not synced yet the derived set is
|
||||
// empty, and the customer still gets the full standard tree instead of a nearly-empty one.
|
||||
//
|
||||
// DELIBERATELY ABSENT: `import`, `import/paperless`, `import/calibre`. They were in the v0.171.0
|
||||
// hardcoded list, and carrying them would have the skeleton RE-CREATE a per-drive drop-zone on every
|
||||
// drive forever — the exact dead-lookalike R-75 exists to remove, and one that is never backed up
|
||||
// (`class: excluded`). Zero-removals is about not DELETING what a box already has, not about
|
||||
// re-creating it on boxes that never had it: nothing here removes the pre-existing dirs on
|
||||
// demo-felhom / demo-hp, they simply stop being maintained and stop appearing on fresh boxes.
|
||||
// Verified before the change: both boxes' old drop-zones held ZERO files (2026-07-26). A box with
|
||||
// pending files in an old drop-zone would need an operator-run move — see REPORT.md.
|
||||
//
|
||||
// ASCII, no spaces (flows through ${} interpolation, shell, and the rsync merge walk).
|
||||
func UserdataSkeleton() []string {
|
||||
func UserdataSkeletonCarry() []string {
|
||||
return []string{
|
||||
"media", "media/movies", "media/tv", "media/music", "media/audiobooks",
|
||||
"media/books", "media/comics", "media/photos",
|
||||
"downloads",
|
||||
"import", "import/paperless", "import/calibre",
|
||||
"roms",
|
||||
"documents",
|
||||
}
|
||||
}
|
||||
|
||||
// BuildUserdataSkeleton merges the catalog-derived dirs with the carry-list into the final, SORTED
|
||||
// set. Each entry is expanded to its ancestor chain ("media/podcasts" implies "media"), deduped, and
|
||||
// sorted.
|
||||
//
|
||||
// SORTING IS A HARD REQUIREMENT, not tidiness. The FileBrowser config is regenerated from this set
|
||||
// and fbNeedsRecreate force-recreates the container on ANY byte difference. Go randomises map
|
||||
// iteration, and the spike measured the naive map-order derivation producing 20 DISTINCT outputs from
|
||||
// 20 identical runs (SPIKE P6) — which across SyncFileBrowserMounts' ~14 call sites is a fleet-wide
|
||||
// FileBrowser restart loop. TestSkeletonDeterminism pins this.
|
||||
func BuildUserdataSkeleton(derived []string) []string {
|
||||
set := make(map[string]bool, len(derived)+16)
|
||||
addChain := func(rel string) {
|
||||
rel = path.Clean(strings.TrimPrefix(filepath.ToSlash(rel), "/"))
|
||||
if rel == "" || rel == "." || rel == ".." || strings.HasPrefix(rel, "../") {
|
||||
return // never let a traversal or an empty entry become a directory to create
|
||||
}
|
||||
parts := strings.Split(rel, "/")
|
||||
for i := range parts {
|
||||
set[strings.Join(parts[:i+1], "/")] = true
|
||||
}
|
||||
}
|
||||
for _, d := range UserdataSkeletonCarry() {
|
||||
addChain(d)
|
||||
}
|
||||
for _, d := range derived {
|
||||
addChain(d)
|
||||
}
|
||||
out := make([]string, 0, len(set))
|
||||
for d := range set { // map order is RANDOM — the sort below is what makes this deterministic
|
||||
out = append(out, d)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// EnsureDirOwned creates path (idempotent) and enforces the convention: mode 2775 via an explicit
|
||||
// Chmod incl. setgid (MkdirAll cannot) + group = gid. Setting an arbitrary group needs CAP_CHOWN —
|
||||
// the in-guest controller runs as root, so this succeeds in production. Returns the first hard error.
|
||||
@@ -60,7 +135,11 @@ func EnsureUserdataDir(path string) error { return EnsureDirOwned(path, SharedCo
|
||||
// EnsureUserdataSkeleton creates the full userdata tree under a namespace root with the convention.
|
||||
// It creates ALL dirs even if one errors (so a single chown/chmod hiccup doesn't truncate the tree),
|
||||
// returning the first error seen for the caller to log.
|
||||
func EnsureUserdataSkeleton(nsRoot string) error {
|
||||
//
|
||||
// dirs is the merged, sorted set from BuildUserdataSkeleton. This function only ever CREATES: there
|
||||
// is no removal path here or anywhere in R-75, so a directory the current catalog no longer implies
|
||||
// simply stays where it is (Scenario D).
|
||||
func EnsureUserdataSkeleton(nsRoot string, dirs []string) error {
|
||||
base := UserdataDir(nsRoot)
|
||||
var firstErr error
|
||||
rec := func(e error) {
|
||||
@@ -69,7 +148,7 @@ func EnsureUserdataSkeleton(nsRoot string) error {
|
||||
}
|
||||
}
|
||||
rec(EnsureUserdataDir(base))
|
||||
for _, sub := range UserdataSkeleton() {
|
||||
for _, sub := range dirs {
|
||||
rec(EnsureUserdataDir(filepath.Join(base, sub)))
|
||||
}
|
||||
return firstErr
|
||||
|
||||
@@ -13,21 +13,31 @@ func TestSharedContentGID(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestUserdataSkeleton_List asserts the locked skeleton subdir set.
|
||||
// TestUserdataSkeleton_List asserts the locked carry-list. R-75 renamed the hardcoded list to
|
||||
// UserdataSkeletonCarry (it is now the non-derived carry-list) and DELIBERATELY dropped the three
|
||||
// `import*` entries: carrying them would re-create a per-drive drop-zone on every drive forever, the
|
||||
// dead lookalike the canonical root exists to remove. That is not a removal — nothing deletes the
|
||||
// dirs an existing box has; they stop being maintained and stop appearing on fresh boxes. Every other
|
||||
// entry is unchanged, which is the zero-removals promise.
|
||||
func TestUserdataSkeleton_List(t *testing.T) {
|
||||
got := map[string]bool{}
|
||||
for _, s := range UserdataSkeleton() {
|
||||
for _, s := range UserdataSkeletonCarry() {
|
||||
got[s] = true
|
||||
}
|
||||
for _, want := range []string{
|
||||
"media/movies", "media/tv", "media/music", "media/audiobooks", "media/books",
|
||||
"media/comics", "media/photos", "downloads", "import/paperless", "import/calibre",
|
||||
"media/comics", "media/photos", "downloads",
|
||||
"roms", "documents",
|
||||
} {
|
||||
if !got[want] {
|
||||
t.Errorf("skeleton missing %q", want)
|
||||
}
|
||||
}
|
||||
for _, gone := range []string{"import", "import/paperless", "import/calibre"} {
|
||||
if got[gone] {
|
||||
t.Errorf("carry-list must NOT hold %q — the drop-zone is canonical on the system drive (R-75)", gone)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestUserdataDir confirms the userdata root is a sibling under the namespace.
|
||||
@@ -41,9 +51,10 @@ func TestUserdataDir(t *testing.T) {
|
||||
// is ignored — dirs + setgid still land). Runs cross-platform.
|
||||
func TestEnsureUserdataSkeleton_Structure(t *testing.T) {
|
||||
ns := t.TempDir()
|
||||
_ = EnsureUserdataSkeleton(ns) // ignore chown error on a non-root CI host
|
||||
dirs := BuildUserdataSkeleton(nil) // no catalog derived → the carry-list floor
|
||||
_ = EnsureUserdataSkeleton(ns, dirs) // ignore chown error on a non-root CI host
|
||||
base := UserdataDir(ns)
|
||||
for _, sub := range append([]string{""}, UserdataSkeleton()...) {
|
||||
for _, sub := range append([]string{""}, dirs...) {
|
||||
p := filepath.Join(base, sub)
|
||||
if fi, err := os.Stat(p); err != nil || !fi.IsDir() {
|
||||
t.Errorf("skeleton dir missing: %s (%v)", p, err)
|
||||
|
||||
@@ -15,8 +15,8 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
magicHeader = "FABE" // Felhom App Bundle Encrypted
|
||||
scryptN = 1 << 15 // 32768
|
||||
magicHeader = "FABE" // Felhom App Bundle Encrypted
|
||||
scryptN = 1 << 15 // 32768
|
||||
scryptR = 8
|
||||
scryptP = 1
|
||||
saltSize = 32
|
||||
|
||||
@@ -23,6 +23,11 @@ type hddProvider struct {
|
||||
func (p *hddProvider) GetStackNeedsHDD(string) bool { return true }
|
||||
func (p *hddProvider) GetStackHDDMounts(string) []string { return p.mounts }
|
||||
func (p *hddProvider) GetStackHDDPath(string) string { return p.hddPath }
|
||||
func (p *hddProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
|
||||
// R-203: these fixtures use ENROLLED drive paths, where the namespace root IS the drive path.
|
||||
// Delegating keeps that identity explicit rather than hardcoding it.
|
||||
func (p *hddProvider) GetStackNamespaceRoot(name string) string { return p.GetStackHDDPath(name) }
|
||||
func (p *hddProvider) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return p.binds, p.hasBinds
|
||||
}
|
||||
|
||||
@@ -96,10 +96,26 @@ type Exporter struct {
|
||||
// computation walks. Nil → the real os.ReadDir-based lister.
|
||||
dirLister func(dir string) []string
|
||||
|
||||
// stopGuard (R-166) marks the stop→export→start window so a controller killed inside it leaves a
|
||||
// durable record that the app is owed a restart. Declared consumer-side as a two-method interface
|
||||
// so this package does not import internal/backup; main.go passes the backup manager's guard, so
|
||||
// BOTH packages write ONE marker file — an exporter with its own file would be a second writer
|
||||
// racing the same recovery. Nil = not wired (tests): the export runs exactly as it did before.
|
||||
stopGuard appStopGuard
|
||||
|
||||
mu sync.Mutex
|
||||
activeJob *Job
|
||||
}
|
||||
|
||||
// appStopGuard is the app-stop crash-marker seam. The REASON is deliberately not a parameter: it is
|
||||
// always "app export" from here, and the adapter in main.go supplies it. Passing it as a string
|
||||
// would duplicate backup.ReasonAppExport's value in a second package with nothing keeping the two in
|
||||
// step — a drift this codebase has paid for before (the offbox key that was guessed, R-7b).
|
||||
type appStopGuard interface {
|
||||
Begin(opID string, stacks []string) error
|
||||
End()
|
||||
}
|
||||
|
||||
// NewExporter creates a new export/import engine.
|
||||
func NewExporter(provider ExportStackProvider, logger *log.Logger, version string) *Exporter {
|
||||
return &Exporter{
|
||||
@@ -109,6 +125,18 @@ func NewExporter(provider ExportStackProvider, logger *log.Logger, version strin
|
||||
}
|
||||
}
|
||||
|
||||
// SetStopGuard wires the app-stop crash marker. INIT-ONLY — call once at startup, before any export.
|
||||
func (e *Exporter) SetStopGuard(g appStopGuard) { e.stopGuard = g }
|
||||
|
||||
// stopGuardBegin records the app-stop marker before an export stops an app. An unwired guard is a
|
||||
// no-op (pre-v0.189.0 behaviour), never an error — a test exporter must not be forced to have one.
|
||||
func (e *Exporter) stopGuardBegin(stackName string) error {
|
||||
if e.stopGuard == nil {
|
||||
return nil
|
||||
}
|
||||
return e.stopGuard.Begin("app-export:"+stackName, []string{stackName})
|
||||
}
|
||||
|
||||
// SetDebug enables or disables verbose debug logging.
|
||||
func (e *Exporter) SetDebug(debug bool) {
|
||||
e.debug = debug
|
||||
@@ -226,6 +254,14 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
// Optionally stop the app
|
||||
wasRunning := false
|
||||
if req.StopApp && e.provider.IsStackRunning(req.StackName) {
|
||||
// R-166: mark BEFORE the stop. The defer below covers the graceful exits; it does NOT cover a
|
||||
// SIGKILL or a power cut, which run no deferred function (Campaign 8 fault 10, on live
|
||||
// hardware) — only this marker does, and a big export is a long window to be killed in.
|
||||
if err := e.stopGuardBegin(req.StackName); err != nil {
|
||||
e.failJob(job, step, "Az alkalmazás leállítása előtti jelölő nem menthető — az exportálás nem indult el.")
|
||||
e.logger.Printf("[ERROR] Export: could not record the app-stop marker for %s (refusing to stop it unprotected): %v", req.StackName, err)
|
||||
return
|
||||
}
|
||||
wasRunning = true
|
||||
e.logger.Printf("[INFO] Export: stopping %s", req.StackName)
|
||||
e.debugf("stopping stack %s before export", req.StackName)
|
||||
@@ -246,6 +282,11 @@ func (e *Exporter) executeExport(req ExportRequest, job *Job) {
|
||||
e.logger.Printf("[WARN] Export: could not restart %s: %v", req.StackName, err)
|
||||
} else {
|
||||
e.debugf("stack %s restarted successfully", req.StackName)
|
||||
// Cleared only on a restart that succeeded — a failed one keeps the marker so the
|
||||
// next startup retries.
|
||||
if e.stopGuard != nil {
|
||||
e.stopGuard.End()
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
@@ -468,8 +509,8 @@ func (e *Exporter) GetDebugInfo() map[string]interface{} {
|
||||
defer e.mu.Unlock()
|
||||
|
||||
info := map[string]interface{}{
|
||||
"debug_enabled": e.debug,
|
||||
"version": e.version,
|
||||
"debug_enabled": e.debug,
|
||||
"version": e.version,
|
||||
"has_active_job": e.activeJob != nil,
|
||||
}
|
||||
|
||||
@@ -618,7 +659,9 @@ func (e *Exporter) exportHDDData(req ExportRequest, dataDir string, manifest *Ma
|
||||
// Task 4: the class-scoped plan. Legacy / no-block apps get an EMPTY plan (all mounts kept, root
|
||||
// tar with zero excludes) → byte-identical v0.130.0 capture.
|
||||
plan := e.computeFabPlan(req, mounts)
|
||||
ud := appbackup.UserdataDir(filepath.Clean(e.provider.GetStackHDDPath(stackName)))
|
||||
// R-203: a NAMESPACE ROOT, not the drive path (identical on an enrolled drive; one segment short
|
||||
// on the system-data fallback).
|
||||
ud := appbackup.UserdataDir(filepath.Clean(e.provider.GetStackNamespaceRoot(stackName)))
|
||||
|
||||
claimed := make(map[string]string) // subdir → mount that claimed it
|
||||
for _, mount := range mounts {
|
||||
|
||||
@@ -189,7 +189,7 @@ func TestExport_HDDMountBasenameCollisionFailsLoud(t *testing.T) {
|
||||
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true},
|
||||
mounts: []string{a, b}, hddPath: hdd,
|
||||
mounts: []string{a, b}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, log.New(io.Discard, "", 0), "test")
|
||||
@@ -229,7 +229,7 @@ func TestFabRoundTrip_UserdataPlacement(t *testing.T) {
|
||||
|
||||
prov := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: true},
|
||||
mounts: []string{ud}, hddPath: hdd,
|
||||
mounts: []string{ud}, hddPath: hdd,
|
||||
}
|
||||
drive := t.TempDir()
|
||||
e := NewExporter(prov, lg, "test")
|
||||
@@ -249,7 +249,7 @@ func TestFabRoundTrip_UserdataPlacement(t *testing.T) {
|
||||
|
||||
prov2 := &hddProvider{
|
||||
rtProvider: &rtProvider{stackDir: srcStack, stacksDir: t.TempDir(), deployed: false},
|
||||
hddPath: hdd,
|
||||
hddPath: hdd,
|
||||
}
|
||||
e2 := NewExporter(prov2, lg, "test")
|
||||
if err := e2.StartImport(ImportRequest{FABPath: fabPath}); err != nil {
|
||||
|
||||
@@ -34,8 +34,12 @@ func (e *Exporter) computeFabPlan(req ExportRequest, mounts []string) fabPlan {
|
||||
if !has {
|
||||
return fabPlan{} // legacy: byte-identical v0.130.0 capture
|
||||
}
|
||||
hddPath := filepath.Clean(e.provider.GetStackHDDPath(req.StackName))
|
||||
fb := appbackup.ComputeFabBuckets(binds, has, hddPath)
|
||||
// R-203: the shared resolver's root parameter is a NAMESPACE ROOT — that is what the off-site
|
||||
// side has always passed (ComputeCaptureSet ← offbox_capture.go). This site passed the bare drive
|
||||
// path, so on the system-data fallback the export's classified paths and the backup's capture set
|
||||
// described DIFFERENT directories for the same declared bind. They now agree by construction.
|
||||
nsRoot := filepath.Clean(e.provider.GetStackNamespaceRoot(req.StackName))
|
||||
fb := appbackup.ComputeFabBuckets(binds, has, nsRoot, e.provider.GetImportRoot())
|
||||
|
||||
deselect := sliceSet(req.DeselectOptional)
|
||||
optIn := sliceSet(req.OptInExcluded)
|
||||
@@ -82,7 +86,10 @@ func (e *Exporter) computeFabPlan(req ExportRequest, mounts []string) fabPlan {
|
||||
}
|
||||
|
||||
plan := fabPlan{SkipMounts: map[string]bool{}}
|
||||
ud := appbackup.UserdataDir(hddPath)
|
||||
// R-203: UserdataDir takes a NAMESPACE ROOT, not the drive path. Identical on an enrolled drive;
|
||||
// one segment short on the system-data fallback, which is where the export plan then skipped (or
|
||||
// failed to skip) the wrong directory.
|
||||
ud := appbackup.UserdataDir(nsRoot)
|
||||
for _, m := range mounts {
|
||||
mc := filepath.Clean(m)
|
||||
if mc == filepath.Clean(ud) {
|
||||
@@ -116,8 +123,8 @@ func (e *Exporter) fabEstimateSplit(stackName string, est *ExportEstimate, volum
|
||||
if !has {
|
||||
return
|
||||
}
|
||||
hddPath := filepath.Clean(e.provider.GetStackHDDPath(stackName))
|
||||
fb := appbackup.ComputeFabBuckets(binds, has, hddPath)
|
||||
nsRoot := filepath.Clean(e.provider.GetStackNamespaceRoot(stackName)) // R-203, as above
|
||||
fb := appbackup.ComputeFabBuckets(binds, has, nsRoot, e.provider.GetImportRoot())
|
||||
est.HasClassification = true
|
||||
|
||||
toItems := func(cps []appbackup.CapturePath) ([]FabItem, int64) {
|
||||
|
||||
@@ -22,7 +22,12 @@ type fabProv struct {
|
||||
mounts []string
|
||||
}
|
||||
|
||||
func (p *fabProv) GetStackHDDPath(string) string { return p.hddPath }
|
||||
func (p *fabProv) GetStackHDDPath(string) string { return p.hddPath }
|
||||
func (p *fabProv) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
|
||||
// R-203: these fixtures use ENROLLED drive paths, where the namespace root IS the drive path.
|
||||
// Delegating keeps that identity explicit rather than hardcoding it.
|
||||
func (p *fabProv) GetStackNamespaceRoot(name string) string { return p.GetStackHDDPath(name) }
|
||||
func (p *fabProv) GetStackHDDMounts(string) []string { return p.mounts }
|
||||
func (p *fabProv) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return p.binds, p.has
|
||||
|
||||
@@ -17,6 +17,14 @@ type ExportStackProvider interface {
|
||||
GetStackHDDMounts(name string) []string
|
||||
// GetStackHDDPath returns the raw HDD_PATH env var from app.yaml.
|
||||
GetStackHDDPath(name string) string
|
||||
// GetImportRoot returns the CANONICAL drop-zone root (R-75), on the SYSTEM drive. ${IMPORT_PATH}
|
||||
// binds resolve against THIS, never against GetStackHDDPath. Empty when unresolvable.
|
||||
GetImportRoot() string
|
||||
// GetStackNamespaceRoot returns the app's felhom-data NAMESPACE ROOT — the directory that directly
|
||||
// contains backups/ and userdata/. It is NOT GetStackHDDPath: on an enrolled drive the two are the
|
||||
// same, and on the system-data fallback the namespace root has one more segment (R-203). Every
|
||||
// appbackup path helper takes THIS, never the drive path. Empty when the app has no HDD_PATH.
|
||||
GetStackNamespaceRoot(name string) string
|
||||
// GetStackClassifiedBinds returns the app's backup-classified compose binds + whether it carries a
|
||||
// (valid) backup block (Task 2). Drives the `.fab` class-scoped export plan (Task 4); a legacy app
|
||||
// (false) exports the v0.130.0 full-root capture unchanged.
|
||||
|
||||
@@ -35,22 +35,27 @@ func (p *rtProvider) GetStackDir(string) (string, bool) { return p.stackDir, tru
|
||||
func (p *rtProvider) GetStackComposePath(string) (string, bool) {
|
||||
return filepath.Join(p.stackDir, "docker-compose.yml"), true
|
||||
}
|
||||
func (p *rtProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *rtProvider) GetStackHDDPath(string) string { return "" }
|
||||
func (p *rtProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *rtProvider) GetStackHDDPath(string) string { return "" }
|
||||
func (p *rtProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
|
||||
// R-203: these fixtures use ENROLLED drive paths, where the namespace root IS the drive path.
|
||||
// Delegating keeps that identity explicit rather than hardcoding it.
|
||||
func (p *rtProvider) GetStackNamespaceRoot(name string) string { return p.GetStackHDDPath(name) }
|
||||
func (p *rtProvider) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
func (p *rtProvider) IsStackRunning(string) bool { return p.running }
|
||||
func (p *rtProvider) StopStack(string) error { p.stopped++; return nil }
|
||||
func (p *rtProvider) StartStack(string) error { p.started = true; return nil }
|
||||
func (p *rtProvider) GetStackDisplayName(n string) string { return "RT " + n }
|
||||
func (p *rtProvider) GetStackNeedsHDD(string) bool { return false }
|
||||
func (p *rtProvider) GetDockerVolumes(string) []string { return p.volumes }
|
||||
func (p *rtProvider) IsStackDeployed(string) bool { return p.deployed }
|
||||
func (p *rtProvider) GetDecryptedEnv(string) map[string]string { return nil }
|
||||
func (p *rtProvider) GetStacksBaseDir() string { return p.stacksDir }
|
||||
func (p *rtProvider) RefreshStacks() error { return nil }
|
||||
func (p *rtProvider) RemoveStackVolumes(string) error { p.removed++; return nil }
|
||||
func (p *rtProvider) IsStackRunning(string) bool { return p.running }
|
||||
func (p *rtProvider) StopStack(string) error { p.stopped++; return nil }
|
||||
func (p *rtProvider) StartStack(string) error { p.started = true; return nil }
|
||||
func (p *rtProvider) GetStackDisplayName(n string) string { return "RT " + n }
|
||||
func (p *rtProvider) GetStackNeedsHDD(string) bool { return false }
|
||||
func (p *rtProvider) GetDockerVolumes(string) []string { return p.volumes }
|
||||
func (p *rtProvider) IsStackDeployed(string) bool { return p.deployed }
|
||||
func (p *rtProvider) GetDecryptedEnv(string) map[string]string { return nil }
|
||||
func (p *rtProvider) GetStacksBaseDir() string { return p.stacksDir }
|
||||
func (p *rtProvider) RefreshStacks() error { return nil }
|
||||
func (p *rtProvider) RemoveStackVolumes(string) error { p.removed++; return nil }
|
||||
func (p *rtProvider) SaveEncryptedAppConfig(stackDir string, env map[string]string) error {
|
||||
p.savedEnv = env
|
||||
return nil
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ── Backup admission (R-181) ─────────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// WHAT WAS WRONG. B2's capture floor (v0.192.0, R-165) shipped as the deliberate replacement for the
|
||||
// bulkhead the `mp1` partition used to give, and it was consulted in exactly ONE place —
|
||||
// `captureAllRecoveryUnits`, which writes a manifest and three compose files: a few KB. The two legs
|
||||
// that write the BULK into the same `backups/primary/<app>` tree — the database dump and the volume
|
||||
// dump — ran FIRST and unguarded. Measured on demo-hp 2026-08-03 06:40:03: opengist's volume dump
|
||||
// wrote 2.0 GB with no check, free fell to 1.0 GB, and the floor then refused the cheap write it had
|
||||
// already lost the argument to. Its refusal message said *"the previous unit is untouched"*, which
|
||||
// was false by then — that app's tar had gone 182,272 B → 2,147,666,432 B under a stale manifest.
|
||||
//
|
||||
// WHAT THIS IS. ONE verdict per app per run, taken before that app's FIRST write of the run, covering
|
||||
// all three legs. The three write under one per-app root (`appbackup.RecoveryUnitPath`), which is
|
||||
// exactly why one verdict can honestly cover them — and why the message may now claim what it claims.
|
||||
//
|
||||
// WHY IT IS DECIDED LAZILY AND NOT ONCE AT THE START OF THE RUN. Space changes during a run: app A's
|
||||
// 2 GB dump can put app B under the reserve. A verdict taken at run start would wave B through on a
|
||||
// reading that was true before the disk filled — the same class of mistake as the one being fixed,
|
||||
// moved one level up.
|
||||
//
|
||||
// WHY IT IS REMEMBERED AND NOT RE-DECIDED PER LEG. Re-deciding between an app's own legs reintroduces
|
||||
// the split this closes: the DB leg admitted, the volume leg admitted, the capture refused — with the
|
||||
// bulk already written. Decide once, remember, reuse; reset per run, because a set carried between
|
||||
// runs is a wrong answer with a confident face.
|
||||
//
|
||||
// IT REFUSES; IT NEVER DELETES. Unchanged from B2 and load-bearing: nothing on this filesystem is
|
||||
// generational (one unit per app at one fixed path, refreshed in place), so "prune the oldest" could
|
||||
// only mean destroying a DIFFERENT app's only local copy. `pruneStalePrimaryDirs` removes ORPHANED
|
||||
// dirs an app left on a drive it moved off — it has no notion of age or of the current app — and must
|
||||
// never be repurposed for headroom.
|
||||
|
||||
// floorReason records WHICH term bound, so the operator can tell "the disk is full" from "this app's
|
||||
// backup is too big for what is left". An alert that says only "refused" sends them to read code.
|
||||
type floorReason int
|
||||
|
||||
const (
|
||||
floorAdmit floorReason = iota // admitted — no term binds
|
||||
floorHeadroom // the filesystem is ALREADY at/below the reserve
|
||||
floorSize // there is room now, but this app's own write would cross the reserve
|
||||
)
|
||||
|
||||
func (r floorReason) String() string {
|
||||
switch r {
|
||||
case floorHeadroom:
|
||||
return "headroom"
|
||||
case floorSize:
|
||||
return "size"
|
||||
default:
|
||||
return "admitted"
|
||||
}
|
||||
}
|
||||
|
||||
// admissionVerdict is one app's decision for one run. It carries everything the alert needs, so the
|
||||
// alert is rendered once from the same value every leg consults.
|
||||
type admissionVerdict struct {
|
||||
admitted bool
|
||||
reason floorReason
|
||||
usage *UnitSpace
|
||||
estGiB float64 // the estimated write in GiB — the arithmetic unit, matching the reserve's terms
|
||||
estBytes int64 // the same estimate in bytes — the RENDERING unit; see floorRefusal
|
||||
hasEst bool // whether an estimate was available at all (§8.2: distinct from "estimated 0")
|
||||
err error // the refusal, nil when admitted
|
||||
}
|
||||
|
||||
// admissionSet is the per-RUN memo. Deliberately not a field with a lifetime of its own: it is
|
||||
// created by beginAdmissionRun and cleared by the returned func, so an absent set means "no run is in
|
||||
// flight" rather than "a stale answer from last night".
|
||||
type admissionSet struct {
|
||||
v map[string]admissionVerdict
|
||||
}
|
||||
|
||||
// beginAdmissionRun opens the per-run admission scope and returns the closer. Called once at the top
|
||||
// of runDBDumpsInternal — which is the single orchestrator of all three legs — so the DB dump, the
|
||||
// volume dump and the capture of one app all consult the SAME verdict.
|
||||
//
|
||||
// A second call while a set is live REPLACES it and the returned closer restores the previous one, so
|
||||
// nesting cannot silently drop a caller's scope.
|
||||
func (m *Manager) beginAdmissionRun() func() {
|
||||
m.admissionMu.Lock()
|
||||
prev := m.admission
|
||||
m.admission = &admissionSet{v: map[string]admissionVerdict{}}
|
||||
m.admissionMu.Unlock()
|
||||
return func() {
|
||||
m.admissionMu.Lock()
|
||||
m.admission = prev
|
||||
m.admissionMu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// admitApp is THE gate. It returns true when this app may write, false when the reserve refuses it.
|
||||
//
|
||||
// On the first refusal for an app it logs and fires EXACTLY ONE operator alert; every later leg in
|
||||
// the same run reads the memo and stays silent, so a refused app produces one email and not three.
|
||||
//
|
||||
// With no run scope open (the periodic status refresh calls captureAllRecoveryUnits directly) it
|
||||
// decides fresh. That is not a gap: each app appears once in that sweep, so "once per app" still
|
||||
// holds — there is simply nothing to remember it across.
|
||||
func (m *Manager) admitApp(stackName string) bool {
|
||||
m.admissionMu.Lock()
|
||||
defer m.admissionMu.Unlock()
|
||||
|
||||
if set := m.admission; set != nil {
|
||||
if v, ok := set.v[stackName]; ok {
|
||||
return v.admitted // already decided this run — do NOT re-decide, do NOT re-alert
|
||||
}
|
||||
}
|
||||
|
||||
v := m.decideAdmission(stackName)
|
||||
if set := m.admission; set != nil {
|
||||
set.v[stackName] = v
|
||||
}
|
||||
if v.admitted {
|
||||
return true
|
||||
}
|
||||
|
||||
// The claim below is now literally true, and that is the whole point of R-181: the verdict is
|
||||
// taken before the FIRST of the three writes, so at this moment nothing under
|
||||
// backups/primary/<app> has been touched by this run. TestAdmission_RefusedAppsTreeIsByteIdentical
|
||||
// pins the consequence by checksumming the tree, not by reading this line.
|
||||
m.logger.Printf("[WARN] [backup] App backup REFUSED for %s (%s) — %v; NO database dump, NO volume "+
|
||||
"dump and NO recovery-unit capture was written for it, the previous unit is untouched and "+
|
||||
"NOTHING was deleted", stackName, v.reason, v.err)
|
||||
if m.unitNotify != nil {
|
||||
m.unitNotify(stackName, v.err, v.usage)
|
||||
}
|
||||
// R-182: the digest entry is recorded HERE, where the verdict is taken — once per app per run.
|
||||
// Not at the three call sites that consult the memo: R-181's whole contract is that ONE verdict
|
||||
// covers all three legs, so noting it per leg listed a single refused app three times and
|
||||
// produced counts like "2 of 1 apps failed". The leg name says what actually happened, which is
|
||||
// that nothing was attempted at all.
|
||||
m.noteFailure(stackName, "whole app (refused before any write)", v.err.Error())
|
||||
return false
|
||||
}
|
||||
|
||||
// decideAdmission applies the floor to a fresh reading plus this app's estimated write.
|
||||
func (m *Manager) decideAdmission(stackName string) admissionVerdict {
|
||||
estBytes, hasEst := m.estimatedWriteBytes(stackName)
|
||||
estGiB := float64(estBytes) / (1024 * 1024 * 1024)
|
||||
usage, reason := m.floorVerdict(m.readUnitSpace(stackName), estGiB)
|
||||
v := admissionVerdict{
|
||||
admitted: reason == floorAdmit,
|
||||
reason: reason,
|
||||
usage: usage,
|
||||
estGiB: estGiB,
|
||||
estBytes: estBytes,
|
||||
hasEst: hasEst,
|
||||
}
|
||||
if !v.admitted {
|
||||
v.err = floorRefusal(reason, usage, estBytes, hasEst)
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// floorRefusal renders the refusal an operator reads. It names the reserve (not an I/O error — this
|
||||
// is a deliberate hold, not broken machinery), says WHICH term bound, and states plainly when the
|
||||
// decision was headroom-only because the app has no previous backup to estimate from (§8.2).
|
||||
//
|
||||
// THE ESTIMATE IS RENDERED IN BYTES-HUMANIZED, NOT GiB, and that is not cosmetic. Fixed to two
|
||||
// decimal GiB, every app under ~10 MB prints `0.00 GiB` — which reads as "no estimate was available"
|
||||
// and is the opposite of what happened. Observed on the live proof run: opengist's real 178 KB
|
||||
// estimate rendered as `estimated 0.00 GiB write`. The arithmetic stays in GiB (the reserve's own
|
||||
// unit); only the rendering changes.
|
||||
func floorRefusal(reason floorReason, usage *UnitSpace, estBytes int64, hasEst bool) error {
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "%%w (reserve: %.0f%%%% used or %.1f GiB free", FloorUsedPercent, FloorFreeGiB)
|
||||
switch {
|
||||
case reason == floorSize:
|
||||
fmt.Fprintf(&b, "; this app's last backup was %s and writing it again would cross the reserve", humanizeBytes(estBytes))
|
||||
case hasEst:
|
||||
fmt.Fprintf(&b, "; the filesystem is already below it, before this app's estimated %s write", humanizeBytes(estBytes))
|
||||
default:
|
||||
b.WriteString("; this app has no previous backup on disk, so only current headroom was considered")
|
||||
}
|
||||
b.WriteString(") — %s")
|
||||
return fmt.Errorf(b.String(), ErrCaptureFloor, usage)
|
||||
}
|
||||
|
||||
// estimatedWriteBytes estimates what this app's three legs are about to write, from what the PREVIOUS
|
||||
// run left in its unit: the `.sql` dumps and the `.tar` volume archives already on disk for this app.
|
||||
//
|
||||
// WHY THIS ESTIMATOR. It is free — two ReadDirs of a directory the caller is about to write into — and
|
||||
// the next write is usually close to the last one. The alternative, a container-based `du` of every
|
||||
// named volume, was measured on the demo box before being rejected; the figure is in REPORT.md §6.
|
||||
//
|
||||
// NO HISTORY → (0, false), and the caller falls back to headroom-only. Refusing an app because it has
|
||||
// never been backed up would make the first backup the one that can never happen (Scenario E).
|
||||
//
|
||||
// It reads the app's CURRENT unit root, so an app that moved drives estimates from its new (probably
|
||||
// empty) location and is treated as history-less — conservative in the admitting direction, which is
|
||||
// the right way round for an estimate that only ever tightens a threshold.
|
||||
func (m *Manager) estimatedWriteBytes(stackName string) (int64, bool) {
|
||||
drivePath := m.GetAppDrivePath(stackName)
|
||||
if drivePath == "" {
|
||||
return 0, false
|
||||
}
|
||||
nsRoot := m.namespaceRoot(drivePath)
|
||||
var total int64
|
||||
var found bool
|
||||
for _, d := range []struct {
|
||||
dir string
|
||||
ext string
|
||||
}{
|
||||
{AppDBDumpPath(nsRoot, stackName), ".sql"},
|
||||
{AppVolumeDumpPath(nsRoot, stackName), ".tar"},
|
||||
} {
|
||||
n, ok := sumFileSizes(d.dir, d.ext)
|
||||
total += n
|
||||
found = found || ok
|
||||
}
|
||||
if !found {
|
||||
return 0, false
|
||||
}
|
||||
return total, true
|
||||
}
|
||||
|
||||
// sumFileSizes totals the sizes of files with the given suffix in dir. The bool reports whether ANY
|
||||
// such file was seen — distinct from a zero total, because a 0-byte dump is history (a real, if
|
||||
// alarming, previous result) while an absent directory is not.
|
||||
//
|
||||
// A stat error on one entry is skipped rather than aborting the sum: an estimate built from the
|
||||
// readable files is worth more than no estimate, and the entry that could not be read is logged
|
||||
// nowhere because this is a hint, not a measurement — it can only tighten a threshold, never relax
|
||||
// one below what the headroom term already enforces.
|
||||
func sumFileSizes(dir, suffix string) (int64, bool) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
var total int64
|
||||
var found bool
|
||||
for _, e := range entries {
|
||||
if e.IsDir() || !strings.HasSuffix(e.Name(), suffix) {
|
||||
continue
|
||||
}
|
||||
fi, err := os.Stat(filepath.Join(dir, e.Name()))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
total += fi.Size()
|
||||
}
|
||||
return total, found
|
||||
}
|
||||
@@ -0,0 +1,741 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-181 — the reserve guards the write that fills the disk, and its promise is true.
|
||||
//
|
||||
// WHAT THESE ASSERT, AND WHY IT IS THE TREE AND NOT THE LOG. The defect being closed is precisely a
|
||||
// log line that claimed something the filesystem contradicted: B2 printed *"the previous unit is
|
||||
// untouched"* while the volume leg had already rewritten that unit's tar 182,272 B → 2,147,666,432 B.
|
||||
// So a test that reads the message and believes it would have passed against the broken code. Every
|
||||
// refusal test here checksums the whole `backups/primary` tree before and after and compares.
|
||||
|
||||
// ── Harness ──────────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
// admissionProvider records the two acts a refused app must never suffer: its recovery info being
|
||||
// read (a capture that was ATTEMPTED) and its stack being stopped (which DumpAppVolumesSafe does as
|
||||
// its first act, before any check of its own).
|
||||
type admissionProvider struct {
|
||||
stacks []string
|
||||
volumes map[string][]string
|
||||
hdd map[string]string // per-app drive path, for the drive-state skip tests
|
||||
dir string
|
||||
infoHits []string
|
||||
stopped []string
|
||||
}
|
||||
|
||||
func (p *admissionProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *admissionProvider) ListDeployedStacks() []StackSummary {
|
||||
out := make([]StackSummary, 0, len(p.stacks))
|
||||
for _, s := range p.stacks {
|
||||
out = append(out, StackSummary{Name: s})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *admissionProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *admissionProvider) GetStackHDDPath(n string) string { return p.hdd[n] }
|
||||
func (p *admissionProvider) GetImportRoot() string { return "" }
|
||||
func (p *admissionProvider) GetDockerVolumes(name string) []string {
|
||||
if p.volumes == nil {
|
||||
return []string{name + "_data"} // every app is volume-bearing unless told otherwise
|
||||
}
|
||||
return p.volumes[name]
|
||||
}
|
||||
func (p *admissionProvider) StopStack(name string) error {
|
||||
p.stopped = append(p.stopped, name)
|
||||
return nil
|
||||
}
|
||||
func (p *admissionProvider) StartStack(string) error { return nil }
|
||||
func (p *admissionProvider) RefreshAndIsRunning(string) bool { return true }
|
||||
func (p *admissionProvider) GetStackRecoveryInfo(name string) (RecoveryInfo, bool) {
|
||||
p.infoHits = append(p.infoHits, name)
|
||||
return RecoveryInfo{StackDir: filepath.Join(p.dir, "stacks", name)}, true
|
||||
}
|
||||
func (p *admissionProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *admissionProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *admissionProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *admissionProvider) GetStackClassifiedBinds(string) ([]ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
type admissionHarness struct {
|
||||
m *Manager
|
||||
prov *admissionProvider
|
||||
events []unitEvent
|
||||
usage map[string]*UnitSpace
|
||||
dir string
|
||||
logs *bytes.Buffer
|
||||
volDumped []string
|
||||
}
|
||||
|
||||
func newAdmissionHarness(t *testing.T, stacks ...string) *admissionHarness {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
h := &admissionHarness{
|
||||
prov: &admissionProvider{stacks: stacks, dir: dir, hdd: map[string]string{}},
|
||||
usage: map[string]*UnitSpace{},
|
||||
dir: dir,
|
||||
logs: &bytes.Buffer{},
|
||||
}
|
||||
h.m = &Manager{
|
||||
logger: log.New(h.logs, "", 0),
|
||||
systemDataPath: dir,
|
||||
stackProvider: h.prov,
|
||||
unitSpaceFn: func(name string) *UnitSpace { return h.usage[name] },
|
||||
}
|
||||
// The volume-dump seam records the leg that writes the BULK — the one B2 never gated. A refused
|
||||
// app must not reach it.
|
||||
h.m.dumpVolumesSafe = func(name string) error {
|
||||
h.volDumped = append(h.volDumped, name)
|
||||
// Write what the real leg writes, so an ungated call is visible in the tree checksum too.
|
||||
dumpDir := AppVolumeDumpPath(h.nsRoot(), name)
|
||||
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(filepath.Join(dumpDir, name+"_data.tar"), []byte("FRESH TAR FROM THIS RUN"), 0o644)
|
||||
}
|
||||
h.m.SetUnitNotify(func(name string, err error, u *UnitSpace) {
|
||||
h.events = append(h.events, unitEvent{app: name, err: err.Error(), usage: u})
|
||||
})
|
||||
return h
|
||||
}
|
||||
|
||||
// markDisconnected / markDecommissioned put a real settings row behind the drive-state skips, so
|
||||
// Scenario F exercises the production guards rather than a stub of them.
|
||||
func (h *admissionHarness) markDisconnected(app string) {
|
||||
h.driveState(app, true, false)
|
||||
}
|
||||
|
||||
func (h *admissionHarness) markDecommissioned(app string) {
|
||||
h.driveState(app, false, true)
|
||||
}
|
||||
|
||||
func (h *admissionHarness) driveState(app string, disconnected, decommissioned bool) {
|
||||
if h.m.settings == nil {
|
||||
sett, err := settings.Load(filepath.Join(h.dir, "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
h.m.settings = sett
|
||||
}
|
||||
// Each such app gets its OWN drive path, or marking one would skip them all.
|
||||
p := filepath.Join(h.dir, "drives", app)
|
||||
if err := os.MkdirAll(p, 0o755); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
h.prov.hdd[app] = p
|
||||
if err := h.m.settings.AddStoragePath(settings.StoragePath{Path: p, Label: app}); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
if disconnected {
|
||||
if err := h.m.settings.SetDisconnected(p, true, nil); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
}
|
||||
if decommissioned {
|
||||
if err := h.m.settings.SetDecommissioned(p, ""); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (h *admissionHarness) nsRoot() string { return filepath.Join(h.dir, "felhom-data") }
|
||||
|
||||
// setSpace states the filesystem's occupancy as a test INPUT — the whole point of the unitSpaceFn
|
||||
// seam, so no test has to manufacture disk pressure on a real disk.
|
||||
func (h *admissionHarness) setSpace(app string, usedPct, availGB, totalGB float64) {
|
||||
h.usage[app] = &UnitSpace{
|
||||
Path: h.dir, UsedPercent: usedPct, AvailGB: availGB,
|
||||
TotalGB: totalGB, UsedGB: totalGB * usedPct / 100,
|
||||
}
|
||||
}
|
||||
|
||||
// seedUnit writes a previous recovery unit for an app: a manifest, a captured app.yaml, a DB dump and
|
||||
// a volume tar of the given size. The tar is SPARSE (Truncate), so a 2 GiB "previous backup" costs no
|
||||
// disk — the estimator reads st_size, which is what the next write will actually cost.
|
||||
func (h *admissionHarness) seedUnit(t *testing.T, app string, tarBytes int64) {
|
||||
t.Helper()
|
||||
ns := h.nsRoot()
|
||||
for _, d := range []string{
|
||||
RecoveryUnitComposePath(ns, app),
|
||||
AppDBDumpPath(ns, app),
|
||||
AppVolumeDumpPath(ns, app),
|
||||
} {
|
||||
if err := os.MkdirAll(d, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write := func(p string, b []byte, mode os.FileMode) {
|
||||
if err := os.WriteFile(p, b, mode); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
write(RecoveryUnitManifestPath(ns, app), []byte(`{"app_name":"`+app+`","created_at":"2026-08-02T00:00:00Z"}`), 0o644)
|
||||
write(filepath.Join(RecoveryUnitComposePath(ns, app), "app.yaml"), []byte("deployed: true\nenv:\n A: previous-good-value\n"), 0o600)
|
||||
write(filepath.Join(AppDBDumpPath(ns, app), app+"-postgres.sql"), []byte("-- previous good dump\n"), 0o644)
|
||||
|
||||
tar := filepath.Join(AppVolumeDumpPath(ns, app), app+"_data.tar")
|
||||
f, err := os.Create(tar)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := f.WriteString("PREVIOUS GOOD TAR"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if tarBytes > 0 {
|
||||
if err := f.Truncate(tarBytes); err != nil { // sparse — st_size is the estimate, blocks are not spent
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// runOneBackupRun performs exactly the sequence runDBDumpsInternal performs for the two legs that can
|
||||
// be driven without Docker: the admission scope is opened, the volume leg runs, then the capture leg.
|
||||
// The DB leg's wiring is pinned structurally by TestAdmission_IsWiredIntoEveryProductionWriteLeg,
|
||||
// because DiscoverDatabases shells out to `docker` and cannot honestly run here.
|
||||
func (h *admissionHarness) runOneBackupRun() {
|
||||
done := h.m.beginAdmissionRun()
|
||||
defer done()
|
||||
h.m.runVolumeDumps()
|
||||
h.m.captureAllRecoveryUnits()
|
||||
}
|
||||
|
||||
// ── The instrument: a checksum of the whole backup tree ──────────────────────────────────────────
|
||||
|
||||
// treeFingerprint walks every file under backups/primary and returns "relpath mode sha256" lines,
|
||||
// sorted. It is the ONLY honest way to check the refusal's claim: it detects a rewritten payload, an
|
||||
// added file and a deleted one alike, which a log line and an exit code both fail to do.
|
||||
func treeFingerprint(t *testing.T, root string) string {
|
||||
t.Helper()
|
||||
var lines []string
|
||||
err := filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
if fi.IsDir() {
|
||||
return nil
|
||||
}
|
||||
f, err := os.Open(p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
sum := sha256.New()
|
||||
if _, err := io.Copy(sum, f); err != nil {
|
||||
return err
|
||||
}
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
lines = append(lines, fmt.Sprintf("%s %o %d %s", rel, fi.Mode().Perm(), fi.Size(), hex.EncodeToString(sum.Sum(nil))))
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("fingerprinting %s: %v", root, err)
|
||||
}
|
||||
sort.Strings(lines)
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
// treeStatFingerprint is the instrument for trees holding a multi-GiB fixture, where hashing every
|
||||
// byte costs more than it proves: name + mode + SIZE. It still catches the act being tested — the
|
||||
// volume leg replacing a 2 GiB tar with a freshly written one — because that changes the size, and it
|
||||
// catches an added or deleted file by name. Content-identical-but-different-bytes is the one thing it
|
||||
// cannot see, which is why the small-tree tests use treeFingerprint instead.
|
||||
func treeStatFingerprint(t *testing.T, root string) string {
|
||||
t.Helper()
|
||||
var lines []string
|
||||
_ = filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err != nil || fi.IsDir() {
|
||||
return nil
|
||||
}
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
lines = append(lines, fmt.Sprintf("%s %o %d", rel, fi.Mode().Perm(), fi.Size()))
|
||||
return nil
|
||||
})
|
||||
sort.Strings(lines)
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
// treeFileList is the weaker instrument used for Scenario F: names only, so the assertion is
|
||||
// specifically about DELETION and cannot be satisfied or broken by a content change.
|
||||
func treeFileList(t *testing.T, root string) []string {
|
||||
t.Helper()
|
||||
var names []string
|
||||
_ = filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err != nil || fi.IsDir() {
|
||||
return nil
|
||||
}
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
names = append(names, rel)
|
||||
return nil
|
||||
})
|
||||
sort.Strings(names)
|
||||
return names
|
||||
}
|
||||
|
||||
func (h *admissionHarness) primaryRoot() string {
|
||||
return PrimaryBackupPath(h.nsRoot())
|
||||
}
|
||||
|
||||
// ── Scenario A — one decision, taken before the first byte ───────────────────────────────────────
|
||||
|
||||
func TestAdmission_RefusedAppWritesNothingAndIsNotStopped(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "privatebin", "opengist", "homebox")
|
||||
h.setSpace("privatebin", 40, 60, 100)
|
||||
h.setSpace("opengist", 98, 0.4, 70) // below the reserve on BOTH terms
|
||||
h.setSpace("homebox", 40, 60, 100)
|
||||
h.seedUnit(t, "opengist", 0)
|
||||
|
||||
// Scoped to the REFUSED app's own unit: its two siblings are admitted and legitimately write
|
||||
// theirs, so a whole-tree fingerprint would change for the right reason and prove nothing here.
|
||||
// Scenario F below takes the whole-tree view, where every app is refused.
|
||||
refusedUnit := RecoveryUnitPath(h.nsRoot(), "opengist")
|
||||
before := treeFingerprint(t, refusedUnit)
|
||||
if before == "" {
|
||||
t.Fatal("the fixture seeded no previous unit, so 'byte-identical' would be vacuously true")
|
||||
}
|
||||
h.runOneBackupRun()
|
||||
after := treeFingerprint(t, refusedUnit)
|
||||
|
||||
// 1. NOT ONE of the three legs ran for the refused app.
|
||||
for _, got := range h.volDumped {
|
||||
if got == "opengist" {
|
||||
t.Fatal("the VOLUME leg ran for a refused app — this is the R-181 defect exactly: the leg " +
|
||||
"that writes the bulk was never gated, so the reserve it protects was consumed by the " +
|
||||
"very step it exists to bound")
|
||||
}
|
||||
}
|
||||
for _, got := range h.prov.infoHits {
|
||||
if got == "opengist" {
|
||||
t.Fatal("the CAPTURE leg was attempted for a refused app — the verdict must be taken before " +
|
||||
"any write is prepared, not partway through one")
|
||||
}
|
||||
}
|
||||
|
||||
// 2. The tree is byte-identical. This is the assertion the broken code could not pass.
|
||||
if after != before {
|
||||
t.Fatalf("the backup tree CHANGED across a refusal.\n--- before ---\n%s\n--- after ---\n%s\n"+
|
||||
"A refusal that has already rewritten the payload is the defect, not the fix", before, after)
|
||||
}
|
||||
|
||||
// 3. The app was never stopped. DumpAppVolumesSafe stops the stack as its FIRST act, so a gate
|
||||
// placed inside it would bounce the app it is refusing to back up.
|
||||
for _, got := range h.prov.stopped {
|
||||
if got == "opengist" {
|
||||
t.Fatal("the refused app was STOPPED — the reserve check has drifted behind the stop")
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Exactly ONE alert, for that app, carrying the space figures. Three legs must not mean three
|
||||
// emails about one disk.
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1 (one app refused, three legs): %+v", len(h.events), h.events)
|
||||
}
|
||||
if h.events[0].app != "opengist" {
|
||||
t.Fatalf("alert names %q, want opengist", h.events[0].app)
|
||||
}
|
||||
if h.events[0].usage == nil || h.events[0].usage.AvailGB != 0.4 {
|
||||
t.Fatalf("the alert carries no/incorrect space figures: %+v", h.events[0].usage)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario B — the other apps are unaffected ───────────────────────────────────────────────────
|
||||
|
||||
func TestAdmission_SiblingAppsProceedAndOnlyTheRefusedOneAlerts(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "privatebin", "opengist", "homebox")
|
||||
h.setSpace("privatebin", 40, 60, 100)
|
||||
h.setSpace("opengist", 99, 0.2, 70)
|
||||
h.setSpace("homebox", 40, 60, 100)
|
||||
|
||||
h.runOneBackupRun()
|
||||
|
||||
for _, app := range []string{"privatebin", "homebox"} {
|
||||
if !hasStr(h.volDumped, app) {
|
||||
t.Errorf("%s was not volume-dumped (dumped=%v) — one app's refusal silenced its siblings", app, h.volDumped)
|
||||
}
|
||||
if !hasStr(h.prov.infoHits, app) {
|
||||
t.Errorf("%s was not captured (attempted=%v) — the loop did not continue past the refusal", app, h.prov.infoHits)
|
||||
}
|
||||
if _, err := os.Stat(RecoveryUnitManifestPath(h.nsRoot(), app)); err != nil {
|
||||
t.Errorf("%s has no manifest after the run: %v — an admitted app must be backed up normally", app, err)
|
||||
}
|
||||
}
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1: %+v", len(h.events), h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario C — the promise is true ─────────────────────────────────────────────────────────────
|
||||
|
||||
// Every claim the shipped message makes is checked against the tree it describes. The wording is NOT
|
||||
// weakened to fit the behaviour; the behaviour was moved so the wording became true (§8.3).
|
||||
func TestAdmission_EveryClaimInTheRefusalMessageHoldsAgainstTheTree(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
h.setSpace("opengist", 98, 0.5, 70)
|
||||
h.seedUnit(t, "opengist", 0)
|
||||
|
||||
beforeFP := treeFingerprint(t, h.primaryRoot())
|
||||
beforeList := treeFileList(t, h.primaryRoot())
|
||||
h.runOneBackupRun()
|
||||
msg := h.logs.String()
|
||||
|
||||
if !strings.Contains(msg, "REFUSED for opengist") {
|
||||
t.Fatalf("no refusal was logged for opengist; log was:\n%s", msg)
|
||||
}
|
||||
|
||||
// Claim 1: "NO database dump, NO volume dump and NO recovery-unit capture was written for it".
|
||||
for _, claim := range []string{"NO database dump", "NO volume dump", "NO recovery-unit capture"} {
|
||||
if !strings.Contains(msg, claim) {
|
||||
t.Fatalf("the message no longer claims %q — if a leg cannot be brought under the verdict the "+
|
||||
"wording must be narrowed deliberately and the gap named, not dropped silently.\n%s", claim, msg)
|
||||
}
|
||||
}
|
||||
if len(h.volDumped) != 0 || len(h.prov.infoHits) != 0 {
|
||||
t.Fatalf("the message claims no leg ran, but volume=%v capture=%v", h.volDumped, h.prov.infoHits)
|
||||
}
|
||||
|
||||
// Claim 2: "the previous unit is untouched" — the claim that was MEASURED FALSE in R-181.
|
||||
if !strings.Contains(msg, "the previous unit is untouched") {
|
||||
t.Fatalf("the message dropped the untouched claim: %s", msg)
|
||||
}
|
||||
if got := treeFingerprint(t, h.primaryRoot()); got != beforeFP {
|
||||
t.Fatalf("the message says the previous unit is untouched; the tree says otherwise.\n"+
|
||||
"--- before ---\n%s\n--- after ---\n%s", beforeFP, got)
|
||||
}
|
||||
|
||||
// Claim 3: "NOTHING was deleted".
|
||||
if !strings.Contains(msg, "NOTHING was deleted") {
|
||||
t.Fatalf("the message dropped the no-deletion claim: %s", msg)
|
||||
}
|
||||
if got := treeFileList(t, h.primaryRoot()); !equalStrs(got, beforeList) {
|
||||
t.Fatalf("files disappeared across a refusal: before=%v after=%v", beforeList, got)
|
||||
}
|
||||
|
||||
// Claim 4: the reason is named, so the operator can tell which term bound.
|
||||
if !strings.Contains(msg, "headroom") {
|
||||
t.Fatalf("the message does not name WHICH term bound — an operator cannot tell 'the disk is "+
|
||||
"full' from 'this app's backup is too big for what is left':\n%s", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario D — size-aware, not just headroom-aware ─────────────────────────────────────────────
|
||||
|
||||
// The live R-181 sequence, reproduced as a unit: the filesystem is ABOVE the reserve on both terms
|
||||
// when the run reaches the app, and the app's own write is what crosses it. Under B2 this app was
|
||||
// admitted at 96% and then allowed to write 2 GB.
|
||||
func TestAdmission_SizeTermRefusesAnAppWhoseOwnWriteWouldCrossTheReserve(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
// 96% used of 70 GiB, 3.0 GiB free — BOTH reserve terms deliberately still clear (97% / 1.0 GiB),
|
||||
// exactly as on demo-hp at 06:40:03, so a headroom-only rule starts the run.
|
||||
h.setSpace("opengist", 96, 3.0, 70)
|
||||
if _, r := h.m.floorVerdict(h.usage["opengist"], 0); r != floorAdmit {
|
||||
t.Fatalf("fixture is wrong: the headroom term already refuses (%v), so this test would pass "+
|
||||
"without a size term and prove nothing", r)
|
||||
}
|
||||
h.seedUnit(t, "opengist", 2<<30) // its last backup was 2 GiB — the figure measured live
|
||||
|
||||
before := treeStatFingerprint(t, h.primaryRoot())
|
||||
h.runOneBackupRun()
|
||||
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want 1 — the app was admitted at 96%% and would have been allowed to "+
|
||||
"write 2 GiB, which is the R-181 sequence: %+v", len(h.events), h.events)
|
||||
}
|
||||
if !strings.Contains(h.logs.String(), "(size)") {
|
||||
t.Fatalf("the refusal was not attributed to the SIZE term:\n%s", h.logs.String())
|
||||
}
|
||||
if !strings.Contains(h.events[0].err, "last backup was 2.0 GB") {
|
||||
t.Fatalf("the alert does not carry the estimate that produced the refusal: %q", h.events[0].err)
|
||||
}
|
||||
if len(h.volDumped) != 0 {
|
||||
t.Fatalf("the volume leg ran anyway: %v", h.volDumped)
|
||||
}
|
||||
if got := treeStatFingerprint(t, h.primaryRoot()); got != before {
|
||||
t.Fatalf("the tree changed despite the size-term refusal.\nbefore=%s\nafter =%s", before, got)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario E — a first-ever backup is not blocked by having no history ─────────────────────────
|
||||
|
||||
func TestAdmission_FirstEverBackupIsAdmitted(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "brandnew")
|
||||
h.setSpace("brandnew", 40, 600, 1000) // ample room, and NO previous unit on disk
|
||||
|
||||
if est, ok := h.m.estimatedWriteBytes("brandnew"); ok || est != 0 {
|
||||
t.Fatalf("estimatedWriteBytes = (%v, %v) for an app with no history, want (0, false)", est, ok)
|
||||
}
|
||||
h.runOneBackupRun()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("a brand-new app was refused: %+v — refusing every app that has no size to estimate "+
|
||||
"from would make the FIRST backup the one that can never happen", h.events)
|
||||
}
|
||||
if !hasStr(h.volDumped, "brandnew") || !hasStr(h.prov.infoHits, "brandnew") {
|
||||
t.Fatalf("the app was not backed up (volume=%v capture=%v)", h.volDumped, h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Scenario F — the reserve still never deletes ─────────────────────────────────────────────────
|
||||
|
||||
func TestAdmission_NothingUnderBackupsIsEverRemoved(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "privatebin", "opengist", "homebox")
|
||||
for _, app := range []string{"privatebin", "opengist", "homebox"} {
|
||||
h.setSpace(app, 99, 0.1, 70) // every app refused — maximum pressure to "make room"
|
||||
h.seedUnit(t, app, 0)
|
||||
}
|
||||
|
||||
before := treeFileList(t, h.primaryRoot())
|
||||
h.runOneBackupRun()
|
||||
after := treeFileList(t, h.primaryRoot())
|
||||
|
||||
if !equalStrs(before, after) {
|
||||
t.Fatalf("the file list changed under the reserve.\nbefore=%v\nafter =%v\n"+
|
||||
"Nothing here is generational — a unit is ONE fixed path per app — so 'prune the oldest' "+
|
||||
"could only mean destroying a DIFFERENT app's only local recovery unit", before, after)
|
||||
}
|
||||
if len(before) == 0 {
|
||||
t.Fatal("the fixture seeded no files, so this test would pass against code that deleted everything")
|
||||
}
|
||||
}
|
||||
|
||||
// ── §8.1 — one verdict per app per run, and it resets between runs ───────────────────────────────
|
||||
|
||||
// The verdict must not be re-taken between an app's own legs. Re-deciding is how the split this fixes
|
||||
// came about: DB leg admitted, volume leg admitted, capture refused — with the bulk already written.
|
||||
func TestAdmission_VerdictIsTakenOncePerAppPerRunAndNotRedecidedBetweenLegs(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
reads := 0
|
||||
h.m.unitSpaceFn = func(string) *UnitSpace {
|
||||
reads++
|
||||
if reads == 1 {
|
||||
return &UnitSpace{Path: h.dir, UsedPercent: 99, AvailGB: 0.1, TotalGB: 70, UsedGB: 69.3}
|
||||
}
|
||||
// The disk "recovers" mid-run. A re-decided verdict would admit the capture leg here — which
|
||||
// is precisely the split R-181 closes, arriving from the other direction.
|
||||
return &UnitSpace{Path: h.dir, UsedPercent: 10, AvailGB: 60, TotalGB: 70, UsedGB: 7}
|
||||
}
|
||||
|
||||
h.runOneBackupRun()
|
||||
|
||||
if reads != 1 {
|
||||
t.Fatalf("the filesystem was read %d times for ONE app in ONE run — the verdict is being "+
|
||||
"re-decided between legs, which reintroduces the split (bulk written, capture refused)", reads)
|
||||
}
|
||||
if len(h.prov.infoHits) != 0 {
|
||||
t.Fatal("the capture leg ran after the app was refused earlier in the same run")
|
||||
}
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1 per app per run: %+v", len(h.events), h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// A set carried between runs is a wrong answer with a confident face: tonight's question answered
|
||||
// with last night's disk.
|
||||
func TestAdmission_TheRememberedSetResetsBetweenRuns(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
h.setSpace("opengist", 99, 0.1, 70)
|
||||
h.runOneBackupRun()
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("run 1: want 1 alert, got %+v", h.events)
|
||||
}
|
||||
|
||||
h.setSpace("opengist", 20, 55, 70) // space freed between runs
|
||||
h.runOneBackupRun()
|
||||
|
||||
if !hasStr(h.volDumped, "opengist") {
|
||||
t.Fatal("the second run still refused the app — the previous run's verdict was carried over, " +
|
||||
"so freeing space could never take effect")
|
||||
}
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("the second (admitted) run alerted again: %+v", h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// ── §8.4 — a nil reading neither refuses nor warns, across ALL THREE legs ────────────────────────
|
||||
|
||||
// Unchanged behaviour, re-pinned because the decision now governs three legs instead of one: an
|
||||
// unreadable filesystem must not silently stop an app being backed up at all.
|
||||
func TestAdmission_UnreadableFilesystemAdmitsEveryLegAndDoesNotWarn(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
// No usage entry → the reader returns nil, which is what system.GetDiskUsage does on error.
|
||||
|
||||
h.runOneBackupRun()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("an unreadable filesystem produced %d alert(s): %+v — that is the drive gate's "+
|
||||
"business and has its own alert", len(h.events), h.events)
|
||||
}
|
||||
if !hasStr(h.volDumped, "opengist") {
|
||||
t.Fatal("the VOLUME leg was refused on an unreadable read — a drive that merely blipped would " +
|
||||
"now stop the bulk of the backup, not just the capture")
|
||||
}
|
||||
if !hasStr(h.prov.infoHits, "opengist") {
|
||||
t.Fatal("the CAPTURE leg was refused on an unreadable read")
|
||||
}
|
||||
}
|
||||
|
||||
// ── The estimator, through the production path (no seam) ─────────────────────────────────────────
|
||||
|
||||
func TestEstimatedWriteBytes_SumsTheAppsPreviousDumpsFromRealFiles(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "opengist")
|
||||
h.seedUnit(t, "opengist", 3<<30) // 3 GiB sparse tar + a small .sql
|
||||
|
||||
est, ok := h.m.estimatedWriteBytes("opengist")
|
||||
if !ok {
|
||||
t.Fatal("history on disk was not recognised as history")
|
||||
}
|
||||
if est < 3<<30 || est > (3<<30)+4096 {
|
||||
t.Fatalf("estimate = %d B, want ~%d (the .tar plus the small .sql)", est, int64(3)<<30)
|
||||
}
|
||||
|
||||
// An app whose unit exists but holds no dumps yet is history-LESS, not a zero-byte estimate.
|
||||
other := AppVolumeDumpPath(h.nsRoot(), "empty")
|
||||
if err := os.MkdirAll(other, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if est, ok := h.m.estimatedWriteBytes("empty"); ok || est != 0 {
|
||||
t.Fatalf("an empty unit reported history (%v, %v) — an absent dump is not a 0-byte one", est, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// ── The seam is WIRED — walked as an AST, not grepped ────────────────────────────────────────────
|
||||
|
||||
// FOUR mechanisms in this project have been built and left disconnected (REUSE.md's seam register).
|
||||
// The behavioural tests above drive the two legs that can run without Docker; the DB leg cannot, so
|
||||
// its gate is pinned HERE, structurally. `strings.Contains` is deliberately not used: a commented-out
|
||||
// call still contains the string, and so does a call inside dead code.
|
||||
func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
file, err := parser.ParseFile(fset, "backup.go", nil, 0) // comments dropped — only real calls survive
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
calls := map[string][]string{} // enclosing func → called names, in source order
|
||||
var current string
|
||||
ast.Inspect(file, func(n ast.Node) bool {
|
||||
switch v := n.(type) {
|
||||
case *ast.FuncDecl:
|
||||
current = v.Name.Name
|
||||
case *ast.CallExpr:
|
||||
name := ""
|
||||
switch fn := v.Fun.(type) {
|
||||
case *ast.Ident:
|
||||
name = fn.Name
|
||||
case *ast.SelectorExpr:
|
||||
name = fn.Sel.Name
|
||||
}
|
||||
if name != "" && current != "" {
|
||||
calls[current] = append(calls[current], name)
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
// 1. The run scope is opened by the orchestrator of all three legs.
|
||||
if !hasStr(calls["runDBDumpsInternal"], "beginAdmissionRun") {
|
||||
t.Fatal("runDBDumpsInternal does not open the admission scope — without it every leg decides " +
|
||||
"independently and the per-run memo never exists, which is the pre-R-181 behaviour")
|
||||
}
|
||||
|
||||
// 2. The DB leg consults it BEFORE the dump. Order is the whole point: a gate after the write is
|
||||
// the defect, relocated.
|
||||
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "DumpOne",
|
||||
"the DATABASE leg dumps before consulting the reserve")
|
||||
|
||||
// 3. The volume leg consults it BEFORE the dump seam — which stops the stack as its first act.
|
||||
assertGateBefore(t, calls["runVolumeDumps"], "admitApp", "dump",
|
||||
"the VOLUME leg — the one that writes the bulk, and the one B2 never gated — dumps before "+
|
||||
"consulting the reserve")
|
||||
|
||||
// 4. The capture leg, in its own file.
|
||||
rfset := token.NewFileSet()
|
||||
rfile, err := parser.ParseFile(rfset, "recovery_unit.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
capCalls := map[string][]string{}
|
||||
current = ""
|
||||
ast.Inspect(rfile, func(n ast.Node) bool {
|
||||
switch v := n.(type) {
|
||||
case *ast.FuncDecl:
|
||||
current = v.Name.Name
|
||||
case *ast.CallExpr:
|
||||
if sel, ok := v.Fun.(*ast.SelectorExpr); ok && current != "" {
|
||||
capCalls[current] = append(capCalls[current], sel.Sel.Name)
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "CaptureRecoveryUnit",
|
||||
"the CAPTURE leg captures before consulting the reserve")
|
||||
}
|
||||
|
||||
// assertGateBefore checks that `gate` appears in the call list before `act`.
|
||||
func assertGateBefore(t *testing.T, calls []string, gate, act, why string) {
|
||||
t.Helper()
|
||||
gi, ai := -1, -1
|
||||
for i, c := range calls {
|
||||
if c == gate && gi < 0 {
|
||||
gi = i
|
||||
}
|
||||
if c == act && ai < 0 {
|
||||
ai = i
|
||||
}
|
||||
}
|
||||
if gi < 0 {
|
||||
t.Fatalf("%s: %q is never called there at all (calls=%v)", why, gate, calls)
|
||||
}
|
||||
if ai < 0 {
|
||||
t.Fatalf("fixture drift: %q is no longer called in that function (calls=%v) — this test can no "+
|
||||
"longer see the act it is ordering the gate against", act, calls)
|
||||
}
|
||||
if gi > ai {
|
||||
t.Fatalf("%s: %q first appears at %d, after %q at %d", why, gate, gi, act, ai)
|
||||
}
|
||||
}
|
||||
|
||||
func hasStr(hay []string, needle string) bool {
|
||||
for _, s := range hay {
|
||||
if s == needle {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func equalStrs(a, b []string) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for i := range a {
|
||||
if a[i] != b[i] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
@@ -100,6 +100,12 @@ func ParseComposeImages(composePath string) []string {
|
||||
return appbackup.ParseComposeImages(composePath)
|
||||
}
|
||||
|
||||
// DBServiceNames forwards to appbackup.DBServiceNames — the compose SERVICE names holding a database,
|
||||
// i.e. the argument list for the DB-only bring-up both restore paths use before a dump replay (R-47).
|
||||
func DBServiceNames(composePath string) ([]string, error) {
|
||||
return appbackup.DBServiceNames(composePath)
|
||||
}
|
||||
|
||||
// humanizeBytes forwards to appbackup.HumanizeBytes; kept unexported so the
|
||||
// many in-package call sites (backup.go, crossdrive.go, restore code) need no edit.
|
||||
func humanizeBytes(b int64) string {
|
||||
@@ -115,6 +121,11 @@ func NamespaceRoot(drivePath string, inGuestDrive bool) string {
|
||||
return appbackup.NamespaceRoot(drivePath, inGuestDrive)
|
||||
}
|
||||
|
||||
// NamespaceRootFor re-exports the ONE drive-kind-aware resolver (R-203).
|
||||
func NamespaceRootFor(drivePath, systemDataPath string) string {
|
||||
return appbackup.NamespaceRootFor(drivePath, systemDataPath)
|
||||
}
|
||||
|
||||
func PrimaryBackupPath(nsRoot string) string {
|
||||
return appbackup.PrimaryBackupPath(nsRoot)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,229 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-174 — the app-stop guard's crash recovery must not start an app onto a MISSING drive.
|
||||
//
|
||||
// The defect these pin, found by review on 2026-08-02 in code shipped 2026-08-01 (v0.189.0):
|
||||
// `appStopGuard.SetStarter(stackMgr)` handed Recover the raw stack manager, whose `StartStack` has
|
||||
// no drive gate. Recover runs at STARTUP — exactly when an external drive may not have come back —
|
||||
// so a backup that stopped an app, followed by a power cut and a drive that did not remount, ended
|
||||
// with the app started onto a missing drive. R-171 one path over.
|
||||
//
|
||||
// THE SEAM UNDER TEST IS THE STARTER, not the gate: `internal/backup` must not import `stacks` or
|
||||
// `settings`, so the production gate lives in `cmd/controller`. What is pinned here is the contract
|
||||
// between them — that a starter returning ErrStartRefused produces a REFUSAL (marker kept, no alarm)
|
||||
// and not a FAILURE. The production wiring itself is pinned by TestMainWiresGatedAppStopStarter.
|
||||
|
||||
// gatingStarter is a starter whose gate refuses a named set of apps, in the shape the production
|
||||
// `gatedAppStopStarter` uses: refuse BEFORE calling through, and wrap ErrStartRefused with a reason.
|
||||
type gatingStarter struct {
|
||||
inner *fakeStarter
|
||||
refuse map[string]string // app → reason
|
||||
refused []string
|
||||
}
|
||||
|
||||
func (s *gatingStarter) StartStack(name string) error {
|
||||
if why, ok := s.refuse[name]; ok {
|
||||
s.refused = append(s.refused, name)
|
||||
return fmt.Errorf("%w: %s", ErrStartRefused, why)
|
||||
}
|
||||
return s.inner.StartStack(name)
|
||||
}
|
||||
|
||||
func newGatedGuard(t *testing.T, dir string, refuse map[string]string) (*AppStopGuard, *gatingStarter) {
|
||||
t.Helper()
|
||||
s := &gatingStarter{inner: &fakeStarter{}, refuse: refuse}
|
||||
g := NewAppStopGuard(filepath.Join(dir, "appstop-state.json"), log.New(io.Discard, "", 0))
|
||||
g.SetStarter(s)
|
||||
return g, s
|
||||
}
|
||||
|
||||
// --- Scenario A — the guard does not start an app onto a missing drive ---------------------------
|
||||
|
||||
func TestRecover_DriveAbsent_RefusesTheStartAndKEEPSTheMarker(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
|
||||
// process 1: a volume dump stops immich, then the box loses power. No End(), no defer.
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatalf("Begin: %v", err)
|
||||
}
|
||||
// <power cut> — and immich's drive does NOT come back.
|
||||
|
||||
// process 2: a fresh controller starts. The drive is absent.
|
||||
g2, starter := newGatedGuard(t, dir, map[string]string{
|
||||
"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint",
|
||||
})
|
||||
res := g2.Recover()
|
||||
|
||||
if len(starter.inner.starts) != 0 {
|
||||
t.Fatalf("started %v — the app was started onto a MISSING drive, which is the whole defect",
|
||||
starter.inner.starts)
|
||||
}
|
||||
if res == nil {
|
||||
t.Fatal("Recover returned nil — the refusal is invisible to the caller, so nothing can report it")
|
||||
}
|
||||
if len(res.Refused) != 1 || res.Refused[0] != "immich" {
|
||||
t.Fatalf("refused=%v, want [immich]", res.Refused)
|
||||
}
|
||||
if len(res.Failed) != 0 {
|
||||
t.Fatalf("failed=%v — a deliberate hold was recorded as a FAILURE. That bucket reaches "+
|
||||
"NotifyBackupFailed, which is customer-enabled by default, so the customer would be "+
|
||||
"emailed \"A biztonsági mentés sikertelen!\" about an app nothing is wrong with (R-171's "+
|
||||
"false-alarm shape one path over)", res.Failed)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was CLEARED after a refused start — the operation is genuinely " +
|
||||
"unfinished, and clearing it erases the only durable record that immich is owed a restart")
|
||||
}
|
||||
// The refusal must name the app AND the reason, or an operator cannot act on it.
|
||||
if d := res.Detail(); !strings.Contains(d, "held_by_drive") || !strings.Contains(d, "immich") {
|
||||
t.Fatalf("detail %q does not name the held app", d)
|
||||
}
|
||||
if msg := res.Message(); !strings.Contains(msg, "HELD") || !strings.Contains(msg, "drive") {
|
||||
t.Fatalf("operator message %q does not say the app is held by an absent drive", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// A refusal-only recovery MUST NOT alarm. This is the assertion that keeps the fix from being the
|
||||
// bug it fixes: the drive gate doing its job is not a backup failure.
|
||||
func TestRecover_RefusalOnly_IsNotAlarming(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g2, _ := newGatedGuard(t, dir, map[string]string{"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint"})
|
||||
res := g2.Recover()
|
||||
|
||||
if res.Alarming() {
|
||||
t.Fatal("a recovery that only REFUSED starts reports as alarming — main.go would push it " +
|
||||
"through NotifyBackupFailed and email the customer about a working drive gate")
|
||||
}
|
||||
}
|
||||
|
||||
// A genuine failure alongside a refusal still alarms, and the two stay in different buckets.
|
||||
func TestRecover_FailureAlongsideRefusal_StillAlarmsAndKeepsThemApart(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:batch", ReasonVolumeDump, []string{"immich", "nextcloud", "homebox"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
g2, starter := newGatedGuard(t, dir, map[string]string{"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint"})
|
||||
starter.inner.failWith = map[string]error{"nextcloud": errors.New("compose up: no such image")}
|
||||
res := g2.Recover()
|
||||
|
||||
if len(res.Refused) != 1 || res.Refused[0] != "immich" {
|
||||
t.Fatalf("refused=%v, want [immich]", res.Refused)
|
||||
}
|
||||
if len(res.Failed) != 1 || res.Failed[0] != "nextcloud" {
|
||||
t.Fatalf("failed=%v, want [nextcloud]", res.Failed)
|
||||
}
|
||||
if len(res.Restarted) != 1 || res.Restarted[0] != "homebox" {
|
||||
t.Fatalf("restarted=%v, want [homebox] — neither a refusal nor a failure may abort the loop",
|
||||
res.Restarted)
|
||||
}
|
||||
if !res.Alarming() {
|
||||
t.Fatal("a genuine restart FAILURE alongside a refusal no longer alarms — the refusal " +
|
||||
"swallowed a real fault")
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared with work still owed")
|
||||
}
|
||||
// The message must not let the held app inflate the failure count.
|
||||
msg := res.Message()
|
||||
if !strings.Contains(msg, "1 of 2 app(s) could NOT be restarted") {
|
||||
t.Fatalf("operator message %q miscounts: the held app must not be counted as a failure", msg)
|
||||
}
|
||||
if !strings.Contains(msg, "not counted as failures") {
|
||||
t.Fatalf("operator message %q does not disclose the held app at all", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario B — a live drive still recovers normally, byte-identical to before -----------------
|
||||
|
||||
func TestRecover_DriveLive_RecoversExactlyAsBefore(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich", "nextcloud"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Nothing refused — the gate says yes for both.
|
||||
g2, starter := newGatedGuard(t, dir, nil)
|
||||
res := g2.Recover()
|
||||
|
||||
if len(starter.inner.starts) != 2 {
|
||||
t.Fatalf("started %v, want both apps — the new gate refused a LEGITIMATE recovery",
|
||||
starter.inner.starts)
|
||||
}
|
||||
if len(res.Refused) != 0 || len(res.Failed) != 0 {
|
||||
t.Fatalf("refused=%v failed=%v, want neither on a live drive", res.Refused, res.Failed)
|
||||
}
|
||||
if len(res.Restarted) != 2 {
|
||||
t.Fatalf("restarted=%v, want both", res.Restarted)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker survived a fully successful recovery — the next boot would restart the apps again")
|
||||
}
|
||||
if !res.Alarming() {
|
||||
t.Fatal("a successful recovery no longer reports to the operator — the interrupted operation " +
|
||||
"itself is what §2.4 wants reported, and it went silent")
|
||||
}
|
||||
}
|
||||
|
||||
// The next startup, with the drive back, completes the recovery and clears the marker. This is what
|
||||
// makes "keep the marker" a recovery rather than a leak.
|
||||
func TestRecover_HeldAppIsRestartedOnceTheDriveReturns(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGatedGuard(t, dir, nil)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Boot 1 — drive absent: refused, marker kept.
|
||||
g2, _ := newGatedGuard(t, dir, map[string]string{"immich": "drive /mnt/felhom-drives/hdd_1 is not a live mountpoint"})
|
||||
if res := g2.Recover(); len(res.Refused) != 1 {
|
||||
t.Fatalf("boot 1 refused=%v, want [immich]", res.Refused)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("boot 1 cleared the marker — boot 2 has nothing to act on and immich stays down forever")
|
||||
}
|
||||
|
||||
// Boot 2 — the drive is back.
|
||||
g3, starter := newGatedGuard(t, dir, nil)
|
||||
res := g3.Recover()
|
||||
if len(starter.inner.starts) != 1 || starter.inner.starts[0] != "immich" {
|
||||
t.Fatalf("boot 2 started %v, want [immich] — the held app was never picked up again",
|
||||
starter.inner.starts)
|
||||
}
|
||||
if len(res.Restarted) != 1 {
|
||||
t.Fatalf("boot 2 restarted=%v, want [immich]", res.Restarted)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("boot 2 kept the marker after a fully successful recovery")
|
||||
}
|
||||
}
|
||||
|
||||
// ErrStartRefused must be matched with errors.Is, i.e. it survives wrapping. A starter that returns
|
||||
// a bare string reason would land in Failed and alarm — the exact collapse this type prevents.
|
||||
func TestErrStartRefused_SurvivesWrapping(t *testing.T) {
|
||||
err := fmt.Errorf("%w: drive /mnt/felhom-drives/hdd_1 is not a live mountpoint", ErrStartRefused)
|
||||
if !errors.Is(err, ErrStartRefused) {
|
||||
t.Fatal("a wrapped ErrStartRefused is no longer matched by errors.Is — every refusal would " +
|
||||
"be recorded as a restart failure and alarm the customer")
|
||||
}
|
||||
if errors.Is(errors.New("compose up: no such image"), ErrStartRefused) {
|
||||
t.Fatal("an ordinary restart failure matches ErrStartRefused — real faults would go silent")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,365 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ── The app-stop marker (R-166 part 2, decision D-b "in-flight operations") ───────────────────────
|
||||
//
|
||||
// Several operations stop a customer's app, do something to its data, and start it again. Between
|
||||
// the stop and the start, NOTHING ON DISK RECORDED THAT AN APP WAS OWED A RESTART. A controller that
|
||||
// died in that window left the app down with no explanation anywhere — and because a stopped app has
|
||||
// zero containers, the boot reconciler read it as a deliberate customer stop and deliberately left
|
||||
// it alone. Silently, indefinitely.
|
||||
//
|
||||
// A `defer` is NOT the fix and must never be described as one. Campaign 8 fault 10 established this
|
||||
// on live hardware: a SIGKILL runs no deferred function, and what brought the quiesce loop's stacks
|
||||
// back was its persisted marker read by Recover() one second after restart. The defer covers the
|
||||
// graceful exits; the marker covers the hard crash and the power cut. This file is that marker for
|
||||
// the app-data path, modelled directly on internal/quiesce's.
|
||||
//
|
||||
// WHY ITS OWN FILE, not quiesce's: one file, one writer. Quiesce's marker records a whole-guest
|
||||
// backup window and is written by the quiesce loop; this one records an app-data operation and is
|
||||
// written by the backup manager and the exporter. Sharing the file would give it two writers with
|
||||
// two lifetimes, and one clearing the other's record is a stranded app by a different route.
|
||||
//
|
||||
// SAFETY (D-b's binding rule): losing this file must never be worse than not having it. A lost or
|
||||
// corrupt marker means the app is not auto-restarted by THIS mechanism — which is precisely the
|
||||
// pre-v0.189.0 position, not a new hazard. It never deletes, restores, or touches a backup artifact.
|
||||
|
||||
// AppStopReason names WHY an app was stopped, so the recovery log tells an operator which operation
|
||||
// was interrupted rather than merely that something was.
|
||||
type AppStopReason string
|
||||
|
||||
const (
|
||||
// ReasonVolumeDump — DumpAppVolumesSafe: stop, tar the volumes consistently, start.
|
||||
ReasonVolumeDump AppStopReason = "volume_dump"
|
||||
// ReasonOffboxReconstitute — a full offsite restore overwriting the app's files.
|
||||
ReasonOffboxReconstitute AppStopReason = "offbox_reconstitute"
|
||||
// ReasonAppExport — a .fab export taken with "stop the app first".
|
||||
ReasonAppExport AppStopReason = "app_export"
|
||||
)
|
||||
|
||||
// humanReason is the operator-facing phrasing for each reason.
|
||||
func (r AppStopReason) humanReason() string {
|
||||
switch r {
|
||||
case ReasonVolumeDump:
|
||||
return "an app-data backup (volume dump)"
|
||||
case ReasonOffboxReconstitute:
|
||||
return "an off-site restore"
|
||||
case ReasonAppExport:
|
||||
return "an app export"
|
||||
default:
|
||||
return string(r)
|
||||
}
|
||||
}
|
||||
|
||||
// AppStopMarker is the persisted "these apps were stopped by an operation that has not reported
|
||||
// finishing — they are owed a restart" note.
|
||||
type AppStopMarker struct {
|
||||
Active bool `json:"active"`
|
||||
OpID string `json:"op_id"`
|
||||
Reason AppStopReason `json:"reason"`
|
||||
Stacks []string `json:"stacks"`
|
||||
StartedAt time.Time `json:"started_at"`
|
||||
}
|
||||
|
||||
// AppStopStarter is the one thing recovery needs: the ability to start a stack. StartStack must be
|
||||
// idempotent (it is — `compose up -d` on a running stack is a no-op).
|
||||
//
|
||||
// R-174: production MUST pass a GATED starter, never the raw stack manager. Recover runs at STARTUP —
|
||||
// exactly when an external drive may not have come back — and `Manager.StartStack` has no drive gate
|
||||
// of its own. See `gatedAppStopStarter` in cmd/controller/main.go.
|
||||
type AppStopStarter interface {
|
||||
StartStack(name string) error
|
||||
}
|
||||
|
||||
// ErrStartRefused is what a gated starter returns when a DELIBERATE HOLDER — today the drive gate —
|
||||
// says an app must not be started. Wrap it (`fmt.Errorf("%w: …", ErrStartRefused)`) so the reason
|
||||
// survives; Recover matches with errors.Is.
|
||||
//
|
||||
// IT IS NOT A FAILURE, AND THE DISTINCTION IS THE WHOLE POINT OF THE TYPE. A refusal means the
|
||||
// holder is doing its job and owns the restart; a failure means the restart was attempted and broke.
|
||||
// Collapsing the two would put a deliberately-held app into `Failed`, which main.go reports through
|
||||
// `NotifyBackupFailed` — a type that is customer-enabled by default (`settings.DefaultEnabledEvents`)
|
||||
// and carries the Hungarian "A biztonsági mentés sikertelen!". That is R-171's defect one path over:
|
||||
// a false alarm about an app the drive gate is deliberately holding. Both buckets keep the marker;
|
||||
// only `Failed` alarms.
|
||||
var ErrStartRefused = errors.New("start refused by a deliberate holder")
|
||||
|
||||
// AppStopGuard owns one marker file. Construct with NewAppStopGuard; the zero value is inert (every
|
||||
// method is a no-op on a nil guard), so a caller that was never wired degrades to pre-v0.189.0
|
||||
// behaviour instead of panicking.
|
||||
type AppStopGuard struct {
|
||||
path string
|
||||
logger *log.Logger
|
||||
now func() time.Time
|
||||
// starter is only needed by Recover; Begin/End work without one.
|
||||
starter AppStopStarter
|
||||
}
|
||||
|
||||
// AppStopRecovery is what Recover found and did. Returned rather than pushed through a notifier
|
||||
// seam, because of a hard ordering constraint: Recover must COMPLETE before the boot reconciler is
|
||||
// launched (§8.4, main.go:236) and the hub notifier is not constructed until main.go:307. A seam
|
||||
// wired after the fact would be a seam that never fires — the "built but never wired" shape this
|
||||
// project has now hit four times. Returning the outcome lets main.go report it the moment the
|
||||
// notifier exists, and makes the reporting decision visible at the call site instead of buried here.
|
||||
type AppStopRecovery struct {
|
||||
Reason AppStopReason
|
||||
OpID string
|
||||
StartedAt time.Time
|
||||
Restarted []string // apps started again by this recovery
|
||||
Failed []string // apps whose restart was ATTEMPTED and broke (the marker was kept for these)
|
||||
// Refused are apps a deliberate holder said must not start — today, an absent data drive
|
||||
// (R-174). The marker is kept for these too, but they are NOT a fault and MUST NOT alarm: the
|
||||
// holder owns the restart. Separate from Failed for the reason recorded on ErrStartRefused.
|
||||
Refused []string
|
||||
}
|
||||
|
||||
// Alarming reports whether this recovery is worth paging an operator about. A recovery that only
|
||||
// REFUSED starts is the drive gate working as designed, and reporting it through the customer-enabled
|
||||
// `backup_failed` type would be the R-171 false alarm one path over.
|
||||
func (r *AppStopRecovery) Alarming() bool {
|
||||
if r == nil {
|
||||
return false
|
||||
}
|
||||
return len(r.Failed) > 0 || len(r.Restarted) > 0
|
||||
}
|
||||
|
||||
// Message is the operator-facing headline for an interrupted operation.
|
||||
func (r *AppStopRecovery) Message() string {
|
||||
if r == nil {
|
||||
return ""
|
||||
}
|
||||
if len(r.Failed) > 0 {
|
||||
m := fmt.Sprintf("%s was interrupted by a controller restart and %d of %d app(s) could NOT be restarted",
|
||||
r.Reason.humanReason(), len(r.Failed), len(r.Restarted)+len(r.Failed))
|
||||
if len(r.Refused) > 0 {
|
||||
m += fmt.Sprintf(" (a further %d are held by an absent drive and are not counted as failures)", len(r.Refused))
|
||||
}
|
||||
return m
|
||||
}
|
||||
if len(r.Refused) > 0 && len(r.Restarted) == 0 {
|
||||
return fmt.Sprintf("%s was interrupted by a controller restart — %d app(s) are left stopped and HELD: their data drive is not available, so the drive gate restarts them when it returns",
|
||||
r.Reason.humanReason(), len(r.Refused))
|
||||
}
|
||||
m := fmt.Sprintf("%s was interrupted by a controller restart — %d app(s) were left stopped and have been restarted",
|
||||
r.Reason.humanReason(), len(r.Restarted))
|
||||
if len(r.Refused) > 0 {
|
||||
m += fmt.Sprintf("; %d more are held by an absent drive", len(r.Refused))
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Detail is the machine-readable tail. App/stack NAMES only — never env values (§9.5).
|
||||
func (r *AppStopRecovery) Detail() string {
|
||||
if r == nil {
|
||||
return ""
|
||||
}
|
||||
d := fmt.Sprintf("op=%s reason=%s started_at=%s restarted=%v", r.OpID, r.Reason,
|
||||
r.StartedAt.UTC().Format(time.RFC3339), r.Restarted)
|
||||
if len(r.Failed) > 0 {
|
||||
d += fmt.Sprintf(" restart_failed=%v", r.Failed)
|
||||
}
|
||||
if len(r.Refused) > 0 {
|
||||
d += fmt.Sprintf(" held_by_drive=%v", r.Refused)
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
// NewAppStopGuard builds a guard over the given marker path.
|
||||
func NewAppStopGuard(path string, logger *log.Logger) *AppStopGuard {
|
||||
if logger == nil {
|
||||
logger = log.Default()
|
||||
}
|
||||
return &AppStopGuard{path: path, logger: logger, now: time.Now}
|
||||
}
|
||||
|
||||
// SetStarter wires the stack-start seam used by Recover. INIT-ONLY — call once at startup, before
|
||||
// Recover. Separate from the constructor because the guard is built alongside the backup manager,
|
||||
// which learns its stack provider later (the same shape as SetStackProvider).
|
||||
func (g *AppStopGuard) SetStarter(s AppStopStarter) {
|
||||
if g == nil {
|
||||
return
|
||||
}
|
||||
g.starter = s
|
||||
}
|
||||
|
||||
// Begin records that `stacks` are about to be stopped by `reason`. It MUST be called BEFORE the
|
||||
// first stop — an error here means the marker could not be written, and the caller must not proceed
|
||||
// to stop an app it cannot promise to restart.
|
||||
func (g *AppStopGuard) Begin(opID string, reason AppStopReason, stackNames []string) error {
|
||||
if g == nil || g.path == "" {
|
||||
return nil // not wired — pre-v0.189.0 behaviour, never a hard failure
|
||||
}
|
||||
if len(stackNames) == 0 {
|
||||
return nil
|
||||
}
|
||||
return g.write(AppStopMarker{
|
||||
Active: true,
|
||||
OpID: opID,
|
||||
Reason: reason,
|
||||
Stacks: append([]string(nil), stackNames...),
|
||||
StartedAt: g.now(),
|
||||
})
|
||||
}
|
||||
|
||||
// End clears the marker after a successful restart. Best-effort by contract: a failure to clear is
|
||||
// logged, never returned as the operation's error — a stale marker costs one idempotent StartStack
|
||||
// on the next boot, which is exactly D-b's "worst acceptable outcome" and far cheaper than failing
|
||||
// a backup that actually succeeded.
|
||||
func (g *AppStopGuard) End() {
|
||||
if g == nil || g.path == "" {
|
||||
return
|
||||
}
|
||||
if err := os.Remove(g.path); err != nil && !os.IsNotExist(err) {
|
||||
g.logger.Printf("[ERROR] [appstop] could not clear the app-stop marker at %s: %v (a stale marker costs one idempotent restart at next startup)", g.path, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Recover restarts any apps left stopped by an operation that died before restarting them, then
|
||||
// clears the marker. Call ONCE at startup, and — critically — call it to COMPLETION before the boot
|
||||
// reconciler is launched, so an app this marker explains is not also reported as an unexplained boot
|
||||
// orphan (§8.4).
|
||||
//
|
||||
// Idempotent: StartStack on a running stack is tolerated, and an absent or inactive marker is a
|
||||
// no-op. On a restart FAILURE the marker is deliberately LEFT IN PLACE — the next startup retries,
|
||||
// and in the meantime the app is down with desired_state:running, so the boot reconciler sees it as
|
||||
// an orphan and the dead-app alarm owns it. Clearing a marker whose restart failed would erase the
|
||||
// only durable record that an app is owed one.
|
||||
//
|
||||
// Returns nil when there was nothing to recover — so "no interrupted operation" and "the recovery
|
||||
// never ran" are distinguishable to the caller, not only in a log (standing rule 3).
|
||||
func (g *AppStopGuard) Recover() *AppStopRecovery {
|
||||
if g == nil || g.path == "" {
|
||||
return nil
|
||||
}
|
||||
m, ok := g.read()
|
||||
if !ok || !m.Active || len(m.Stacks) == 0 {
|
||||
return nil
|
||||
}
|
||||
if g.starter == nil {
|
||||
g.logger.Printf("[ERROR] [appstop] crash recovery: %d app(s) were stopped by %s and are owed a restart, but no stack starter is wired — leaving the marker for the next startup: %v",
|
||||
len(m.Stacks), m.Reason.humanReason(), m.Stacks)
|
||||
return nil
|
||||
}
|
||||
|
||||
g.logger.Printf("[WARN] [appstop] crash recovery: %s (op %q) was interrupted and left %d app(s) stopped — restarting them: %v",
|
||||
m.Reason.humanReason(), m.OpID, len(m.Stacks), m.Stacks)
|
||||
|
||||
res := &AppStopRecovery{Reason: m.Reason, OpID: m.OpID, StartedAt: m.StartedAt}
|
||||
for _, name := range m.Stacks {
|
||||
if err := g.starter.StartStack(name); err != nil {
|
||||
// R-174: a REFUSAL is not a failure. The starter's gate has said this app must not be
|
||||
// started (an absent data drive), so the app is left down deliberately and the holder
|
||||
// owns the restart. Logged at WARN with the reason, and kept out of Failed so it never
|
||||
// reaches the customer-enabled backup_failed alarm — see ErrStartRefused.
|
||||
if errors.Is(err, ErrStartRefused) {
|
||||
g.logger.Printf("[WARN] [appstop] crash recovery: NOT restarting %s — %v; the marker is KEPT and the holder owns the restart", name, err)
|
||||
res.Refused = append(res.Refused, name)
|
||||
continue
|
||||
}
|
||||
g.logger.Printf("[ERROR] [appstop] crash recovery: restart %s failed: %v", name, err)
|
||||
res.Failed = append(res.Failed, name)
|
||||
continue
|
||||
}
|
||||
g.logger.Printf("[INFO] [appstop] crash recovery: restarted %s after the interrupted %s", name, m.Reason.humanReason())
|
||||
res.Restarted = append(res.Restarted, name)
|
||||
}
|
||||
sort.Strings(res.Failed)
|
||||
sort.Strings(res.Refused)
|
||||
sort.Strings(res.Restarted)
|
||||
|
||||
// The marker is kept for BOTH unfinished outcomes, for the same reason and with different
|
||||
// urgency: a failed restart is retried next startup, and a refused one is genuinely unfinished
|
||||
// until its drive returns. Clearing it in either case would erase the only durable record that
|
||||
// an app is owed a restart.
|
||||
if len(res.Failed) > 0 {
|
||||
g.logger.Printf("[ERROR] [appstop] crash recovery: %d app(s) could not be restarted — KEEPING the marker so the next startup retries; the dead-app alarm owns them meanwhile: %v",
|
||||
len(res.Failed), res.Failed)
|
||||
return res
|
||||
}
|
||||
if len(res.Refused) > 0 {
|
||||
g.logger.Printf("[WARN] [appstop] crash recovery: %d app(s) were deliberately NOT restarted (drive absent) — KEEPING the marker; this is the gate working, not a fault: %v",
|
||||
len(res.Refused), res.Refused)
|
||||
return res
|
||||
}
|
||||
g.End()
|
||||
return res
|
||||
}
|
||||
|
||||
// HeldStacks returns the stacks an app-data operation is CURRENTLY holding down, or nil.
|
||||
//
|
||||
// Read-only and nil-safe. It exists for the boot reconciler (§8.2): once R-157 mechanism A widened
|
||||
// the boot window, the sweep could overlap a running volume dump or export and "recover" an app that
|
||||
// is deliberately stopped mid-operation — restarting it under a tar, which is the inconsistency the
|
||||
// stop was taken to avoid. Recover() has already run to completion by then, so a marker seen through
|
||||
// this method belongs to an operation running NOW, not to a crashed one.
|
||||
func (g *AppStopGuard) HeldStacks() []string {
|
||||
if g == nil || g.path == "" {
|
||||
return nil
|
||||
}
|
||||
m, ok := g.read()
|
||||
if !ok || !m.Active {
|
||||
return nil
|
||||
}
|
||||
return append([]string(nil), m.Stacks...)
|
||||
}
|
||||
|
||||
// ---- marker persistence (atomic, 0600) — the quiesce shape ------------------------------------
|
||||
|
||||
func (g *AppStopGuard) write(m AppStopMarker) error {
|
||||
data, err := json.MarshalIndent(m, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(g.path), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
tmp := g.path + ".tmp"
|
||||
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := f.Write(data); err != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
// fsync before rename: the whole point is surviving a power cut, and a rename that lands ahead
|
||||
// of the bytes it points at is a marker that reads as corrupt at exactly the wrong moment.
|
||||
if err := f.Sync(); err != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp, g.path)
|
||||
}
|
||||
|
||||
func (g *AppStopGuard) read() (AppStopMarker, bool) {
|
||||
data, err := os.ReadFile(g.path)
|
||||
if err != nil {
|
||||
return AppStopMarker{}, false
|
||||
}
|
||||
var m AppStopMarker
|
||||
if err := json.Unmarshal(data, &m); err != nil {
|
||||
// Never a silent skip (§9.4): a corrupt marker is LOUD and the bad file is quarantined, so a
|
||||
// genuinely interrupted operation leaves a trace instead of vanishing. Still returns false —
|
||||
// "no usable marker ⇒ no recovery" is the correct contract, and matches quiesce's.
|
||||
g.logger.Printf("[WARN] [appstop] the app-stop marker at %s is corrupt (%v) — quarantining; apps are NOT auto-restarted from it", g.path, err)
|
||||
_ = os.Rename(g.path, fmt.Sprintf("%s.corrupt-%d", g.path, g.now().Unix()))
|
||||
return AppStopMarker{}, false
|
||||
}
|
||||
return m, true
|
||||
}
|
||||
@@ -0,0 +1,395 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-166 part 2 — the app-stop crash marker.
|
||||
//
|
||||
// THE DISCIPLINE THAT MATTERS HERE (§10): a `defer` is not crash-safety, so a test that lets the
|
||||
// deferred cleanup run proves nothing about a crash. Every "interrupted" test below simulates a
|
||||
// SIGKILL by never reaching the restart — the marker is written, the process conceptually dies, and
|
||||
// a FRESH guard over the SAME file does the recovering. That is exactly what Campaign 8 fault 10
|
||||
// established on live hardware: a SIGKILL runs no deferred function, and what brought the stacks
|
||||
// back was the marker read at startup.
|
||||
|
||||
type fakeStarter struct {
|
||||
starts []string
|
||||
failWith map[string]error
|
||||
}
|
||||
|
||||
func (f *fakeStarter) StartStack(name string) error {
|
||||
f.starts = append(f.starts, name)
|
||||
if err := f.failWith[name]; err != nil {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func newGuard(t *testing.T, dir string) (*AppStopGuard, *fakeStarter) {
|
||||
t.Helper()
|
||||
s := &fakeStarter{}
|
||||
g := NewAppStopGuard(filepath.Join(dir, "appstop-state.json"), log.New(io.Discard, "", 0))
|
||||
g.SetStarter(s)
|
||||
return g, s
|
||||
}
|
||||
|
||||
func markerPath(dir string) string { return filepath.Join(dir, "appstop-state.json") }
|
||||
|
||||
func markerExists(t *testing.T, dir string) bool {
|
||||
t.Helper()
|
||||
_, err := os.Stat(markerPath(dir))
|
||||
if err != nil && !os.IsNotExist(err) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// --- Scenario E — a crash mid-backup brings the app back -----------------------------------------
|
||||
|
||||
func TestRecover_InterruptedVolumeDump_RestartsTheAppAndClearsTheMarker(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
|
||||
// --- process 1: an operation stops the app and is KILLED. No End(), no defer, no cleanup. ---
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatalf("Begin: %v", err)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("Begin did not write a marker — nothing would survive the kill")
|
||||
}
|
||||
// <SIGKILL here> — g1 is abandoned deliberately; nothing else is called on it.
|
||||
|
||||
// --- process 2: a fresh controller starts and recovers from the file alone. ---
|
||||
g2, starter := newGuard(t, dir)
|
||||
res := g2.Recover()
|
||||
|
||||
if len(starter.starts) != 1 || starter.starts[0] != "immich" {
|
||||
t.Fatalf("started %v, want exactly [immich] — the app was left stranded by the interrupted backup", starter.starts)
|
||||
}
|
||||
if res == nil || len(res.Restarted) != 1 || res.Restarted[0] != "immich" {
|
||||
t.Fatalf("recovery result = %+v, want immich restarted", res)
|
||||
}
|
||||
if res.Reason != ReasonVolumeDump {
|
||||
t.Fatalf("reason = %q, want %q — the operator must be told WHICH operation was interrupted", res.Reason, ReasonVolumeDump)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker survived a successful recovery — the next boot would restart the app again")
|
||||
}
|
||||
// The operator-facing text must name the interruption, not merely report a restart.
|
||||
if msg := res.Message(); msg == "" || !strings.Contains(msg, "interrupted") {
|
||||
t.Fatalf("operator message %q does not say the operation was interrupted", msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_NoMarker_IsASilentNoOp(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g, starter := newGuard(t, dir)
|
||||
if res := g.Recover(); res != nil {
|
||||
t.Fatalf("Recover reported %+v on a box with no marker", res)
|
||||
}
|
||||
if len(starter.starts) != 0 {
|
||||
t.Fatalf("started %v with no marker present", starter.starts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_FailedRestart_KEEPSTheMarkerForTheNextStartup(t *testing.T) {
|
||||
// The single most important failure behaviour: clearing a marker whose restart failed would
|
||||
// erase the only durable record that an app is owed one. The app is genuinely still down.
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich", "nextcloud"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
g2, starter := newGuard(t, dir)
|
||||
starter.failWith = map[string]error{"immich": errors.New("compose up: no such image")}
|
||||
res := g2.Recover()
|
||||
|
||||
if len(res.Failed) != 1 || res.Failed[0] != "immich" {
|
||||
t.Fatalf("failed=%v, want [immich]", res.Failed)
|
||||
}
|
||||
if len(res.Restarted) != 1 || res.Restarted[0] != "nextcloud" {
|
||||
t.Fatalf("restarted=%v, want [nextcloud] — one app failing must not abort the others", res.Restarted)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared even though a restart FAILED — the next startup would not retry")
|
||||
}
|
||||
if msg := res.Message(); !strings.Contains(msg, "NOT be restarted") {
|
||||
t.Fatalf("operator message %q does not report the failure", msg)
|
||||
}
|
||||
if d := res.Detail(); !strings.Contains(d, "restart_failed") || !strings.Contains(d, "immich") {
|
||||
t.Fatalf("detail %q does not name which app failed", d)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_IsIdempotentAcrossRepeatedStartups(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("op", ReasonOffboxReconstitute, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g2, s2 := newGuard(t, dir)
|
||||
g2.Recover()
|
||||
g3, s3 := newGuard(t, dir)
|
||||
g3.Recover()
|
||||
|
||||
if len(s2.starts) != 1 {
|
||||
t.Fatalf("first recovery started %v", s2.starts)
|
||||
}
|
||||
if len(s3.starts) != 0 {
|
||||
t.Fatalf("a SECOND startup restarted %v again — the marker was not cleared", s3.starts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_CorruptMarkerIsQuarantinedNotSilentlySkipped(t *testing.T) {
|
||||
// §9.4: never a silent skip. A corrupt marker cannot be acted on, but it must leave a trace —
|
||||
// otherwise a genuinely interrupted operation vanishes without evidence.
|
||||
dir := t.TempDir()
|
||||
if err := os.WriteFile(markerPath(dir), []byte("{not json"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g, starter := newGuard(t, dir)
|
||||
if res := g.Recover(); res != nil {
|
||||
t.Fatalf("a corrupt marker produced a recovery result %+v", res)
|
||||
}
|
||||
if len(starter.starts) != 0 {
|
||||
t.Fatalf("apps were started from a corrupt marker: %v", starter.starts)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the corrupt marker was left in place — it would be re-read forever")
|
||||
}
|
||||
quarantined, _ := filepath.Glob(markerPath(dir) + ".corrupt-*")
|
||||
if len(quarantined) != 1 {
|
||||
t.Fatalf("the corrupt marker was not quarantined (found %d) — it was silently dropped", len(quarantined))
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecover_NoStarterWiredKeepsTheMarker(t *testing.T) {
|
||||
// D-b's safety rule: never worse than not having the file. With no starter the guard cannot act,
|
||||
// so it must keep the record for a startup that can, rather than clear it and lose the app.
|
||||
dir := t.TempDir()
|
||||
g1, _ := newGuard(t, dir)
|
||||
if err := g1.Begin("op", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
g2 := NewAppStopGuard(markerPath(dir), log.New(io.Discard, "", 0)) // deliberately no SetStarter
|
||||
if res := g2.Recover(); res != nil {
|
||||
t.Fatalf("recovered without a starter: %+v", res)
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared with no starter wired — the app would never come back")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNilGuardIsInert(t *testing.T) {
|
||||
// A caller that was never wired must degrade to pre-v0.189.0 behaviour, not panic.
|
||||
var g *AppStopGuard
|
||||
if err := g.Begin("op", ReasonVolumeDump, []string{"x"}); err != nil {
|
||||
t.Fatalf("nil guard Begin returned %v", err)
|
||||
}
|
||||
g.End()
|
||||
if res := g.Recover(); res != nil {
|
||||
t.Fatalf("nil guard recovered %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkerContentsAreDiagnosable(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g, _ := newGuard(t, dir)
|
||||
if err := g.Begin("volume-dump:immich", ReasonVolumeDump, []string{"immich"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, err := os.ReadFile(markerPath(dir))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var m AppStopMarker
|
||||
if err := json.Unmarshal(raw, &m); err != nil {
|
||||
t.Fatalf("the marker on disk is not readable JSON: %v", err)
|
||||
}
|
||||
if !m.Active || m.OpID != "volume-dump:immich" || m.Reason != ReasonVolumeDump ||
|
||||
len(m.Stacks) != 1 || m.Stacks[0] != "immich" || m.StartedAt.IsZero() {
|
||||
t.Fatalf("the marker does not record enough to diagnose the interruption: %+v", m)
|
||||
}
|
||||
// 0600 — it names customer apps.
|
||||
fi, err := os.Stat(markerPath(dir))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if fi.Mode().Perm() != 0o600 {
|
||||
t.Fatalf("marker mode = %v, want 0600", fi.Mode().Perm())
|
||||
}
|
||||
}
|
||||
|
||||
func TestBeginWithNoStacksWritesNothing(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
g, _ := newGuard(t, dir)
|
||||
if err := g.Begin("op", ReasonVolumeDump, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("a marker was written for an operation that stops nothing")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenarios E/F — DumpAppVolumesSafe, the primary site ----------------------------------------
|
||||
|
||||
// inspectingProvider is the StackDataProvider slice DumpAppVolumesSafe touches. It records whether
|
||||
// the marker file EXISTED at each step — the positive observable for the ordering property. An
|
||||
// absent log line is not evidence (standing rule 3); the file's presence at the moment of the stop
|
||||
// is.
|
||||
//
|
||||
// GetDockerVolumes returns nothing, so the dump itself is a no-op and no Docker is involved — the
|
||||
// stop/start bracket around it is what is under test.
|
||||
type inspectingProvider struct {
|
||||
StackDataProvider
|
||||
markerFile string
|
||||
events []string
|
||||
stopErr error
|
||||
startErr error
|
||||
markerPresentAtStop bool
|
||||
markerAtStartCall bool
|
||||
// panicOnVolumes simulates a hard abort (SIGKILL/power cut) at the point the dump begins: the
|
||||
// unwind skips the restart statement, exactly as a kill would.
|
||||
panicOnVolumes bool
|
||||
}
|
||||
|
||||
func (p *inspectingProvider) GetDockerVolumes(string) []string {
|
||||
if p.panicOnVolumes {
|
||||
panic("simulated hard abort mid-dump")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (p *inspectingProvider) StopStack(name string) error {
|
||||
_, err := os.Stat(p.markerFile)
|
||||
p.markerPresentAtStop = err == nil
|
||||
p.events = append(p.events, "stop:"+name)
|
||||
return p.stopErr
|
||||
}
|
||||
|
||||
func (p *inspectingProvider) StartStack(name string) error {
|
||||
_, err := os.Stat(p.markerFile)
|
||||
p.markerAtStartCall = err == nil
|
||||
p.events = append(p.events, "start:"+name)
|
||||
return p.startErr
|
||||
}
|
||||
|
||||
func newDumpManager(t *testing.T, dir string, p *inspectingProvider) *Manager {
|
||||
t.Helper()
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
m := &Manager{logger: lg, stackProvider: p, systemDataPath: dir}
|
||||
m.appStop = NewAppStopGuard(markerPath(dir), lg)
|
||||
return m
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_MarkerCoversTheWholeStopStartWindow(t *testing.T) {
|
||||
// Scenario F, the happy path: the marker is on disk BEFORE the stop, still on disk for the whole
|
||||
// time the app is down, and GONE once the restart succeeds.
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir)}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
if err := m.DumpAppVolumesSafe("immich"); err != nil {
|
||||
t.Fatalf("DumpAppVolumesSafe: %v", err)
|
||||
}
|
||||
|
||||
if !p.markerPresentAtStop {
|
||||
t.Fatal("the marker was NOT on disk when the app was stopped — a crash one instruction later " +
|
||||
"strands the app, which is the entire failure this marker exists to prevent")
|
||||
}
|
||||
if !p.markerAtStartCall {
|
||||
t.Fatal("the marker was already gone while the app was still down")
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker survived a dump whose restart succeeded — the next boot would restart the app again")
|
||||
}
|
||||
if len(p.events) != 2 || p.events[0] != "stop:immich" || p.events[1] != "start:immich" {
|
||||
t.Fatalf("events=%v, want [stop:immich start:immich]", p.events)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_Interrupted_RecoveryBringsTheAppBack(t *testing.T) {
|
||||
// Scenario E end-to-end THROUGH THE PRODUCTION PATH, and WITHOUT running any cleanup.
|
||||
//
|
||||
// The abort is real: GetDockerVolumes panics, which unwinds out of DumpAppVolumesSafe AFTER the
|
||||
// marker was written and the app stopped, and BEFORE the restart statement — and because that
|
||||
// restart is a plain statement, not a defer, it never runs. That is the shape of a hard kill.
|
||||
//
|
||||
// The earlier version of this test called m.appStop.Begin itself, which meant it proved the
|
||||
// marker type worked and NOT that DumpAppVolumesSafe uses it — it survived the red-proof that
|
||||
// deleted the production Begin call. Driving the real function is what makes the proof bite.
|
||||
//
|
||||
// RED-PROOF: delete the `m.appStop.Begin(...)` call from DumpAppVolumesSafe and this test fails —
|
||||
// nothing is written, so nothing is recovered. Demonstrated in REPORT.md §5.
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir), panicOnVolumes: true}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
func() {
|
||||
defer func() {
|
||||
if recover() == nil {
|
||||
t.Error("the simulated abort did not fire — this test proves nothing")
|
||||
}
|
||||
}()
|
||||
_ = m.DumpAppVolumesSafe("immich")
|
||||
}()
|
||||
|
||||
if !p.markerPresentAtStop {
|
||||
t.Fatal("the app was stopped before any marker existed")
|
||||
}
|
||||
if p.markerAtStartCall {
|
||||
t.Fatal("the restart ran despite the abort — the simulation is wrong, not the code")
|
||||
}
|
||||
|
||||
// <the controller is gone> — a fresh one starts and recovers from the file alone.
|
||||
g, starter := newGuard(t, dir)
|
||||
res := g.Recover()
|
||||
|
||||
if len(starter.starts) != 1 || starter.starts[0] != "immich" {
|
||||
t.Fatalf("started %v — the app stopped by the interrupted dump was not brought back", starter.starts)
|
||||
}
|
||||
if res == nil || res.Reason != ReasonVolumeDump {
|
||||
t.Fatalf("recovery did not name the volume dump as the interrupted operation: %+v", res)
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("the marker was not cleared after a successful recovery")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_FailedRestartKeepsTheMarker(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir), startErr: errors.New("compose up failed")}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
if err := m.DumpAppVolumesSafe("immich"); err == nil {
|
||||
t.Fatal("a failed restart must surface as an error")
|
||||
}
|
||||
if !markerExists(t, dir) {
|
||||
t.Fatal("the marker was cleared even though the restart FAILED — the app is still down and " +
|
||||
"nothing records that it is owed a restart")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDumpAppVolumesSafe_FailedStopClearsTheMarker(t *testing.T) {
|
||||
// Nothing was stopped, so nothing is owed a restart. A stranded marker here would cost a
|
||||
// spurious restart at the next startup AND a false "a backup was interrupted" alert.
|
||||
dir := t.TempDir()
|
||||
p := &inspectingProvider{markerFile: markerPath(dir), stopErr: errors.New("stack is protected")}
|
||||
m := newDumpManager(t, dir, p)
|
||||
|
||||
if err := m.DumpAppVolumesSafe("traefik"); err == nil {
|
||||
t.Fatal("a failed stop must surface as an error")
|
||||
}
|
||||
if markerExists(t, dir) {
|
||||
t.Fatal("a marker was left behind for an app that was never stopped")
|
||||
}
|
||||
}
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
@@ -32,13 +33,69 @@ type Manager struct {
|
||||
// tier2Notify, if set, is called after each Tier 2 copy (success: err==nil) for notifications.
|
||||
tier2Notify func(stackName, destLabel string, dur time.Duration, err error)
|
||||
|
||||
// unitNotify (R-158 / R-167), if set, is called ONCE PER APP whose Tier-1 recovery-unit capture
|
||||
// FAILED, and the capture loop continues to the next app. Wired in cmd/controller/main.go.
|
||||
//
|
||||
// WHY IT EXISTS. `/backups/apps` is the page a person opens to ask whether ONE app is backed up,
|
||||
// and until now it was the one page that never said: a per-app capture failure was a `[WARN]`
|
||||
// line and went no further. The manager had three notify seams and none for the unit capture —
|
||||
// the FIFTH instance in this project of a mechanism built and left disconnected.
|
||||
//
|
||||
// IT CARRIES THE SPACE FIGURES DELIBERATELY. The overwhelmingly likely cause is a full
|
||||
// filesystem, and an operator who has the used/free bytes at the moment of failure can act
|
||||
// without logging in. It is the same pair of numbers the customer-facing fill warning reports,
|
||||
// which is why the two ship together.
|
||||
//
|
||||
// OPERATOR-TIER. Routed to a hub event type that is in `notify.operatorOnlyEvents` — a customer
|
||||
// can take no action on a capture failure. Deliberately NOT `backup_failed`, which is
|
||||
// customer-enabled by default and would email them in Hungarian about it (D-c).
|
||||
//
|
||||
// NO CONTROLLER-SIDE COOLDOWN — the hub owns cooldown, per the offboxEnlargeBlockedNotify
|
||||
// precedent.
|
||||
unitNotify func(stackName string, err error, usage *UnitSpace)
|
||||
|
||||
// unitSpaceFn (R-165 / B2), if set, replaces the real statfs behind the capture floor so a test
|
||||
// can state a filesystem's occupancy as an input. Nil in production → `unitTargetSpace`.
|
||||
unitSpaceFn func(stackName string) *UnitSpace
|
||||
|
||||
// admission (R-181) is the per-RUN memo of the reserve's per-app verdict, guarded by admissionMu.
|
||||
// Non-nil only for the duration of a backup run (beginAdmissionRun → its closer). One verdict per
|
||||
// app covers all THREE write legs — DB dump, volume dump, unit capture — because all three write
|
||||
// under one per-app root; see admission.go for why it is decided lazily and never re-decided.
|
||||
admissionMu sync.Mutex
|
||||
admission *admissionSet
|
||||
|
||||
// summary (R-182) is the per-RUN digest collector, guarded by summaryMu. Same lifetime as
|
||||
// `admission` and for the same reason: an absent collector means "no run in flight", never a
|
||||
// stale answer from last night. runSummaryNotify is the operator digest seam, wired in main.go.
|
||||
summaryMu sync.Mutex
|
||||
summary *runSummary
|
||||
runSummaryNotify func(RunSummary)
|
||||
// manualRun tags the NEXT run as operator-triggered (cleared as the run starts), so the digest
|
||||
// can say which kind it was and the hub can decline to collapse a manual run into a nightly one.
|
||||
manualRun atomic.Bool
|
||||
|
||||
// appStop (R-166) is the crash marker for operations that stop an app, work on its data, and
|
||||
// start it again. Written BEFORE the stop and cleared AFTER the restart, so a SIGKILL or a power
|
||||
// cut in that window leaves a durable record that Recover honours at the next startup. Built in
|
||||
// NewManager from cfg.Paths.DataDir — see appstop_marker.go for why it is not quiesce's file.
|
||||
appStop *AppStopGuard
|
||||
|
||||
// offbox (Part B): the restic-SFTP exec seam (nil → real restic) + the failure→operator-alert hook.
|
||||
offboxRunner offboxRunner
|
||||
offboxNotify func(dur time.Duration, snapshots int, err error)
|
||||
// offboxStreamRunner + offboxProgress: the MANUAL run's live progress (v0.147.0, 4c). The stream
|
||||
// seam scans restic's `--json` stdout line-by-line; the state is what the page polls. Both are
|
||||
// inert on the nightly path — the sink is installed only for the duration of a manual run.
|
||||
offboxStreamRunner offboxStreamRunner
|
||||
offboxProgress offboxProgressState
|
||||
// offboxOrphanEvent (v0.142.0), if set, pushes a hub event on offsite-repo continuity transitions
|
||||
// ("offbox_repo_orphaned" / "offbox_repo_reset"); renamedTo names the move-aside path (reset only).
|
||||
// Wired in main.go to the notifier. Nil-safe.
|
||||
offboxOrphanEvent func(eventType, renamedTo string)
|
||||
// offboxGapNotify (R-203) fires when a COMPLETED offsite run could not capture a directory an
|
||||
// app declares MANDATORY — a coverage gap, not a failed run. nil → no signal.
|
||||
offboxGapNotify func(gaps map[string][]string)
|
||||
// offboxSSH (v0.142.0) is the raw-ssh exec seam for the orphaned-repo move-aside (restic has no
|
||||
// rename); tests inject a fake. Nil → the real ssh invocation (defaultOffboxSSH).
|
||||
offboxSSH func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error)
|
||||
@@ -46,6 +103,13 @@ type Manager struct {
|
||||
// offboxSizer (3a) — the mandatory-set byte estimator for the pre-push enlargement gate, overridable
|
||||
// in tests so the gate is unit-testable without a real du. Nil → the real dirSizeBytes (du -sb).
|
||||
offboxSizer func(path string) int64
|
||||
// offboxNow (v0.206.0, R-241) is the abandonment countdown's clock. Nil → time.Now.
|
||||
//
|
||||
// IT EXISTS SO THE TERMINAL STEP IS TESTABLE WITHOUT SHORTENING A LIVE TIMER (§7.4). The sweep is
|
||||
// the only thing in the product that deletes a customer's off-site history; driving it with a
|
||||
// clock keeps that step exercised on every run of the suite instead of once, on real data, by an
|
||||
// operator who then has to hope.
|
||||
offboxNow func() time.Time
|
||||
// offboxEnlargeBlockedNotify (3a), if set, is called ONCE per app that NEWLY enters the
|
||||
// quota-blocked (enlargement-refused) state — edge-triggered against the persisted EnlargedBlocked
|
||||
// set so a nightly schedule can't re-notify a persistently-blocked app (the hub owns cooldown; the
|
||||
@@ -54,6 +118,16 @@ type Manager struct {
|
||||
// offboxPlaceCopier (3a) — the place-to-live missing-only merge seam (nil → rsyncRestoreMissing,
|
||||
// the `-a --ignore-existing` additive copy). Never rsyncMirror (--delete trap).
|
||||
offboxPlaceCopier func(src, dst string) (int, error)
|
||||
// offboxFullPlaceCopier (R-43, v0.148.0) — the FULL-restore overwrite seam (nil →
|
||||
// rsyncRestoreOverwrite: `-a` with NO --ignore-existing and NO --delete). Distinct from
|
||||
// offboxPlaceCopier on purpose: the two have opposite semantics for an existing file.
|
||||
offboxFullPlaceCopier func(src, dst string) (int, error)
|
||||
// safetyDumpFn (R-43) — the pre-restore safety-dump seam (nil → the real DumpOne), so the
|
||||
// "never replay without an undo on disk" refusal is unit-testable without Docker.
|
||||
safetyDumpFn func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult
|
||||
// offsitePreDumpFn (R-44) — the offsite dump pre-phase seam (nil → runDBDumpsInternal), so the
|
||||
// dumps-strictly-before-capture ordering is observable in a test without Docker or restic.
|
||||
offsitePreDumpFn func(ctx context.Context) error
|
||||
// offboxFreeFn (3a) — the free-space probe for the restore headroom gate, overridable in tests (the
|
||||
// Windows `go test` host has no `df`). Nil → the real diskFreeBytes (df --output=avail).
|
||||
offboxFreeFn func(path string) int64
|
||||
@@ -92,11 +166,33 @@ type Manager struct {
|
||||
// (`-a --delete`, contents-of-src semantics).
|
||||
tier2Mirror func(src, dst string) error
|
||||
|
||||
// sharesPassdbCapture (R-7b) — the samba passdb capture seam (a `docker exec … tar cf -`),
|
||||
// overridable so the shares payload builder is unit-testable without docker. Nil → the real
|
||||
// defaultSharesPassdbCapture. Best-effort by contract: an error yields a manifest-only payload.
|
||||
sharesPassdbCapture func() ([]byte, error)
|
||||
|
||||
// sharesPassdbRestore (R-7b) — the mirror seam for putting a captured passdb archive BACK into the
|
||||
// samba named volume (`docker exec -i … tar xf -`). Nil → the real defaultSharesPassdbRestore.
|
||||
sharesPassdbRestore func(tar []byte) error
|
||||
|
||||
// sharesReconcile (R-7b), if set, re-renders and applies the samba stack after a shares restore
|
||||
// re-adds definitions to the registry (wired in main.go to stacks.Manager.ReconcileSamba). It is a
|
||||
// SEAM rather than a direct call because the backup package must not depend on the stacks package.
|
||||
// Nil → the registry is updated and a WARN says smb.conf will catch up on the next health tick.
|
||||
sharesReconcile func() error
|
||||
|
||||
// tier2SSDFits (3b) — the SSD-headroom predicate seam, overridable in tests (system.GetDiskUsage is
|
||||
// Linux-only → nil on the Windows test host, which would always refuse the SSD branch). Nil → the
|
||||
// real tier2FitsSystemDrive.
|
||||
tier2SSDFits func(sys string, sizeBytes int64) bool
|
||||
|
||||
// samePhysicalDevice — the off-drive identity predicate behind every Tier-2 "is this really a
|
||||
// SECOND disk?" guard, overridable in tests. The real check is `st_dev` equality, so on a host
|
||||
// where every `t.TempDir()` lands on one filesystem the fixture's "two drives" are indistinguishable
|
||||
// and Tier-2 correctly refuses them — which makes the off-drive tests unrunnable rather than wrong.
|
||||
// Nil → the real system.SamePhysicalDevice (production always takes this path).
|
||||
samePhysicalDevice func(a, b string) bool
|
||||
|
||||
// migrationRunning, if set, reports whether a data migration is in progress. The scheduled
|
||||
// backup paths skip when it returns true (Change 3 — backup ↔ migration mutual exclusion), so a
|
||||
// nightly dump/Tier-2 can't race a migration copy/cleanup on the same drive.
|
||||
@@ -106,6 +202,13 @@ type Manager struct {
|
||||
lastDBDump *DBDumpStatus
|
||||
running bool
|
||||
|
||||
// R-43/R-44 (v0.148.0) — the coherence stamp of the offsite run in flight, read by
|
||||
// CaptureRecoveryUnit so each unit records WHICH run took the dumps sitting beside its files.
|
||||
// Set for the duration of the dump pre-phase + capture, cleared after; "" means "no offsite run
|
||||
// is establishing coherence right now" (the periodic refresh and the local 02:30 dump leg).
|
||||
offsiteRunID string
|
||||
offsiteRunDumpAt string
|
||||
|
||||
// Restore op-status (Part B, opstatus.go) — display-only async-restore progress, under `mu`.
|
||||
opRunning bool
|
||||
opName string
|
||||
@@ -172,10 +275,30 @@ func NewManager(cfg *config.Config, sett *settings.Settings, logger *log.Logger)
|
||||
settings: sett,
|
||||
systemDataPath: cfg.Paths.SystemDataPath,
|
||||
}
|
||||
// R-166: its OWN file next to quiesce-state.json, never inside it — one file, one writer.
|
||||
m.appStop = NewAppStopGuard(filepath.Join(cfg.Paths.DataDir, "appstop-state.json"), logger)
|
||||
m.reconcileCrashedRun()
|
||||
return m
|
||||
}
|
||||
|
||||
// AppStopGuard exposes the app-stop crash marker so the exporter (a different package with the same
|
||||
// stop-work-start shape) can share the one marker file rather than opening a second one.
|
||||
func (m *Manager) AppStopGuard() *AppStopGuard { return m.appStop }
|
||||
|
||||
// SetAppStopGuard injects the guard instead of using the one NewManager built. INIT-ONLY — call once
|
||||
// during single-threaded startup, before any backup runs.
|
||||
//
|
||||
// It exists because of a startup ORDERING constraint, not for testing: the guard's Recover must
|
||||
// complete before the boot reconciler is launched (main.go:~236) and this manager is not constructed
|
||||
// until ~line 272. So main.go builds the guard early, recovers, and hands the SAME object here —
|
||||
// rather than a second guard over the same file, which would be one file with two owners, the exact
|
||||
// shape this marker was kept out of quiesce's file to avoid.
|
||||
func (m *Manager) SetAppStopGuard(g *AppStopGuard) {
|
||||
if g != nil {
|
||||
m.appStop = g
|
||||
}
|
||||
}
|
||||
|
||||
// reconcileCrashedRun makes the persisted offbox status truthful after a crash (campaign C1): a controller
|
||||
// that died mid-run left LastStatus="running" on disk (the in-memory single-flight mutex is gone with the
|
||||
// process, but the persisted status keeps lying "running" forever). Flip it to error with a Hungarian
|
||||
@@ -216,7 +339,10 @@ func (m *Manager) GetAppDrivePath(stackName string) string {
|
||||
// as-is; only the SSD-only system-data fallback gets the felhom-data subdir appended. This is what
|
||||
// keeps a drive-resident app's backups single-nested instead of .../felhom-data/felhom-data/... .
|
||||
func (m *Manager) namespaceRoot(drivePath string) string {
|
||||
return NamespaceRoot(drivePath, drivePath != m.systemDataPath)
|
||||
// R-203: delegates to the ONE expression of the rule (appbackup.NamespaceRootFor). This used to
|
||||
// hold its own copy — `drivePath != m.systemDataPath`, without Clean on either side — while
|
||||
// stacks.Manager.inGuest held a second copy WITH Clean. Two copies that already differed.
|
||||
return NamespaceRootFor(drivePath, m.systemDataPath)
|
||||
}
|
||||
|
||||
// AppNamespaceRoot returns the felhom-data namespace root for a stack's keep-side backups, resolving
|
||||
@@ -290,11 +416,48 @@ func (m *Manager) RunDBDumps(ctx context.Context) error {
|
||||
return m.runDBDumpsInternal(ctx)
|
||||
}
|
||||
|
||||
// offsiteRunStamp returns the in-flight offsite run's coherence stamp ("" when none).
|
||||
func (m *Manager) offsiteRunStamp() (runID, dumpsAt string) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
return m.offsiteRunID, m.offsiteRunDumpAt
|
||||
}
|
||||
|
||||
// beginOffsiteRunStamp marks the start of an offsite run's coherence window and returns the cleanup.
|
||||
// The stamp is what CaptureRecoveryUnit writes into each unit manifest, so it must be live across
|
||||
// BOTH the dump leg and the unit capture that follows it — those two together are the pair.
|
||||
func (m *Manager) beginOffsiteRunStamp(runID string) func() {
|
||||
m.mu.Lock()
|
||||
m.offsiteRunID = runID
|
||||
m.offsiteRunDumpAt = time.Now().UTC().Format(time.RFC3339)
|
||||
m.mu.Unlock()
|
||||
return func() {
|
||||
m.mu.Lock()
|
||||
m.offsiteRunID, m.offsiteRunDumpAt = "", ""
|
||||
m.mu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// runDBDumpsInternal is the implementation of RunDBDumps. Caller must hold the running flag.
|
||||
func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
start := time.Now()
|
||||
m.logger.Printf("[INFO] [backup] Starting database dump run")
|
||||
|
||||
// R-181: open the per-run admission scope HERE, because this function is the single orchestrator
|
||||
// of all three write legs. Each app's reserve verdict is taken at its first write of this run and
|
||||
// then reused by the other two legs, so a refused app writes nothing at all and is alerted once.
|
||||
// The scope is closed on every exit path — a set that outlived its run would answer tonight's
|
||||
// question with last night's disk.
|
||||
defer m.beginAdmissionRun()()
|
||||
|
||||
// R-182: the digest scope has the same lifetime. `emitRunSummary` runs BEFORE the closer (defers
|
||||
// unwind last-in-first-out), so the summary is still populated when it is sent, and it sends
|
||||
// nothing at all when the run was clean.
|
||||
kind := m.runKindFor()
|
||||
m.manualRun.Store(false) // tags exactly ONE run; a stale flag would mislabel every later nightly
|
||||
defer m.beginRunSummary(kind, newRunID())()
|
||||
defer m.emitRunSummary()
|
||||
|
||||
dbs, err := DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Database discovery failed: %v", err)
|
||||
@@ -330,6 +493,16 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
continue
|
||||
}
|
||||
|
||||
// R-181: the reserve, BEFORE the first byte of this app's backup is written. This is usually
|
||||
// where an app's verdict is taken, because the DB leg runs first; the volume leg and the
|
||||
// capture then read the same memo. SKIP, not FAIL — a deliberate hold is not a broken dump,
|
||||
// and the operator alert (fired once, inside admitApp) is the signal that it happened.
|
||||
m.noteAttempted(db.StackName)
|
||||
if !m.admitApp(db.StackName) {
|
||||
summary = append(summary, fmt.Sprintf("SKIP %s (reserve — app backup refused)", db.ContainerName))
|
||||
continue
|
||||
}
|
||||
|
||||
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), db.StackName)
|
||||
|
||||
result := DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
|
||||
@@ -338,6 +511,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
if result.Error != nil {
|
||||
allOK = false
|
||||
summary = append(summary, fmt.Sprintf("FAIL %s: %v", result.DB.ContainerName, result.Error))
|
||||
m.noteFailure(db.StackName, "database dump", result.Error.Error())
|
||||
m.logger.Printf("[ERROR] [backup] DB dump failed for %s: %v", result.DB.ContainerName, result.Error)
|
||||
} else {
|
||||
totalSize += result.Size
|
||||
@@ -424,6 +598,12 @@ func failedSummaryLines(summary []string) []string {
|
||||
// variant stops the stack before its own volume check — calling it unconditionally would bounce
|
||||
// every volume-less app on every nightly run. Per-stack isolation mirrors the DB loop: one app's
|
||||
// failure is recorded and does not abort the others.
|
||||
//
|
||||
// R-181 adds the reserve to that order, and for the SAME reason: it sits ahead of DumpAppVolumesSafe,
|
||||
// so a refused app is never stopped. A refusal decided inside the Safe variant would already have
|
||||
// bounced the app it was refusing to back up. It sits AFTER the volume-less check because an app with
|
||||
// no named volumes writes nothing in this leg — there is no first write here to gate, and consulting
|
||||
// the reserve for it would only decide a verdict early on a stale reading.
|
||||
func (m *Manager) runVolumeDumps() (summary []string, dumped int, allOK bool) {
|
||||
allOK = true
|
||||
if m.stackProvider == nil {
|
||||
@@ -459,9 +639,19 @@ func (m *Manager) runVolumeDumps() (summary []string, dumped int, allOK bool) {
|
||||
continue
|
||||
}
|
||||
|
||||
// R-181: the reserve, ahead of DumpAppVolumesSafe so a refused app is NOT stopped. For an app
|
||||
// that already has a DB this is a memo lookup taken before its DB dump; for a volume-only app
|
||||
// this is where its verdict is taken, still before its first byte.
|
||||
m.noteAttempted(stack.Name)
|
||||
if !m.admitApp(stack.Name) {
|
||||
summary = append(summary, fmt.Sprintf("SKIP %s volumes (reserve — app backup refused)", stack.Name))
|
||||
continue
|
||||
}
|
||||
|
||||
if err := dump(stack.Name); err != nil {
|
||||
allOK = false
|
||||
summary = append(summary, fmt.Sprintf("FAIL %s volumes: %v", stack.Name, err))
|
||||
m.noteFailure(stack.Name, "volume dump", err.Error())
|
||||
m.logger.Printf("[ERROR] [backup] Volume dump failed for %s: %v", stack.Name, err)
|
||||
continue
|
||||
}
|
||||
@@ -613,13 +803,28 @@ func atomicPromoteTar(tmpPath, finalPath string) error {
|
||||
// DumpAppVolumesSafe stops the stack before dumping volumes and restarts after.
|
||||
// Prevents inconsistent tars of live database volumes (e.g. PostgreSQL).
|
||||
// Protected stacks that reject StopStack will return an error — callers handle as warning.
|
||||
//
|
||||
// R-166: the stop→dump→start window is marked. Before this, a controller killed between the stop
|
||||
// and the start left the app down with NOTHING on disk saying why or that it was owed a restart —
|
||||
// and a stopped app has zero containers, which the boot reconciler then read as a deliberate
|
||||
// customer stop and left alone. The marker is the mechanism, not the restart call below: a SIGKILL
|
||||
// runs no deferred function (Campaign 8 fault 10, on live hardware), so only something already
|
||||
// written to disk can survive it.
|
||||
func (m *Manager) DumpAppVolumesSafe(stackName string) error {
|
||||
if m.stackProvider == nil {
|
||||
return fmt.Errorf("no stack provider")
|
||||
}
|
||||
|
||||
// Intent before the act: refuse to stop an app we cannot promise to restart.
|
||||
if err := m.appStop.Begin("volume-dump:"+stackName, ReasonVolumeDump, []string{stackName}); err != nil {
|
||||
return fmt.Errorf("could not record the app-stop marker for %s (refusing to stop it unprotected): %w", stackName, err)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [backup] Stopping %s for safe volume dump", stackName)
|
||||
if err := m.stackProvider.StopStack(stackName); err != nil {
|
||||
// Nothing was stopped, so nothing is owed a restart — clear rather than strand a marker that
|
||||
// would cost a spurious (if harmless) restart at the next startup.
|
||||
m.appStop.End()
|
||||
return fmt.Errorf("could not stop %s for volume dump: %w", stackName, err)
|
||||
}
|
||||
|
||||
@@ -629,6 +834,10 @@ func (m *Manager) DumpAppVolumesSafe(stackName string) error {
|
||||
startErr := m.stackProvider.StartStack(stackName)
|
||||
if startErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Failed to restart %s after volume dump: %v", stackName, startErr)
|
||||
} else {
|
||||
// Cleared ONLY on a restart that succeeded. A failed restart keeps the marker so the next
|
||||
// startup retries — the app really is still owed one.
|
||||
m.appStop.End()
|
||||
}
|
||||
|
||||
// Surface both errors — callers must know if the app is left stopped
|
||||
@@ -655,6 +864,12 @@ func (m *Manager) IsRunning() bool {
|
||||
return m.running
|
||||
}
|
||||
|
||||
// AcquireRunningForTest / ReleaseRunningForTest occupy the single-flight from another package's
|
||||
// test, so the "a run is already in flight" branch can be exercised without racing a real run.
|
||||
// Test-only seam, in the same spirit as SetOffboxRunner; nothing in production calls them.
|
||||
func (m *Manager) AcquireRunningForTest() error { return m.acquireRunning() }
|
||||
func (m *Manager) ReleaseRunningForTest() { m.releaseRunning() }
|
||||
|
||||
// acquireRunning atomically sets the running flag. Returns error if already running.
|
||||
func (m *Manager) acquireRunning() error {
|
||||
m.mu.Lock()
|
||||
@@ -828,7 +1043,19 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
|
||||
// Phase 2: keep each app's recovery unit current with its definition. Idempotent
|
||||
// (checksum-skip), so this periodic refresh only writes when the config actually changed,
|
||||
// and ensures units exist shortly after startup without waiting for the daily DB dump.
|
||||
m.captureAllRecoveryUnits()
|
||||
//
|
||||
// R-182: this sweep gets its OWN digest scope. It has to, and the reason is the whole
|
||||
// balance of this change. The per-app event is now record-only, so without a digest here a
|
||||
// capture failure detected between runs would be recorded and NEVER notified — a new
|
||||
// silence introduced while closing one. But this path can fire on every status poll, so its
|
||||
// digest deliberately carries NO run id: the hub's ordinary 1-hour operator cooldown then
|
||||
// applies, which caps it at one mail an hour exactly as before, while the mail now lists
|
||||
// EVERY failing app instead of whichever one happened to be first.
|
||||
func() {
|
||||
defer m.beginRunSummary(runKindRefresh, "")()
|
||||
defer m.emitRunSummary()
|
||||
m.captureAllRecoveryUnits()
|
||||
}()
|
||||
}
|
||||
|
||||
// Fill in dynamic fields under lock.
|
||||
@@ -932,6 +1159,15 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
return status
|
||||
}
|
||||
|
||||
// sameDevice reports whether two paths sit on the same physical device, through the test seam when
|
||||
// one is installed. Nil seam → system.SamePhysicalDevice, i.e. byte-for-byte the previous behaviour.
|
||||
func (m *Manager) sameDevice(a, b string) bool {
|
||||
if m.samePhysicalDevice != nil {
|
||||
return m.samePhysicalDevice(a, b)
|
||||
}
|
||||
return system.SamePhysicalDevice(a, b)
|
||||
}
|
||||
|
||||
// hasOffDriveTarget reports whether any registered, schedulable storage path lives on a physical disk
|
||||
// OTHER than the system drive — i.e. whether a genuine off-drive (tier-2) copy is possible at all.
|
||||
// When false the box is single-drive: tier-1 is the ONLY local copy and 3-2-1 needs a 2nd drive or
|
||||
@@ -941,7 +1177,7 @@ func (m *Manager) hasOffDriveTarget() bool {
|
||||
return false
|
||||
}
|
||||
for _, sp := range m.settings.GetSchedulableStoragePaths() {
|
||||
if sp.Path == m.systemDataPath || system.SamePhysicalDevice(m.systemDataPath, sp.Path) {
|
||||
if sp.Path == m.systemDataPath || m.sameDevice(m.systemDataPath, sp.Path) {
|
||||
continue
|
||||
}
|
||||
return true
|
||||
|
||||
@@ -0,0 +1,305 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/fillwatch"
|
||||
)
|
||||
|
||||
// R-165 / decision B2 — the capture floor that replaces the `mp1` bulkhead.
|
||||
//
|
||||
// Before the merge, the 20 G backup partition kept a runaway capture from reaching
|
||||
// `/var/lib/docker`, because it was a different filesystem. After the merge it is the same one, and a
|
||||
// full Docker data-root is a stopped box. These pin the replacement.
|
||||
|
||||
// floorProvider lists stacks and always resolves recovery info — the floor must refuse BEFORE any of
|
||||
// that is consulted, so a capture that gets as far as GetStackRecoveryInfo has already lost.
|
||||
type floorProvider struct {
|
||||
stacks []string
|
||||
dir string
|
||||
infoHits []string // records every app whose recovery info was read = a capture that was ATTEMPTED
|
||||
}
|
||||
|
||||
func (p *floorProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *floorProvider) ListDeployedStacks() []StackSummary {
|
||||
out := make([]StackSummary, 0, len(p.stacks))
|
||||
for _, s := range p.stacks {
|
||||
out = append(out, StackSummary{Name: s})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *floorProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *floorProvider) GetStackHDDPath(string) string { return "" }
|
||||
func (p *floorProvider) GetImportRoot() string { return "" }
|
||||
func (p *floorProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *floorProvider) StopStack(string) error { return nil }
|
||||
func (p *floorProvider) StartStack(string) error { return nil }
|
||||
func (p *floorProvider) RefreshAndIsRunning(string) bool { return true }
|
||||
func (p *floorProvider) GetStackRecoveryInfo(name string) (RecoveryInfo, bool) {
|
||||
p.infoHits = append(p.infoHits, name)
|
||||
return RecoveryInfo{StackDir: filepath.Join(p.dir, "stacks", name)}, true
|
||||
}
|
||||
func (p *floorProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *floorProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *floorProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *floorProvider) GetStackClassifiedBinds(string) ([]ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
type floorHarness struct {
|
||||
m *Manager
|
||||
prov *floorProvider
|
||||
events []unitEvent
|
||||
usage map[string]*UnitSpace
|
||||
dir string
|
||||
}
|
||||
|
||||
// newFloorHarness injects the usage read, so the filesystem's occupancy is a test input rather than
|
||||
// something the test has to manufacture on a real disk.
|
||||
func newFloorHarness(t *testing.T, stacks ...string) *floorHarness {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
h := &floorHarness{
|
||||
prov: &floorProvider{stacks: stacks, dir: dir},
|
||||
usage: map[string]*UnitSpace{},
|
||||
dir: dir,
|
||||
}
|
||||
h.m = &Manager{
|
||||
logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: dir,
|
||||
stackProvider: h.prov,
|
||||
unitSpaceFn: func(name string) *UnitSpace { return h.usage[name] },
|
||||
}
|
||||
h.m.SetUnitNotify(func(name string, err error, u *UnitSpace) {
|
||||
h.events = append(h.events, unitEvent{app: name, err: err.Error(), usage: u})
|
||||
})
|
||||
return h
|
||||
}
|
||||
|
||||
func (h *floorHarness) setSpace(app string, usedPct, availGB float64) {
|
||||
h.usage[app] = &UnitSpace{
|
||||
Path: h.dir, UsedPercent: usedPct, AvailGB: availGB,
|
||||
TotalGB: 100, UsedGB: usedPct,
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario D — the floor refuses, per app, and says so ----------------------------------------
|
||||
|
||||
func TestFloor_RefusesTheAppAndLeavesItsPreviousUnitByteIdentical(t *testing.T) {
|
||||
h := newFloorHarness(t, "homebox", "immich", "nextcloud")
|
||||
h.setSpace("homebox", 40, 60)
|
||||
h.setSpace("immich", 98, 0.4) // below the floor on BOTH terms
|
||||
h.setSpace("nextcloud", 40, 60)
|
||||
|
||||
// A previous unit exists for the app about to be refused. Checksum it before and after.
|
||||
unitDir := filepath.Join(h.dir, "felhom-data", "backups", "primary", "immich", "compose")
|
||||
if err := os.MkdirAll(unitDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prev := filepath.Join(unitDir, "app.yaml")
|
||||
if err := os.WriteFile(prev, []byte("deployed: true\nenv:\n A: previous-good-value\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before := checksumFile(t, prev)
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
// The refused app must NOT have been attempted at all — the floor is checked BEFORE any write.
|
||||
for _, hit := range h.prov.infoHits {
|
||||
if hit == "immich" {
|
||||
t.Fatal("the refused app's recovery info was read — the capture was ATTEMPTED rather than " +
|
||||
"refused up front, so a write could have started and failed partway")
|
||||
}
|
||||
}
|
||||
|
||||
if after := checksumFile(t, prev); after != before {
|
||||
t.Fatalf("the previous unit changed (%s → %s) — a refused capture must leave the last good "+
|
||||
"copy byte-identical", before, after)
|
||||
}
|
||||
if _, err := os.Stat(prev); err != nil {
|
||||
t.Fatalf("the previous unit is gone: %v — the floor REFUSES, it never deletes", err)
|
||||
}
|
||||
|
||||
// Exactly one alert, for the refused app, carrying the space figures.
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("got %d alerts, want exactly 1: %+v", len(h.events), h.events)
|
||||
}
|
||||
e := h.events[0]
|
||||
if e.app != "immich" {
|
||||
t.Fatalf("alert names %q, want immich", e.app)
|
||||
}
|
||||
if e.usage == nil || e.usage.AvailGB != 0.4 {
|
||||
t.Fatalf("the alert carries no/incorrect space figures: %+v", e.usage)
|
||||
}
|
||||
if !strings.Contains(e.err, "reserve") {
|
||||
t.Fatalf("the alert message %q does not say it was a reserve refusal — an operator would read "+
|
||||
"it as a broken capture rather than a deliberate hold", e.err)
|
||||
}
|
||||
|
||||
// The other two must have been captured normally — one app's refusal must not silence its siblings.
|
||||
got := strings.Join(h.prov.infoHits, ",")
|
||||
if !strings.Contains(got, "homebox") || !strings.Contains(got, "nextcloud") {
|
||||
t.Fatalf("attempted=%v — the loop did not continue past the refusal", h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing may be deleted to make room, under any threshold. Nothing on this filesystem is
|
||||
// generational, so "the oldest" is always a DIFFERENT app's only local copy.
|
||||
func TestFloor_NeverDeletesAnotherAppsUnit(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich", "nextcloud")
|
||||
h.setSpace("immich", 99, 0.1)
|
||||
h.setSpace("nextcloud", 99, 0.1)
|
||||
|
||||
other := filepath.Join(h.dir, "felhom-data", "backups", "primary", "nextcloud")
|
||||
if err := os.MkdirAll(other, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
keep := filepath.Join(other, "manifest.json")
|
||||
if err := os.WriteFile(keep, []byte(`{"app_name":"nextcloud"}`), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before := checksumFile(t, keep)
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
if _, err := os.Stat(keep); err != nil {
|
||||
t.Fatalf("another app's unit was DELETED to make room: %v — nothing here is generational, so "+
|
||||
"pruning could only destroy an app's only local copy", err)
|
||||
}
|
||||
if after := checksumFile(t, keep); after != before {
|
||||
t.Fatal("another app's unit was modified while the filesystem was under the floor")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario E — the floor is not a wall by another name ----------------------------------------
|
||||
|
||||
// The floor is about the FILESYSTEM's remaining headroom, never the unit's size. A per-unit cap would
|
||||
// be R-163 rebuilt inside one volume.
|
||||
func TestFloor_LargeUnitWithAmpleSpaceIsCaptured(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// A huge app on a huge, mostly-empty filesystem: 40% used, 600 GB free.
|
||||
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 600, TotalGB: 1000, UsedGB: 400}
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("a capture was refused on a filesystem with 600 GB free (%+v) — the floor has become "+
|
||||
"a per-unit size cap, which is exactly the ceiling R-165 removed", h.events)
|
||||
}
|
||||
if len(h.prov.infoHits) != 1 || h.prov.infoHits[0] != "immich" {
|
||||
t.Fatalf("attempted=%v, want [immich] — the capture was not even tried", h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// The old 20 G ceiling must not survive anywhere: a unit far larger than the retired partition is
|
||||
// captured when the filesystem has room.
|
||||
func TestFloor_TheOld20GCeilingIsGone(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// 180 GB free, and the app's own data is 120 GB — SIX TIMES the retired 20 G area. The figure is
|
||||
// deliberately far above 20 so that a literal `UsedGB > 20` cap cannot survive this test: a
|
||||
// fixture sitting exactly on the old boundary would pass under the very shape it forbids.
|
||||
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 180, TotalGB: 300, UsedGB: 120}
|
||||
h.m.captureAllRecoveryUnits()
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("refused with 180 GB free: %+v — a fixed per-area limit survives somewhere", h.events)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group E — the floor sits BELOW the critical warning band -------------------------------------
|
||||
|
||||
// A floor that fires before its own warning is a silent failure wearing a threshold: the customer
|
||||
// would get a refusal with no prior notice that anything was wrong. The customer's `disk_critical`
|
||||
// must always come first.
|
||||
func TestFloorSitsBelowTheCriticalWarningBand(t *testing.T) {
|
||||
if FloorUsedPercent <= fillwatch.CritUsedPercent {
|
||||
t.Fatalf("FloorUsedPercent (%.1f) must be strictly ABOVE fillwatch.CritUsedPercent (%.1f) — "+
|
||||
"otherwise a capture can be refused before the customer was ever warned that the disk was "+
|
||||
"filling, which is a silent failure wearing a threshold",
|
||||
FloorUsedPercent, fillwatch.CritUsedPercent)
|
||||
}
|
||||
if FloorFreeGiB >= fillwatch.CritFreeGiB {
|
||||
t.Fatalf("FloorFreeGiB (%.1f) must be strictly BELOW fillwatch.CritFreeGiB (%.1f) — the "+
|
||||
"free-byte term needs the same ordering as the percentage term, or the free-byte path "+
|
||||
"refuses before it warns", FloorFreeGiB, fillwatch.CritFreeGiB)
|
||||
}
|
||||
// And below the WARNING band too, transitively — stated explicitly so the chain is readable.
|
||||
if FloorUsedPercent <= fillwatch.WarnUsedPercent || FloorFreeGiB >= fillwatch.WarnFreeGiB {
|
||||
t.Fatal("the floor is not beyond the warning band — the customer must be warned, then warned " +
|
||||
"critically, and only then can a capture be refused")
|
||||
}
|
||||
|
||||
// Both terms must be able to refuse INDEPENDENTLY — that is why there are two. estGiB=0 is the
|
||||
// history-less case, which exercises the headroom term alone.
|
||||
if _, r := (&Manager{}).floorVerdict(&UnitSpace{UsedPercent: 50, AvailGB: 0.5}, 0); r != floorHeadroom {
|
||||
t.Fatal("a filesystem with 0.5 GiB free at only 50% used was NOT refused — the free-byte term " +
|
||||
"does not trip on its own, so a very large volume can run out without the floor engaging")
|
||||
}
|
||||
if _, r := (&Manager{}).floorVerdict(&UnitSpace{UsedPercent: 98, AvailGB: 40}, 0); r != floorHeadroom {
|
||||
t.Fatal("a filesystem 98% used was NOT refused — the percentage term does not trip on its own")
|
||||
}
|
||||
if _, r := (&Manager{}).floorVerdict(&UnitSpace{UsedPercent: 50, AvailGB: 50}, 0); r != floorAdmit {
|
||||
t.Fatal("a healthy filesystem was refused")
|
||||
}
|
||||
}
|
||||
|
||||
// --- §8.4 — a nil usage read neither refuses nor warns --------------------------------------------
|
||||
|
||||
func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// No entry → the injected reader returns nil, which is what system.GetDiskUsage does on error.
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("an UNREADABLE filesystem produced %d alert(s): %+v — an absent, unmounted or "+
|
||||
"unreadable filesystem is the drive gate's business and already has its own alert; "+
|
||||
"refusing here would block every capture on a box whose drive merely blipped", len(h.events), h.events)
|
||||
}
|
||||
if len(h.prov.infoHits) != 1 {
|
||||
t.Fatalf("the capture was not attempted on an unreadable read (attempted=%v) — a nil reading "+
|
||||
"must not refuse", h.prov.infoHits)
|
||||
}
|
||||
}
|
||||
|
||||
// ErrCaptureFloor must be matchable, so a caller can tell a deliberate refusal from a broken capture.
|
||||
func TestErrCaptureFloor_IsMatchable(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
h.setSpace("immich", 99, 0.2)
|
||||
h.m.captureAllRecoveryUnits()
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("want 1 event, got %d", len(h.events))
|
||||
}
|
||||
// The seam hands a string, so assert on the sentinel's own text being present and distinct.
|
||||
if !errors.Is(errWrapForTest(), ErrCaptureFloor) {
|
||||
t.Fatal("ErrCaptureFloor does not survive wrapping")
|
||||
}
|
||||
if !strings.Contains(h.events[0].err, "Refused") && !strings.Contains(h.events[0].err, "refused") {
|
||||
t.Fatalf("the alert %q does not identify itself as a refusal", h.events[0].err)
|
||||
}
|
||||
}
|
||||
|
||||
func errWrapForTest() error { return errors.Join(ErrCaptureFloor, errors.New("ctx")) }
|
||||
|
||||
func checksumFile(t *testing.T, path string) string {
|
||||
t.Helper()
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
h := sha256.New()
|
||||
if _, err := io.Copy(h, f); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return hex.EncodeToString(h.Sum(nil))
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// oneDrivePerSubtree is the test stand-in for system.SamePhysicalDevice (st_dev equality).
|
||||
//
|
||||
// Why it exists: the real predicate asks "are these two paths on the same physical disk?", and
|
||||
// Tier-2's whole purpose is to refuse a target that is. On a host where every t.TempDir() lands on
|
||||
// one filesystem — DooPlex, and any CI box with a single volume — a fixture's "usb" and "flash"
|
||||
// dirs share one st_dev, so the guard correctly refuses them and the off-drive tests can never
|
||||
// exercise their subject. This models what the fixture is actually depicting: one drive per
|
||||
// directory subtree, so two paths share a device only when one contains the other (a path inside a
|
||||
// drive IS on that drive). Unrelated subtrees are distinct devices, exactly as real mountpoints are.
|
||||
//
|
||||
// It does NOT relax any assertion — the guard still runs, still refuses same-device targets (see
|
||||
// TestSharesTier2NeverTargetsItsOwnSourceDrive, which passes under this seam), and production keeps
|
||||
// using the real st_dev check because the seam is nil there.
|
||||
func oneDrivePerSubtree(a, b string) bool {
|
||||
a, b = filepath.Clean(a), filepath.Clean(b)
|
||||
if a == b {
|
||||
return true
|
||||
}
|
||||
sep := string(filepath.Separator)
|
||||
return strings.HasPrefix(a, b+sep) || strings.HasPrefix(b, a+sep)
|
||||
}
|
||||
@@ -54,8 +54,18 @@ func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err
|
||||
m.offboxNotify = fn
|
||||
}
|
||||
|
||||
// SetOffboxGapNotify wires the R-203 operator signal: a run that completed but could NOT capture a
|
||||
// directory an app declares MANDATORY. Distinct from offboxNotify, which fires only on a hard run
|
||||
// failure — a coverage gap is not a failed run, and until v0.197.0 it reached nobody at all.
|
||||
// gaps is app → the relative paths that were missed. nil → no signal (pre-R-203 behaviour).
|
||||
func (m *Manager) SetOffboxGapNotify(fn func(gaps map[string][]string)) {
|
||||
m.offboxGapNotify = fn
|
||||
}
|
||||
|
||||
// SetOffboxOrphanEvent wires the offsite-repo continuity event push (main.go → notifier).
|
||||
func (m *Manager) SetOffboxOrphanEvent(fn func(eventType, renamedTo string)) { m.offboxOrphanEvent = fn }
|
||||
func (m *Manager) SetOffboxOrphanEvent(fn func(eventType, renamedTo string)) {
|
||||
m.offboxOrphanEvent = fn
|
||||
}
|
||||
|
||||
// SetOffboxSSH overrides the raw-ssh exec used for the orphaned-repo move-aside (tests).
|
||||
func (m *Manager) SetOffboxSSH(fn func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error)) {
|
||||
@@ -67,6 +77,16 @@ func (m *Manager) SetOffboxSSH(fn func(ctx context.Context, host, user string, p
|
||||
// the orphan card instead of the raw restic error.
|
||||
var ErrOffboxOrphaned = fmt.Errorf("offbox repo orphaned: exists but keyed under a previous, no-longer-available passphrase")
|
||||
|
||||
// ErrOffboxRunInFlight is returned to the MANUAL caller only, when the single-flight dropped the
|
||||
// request because a run was already going (R-234). It is not a failure of anything — the run in
|
||||
// flight is doing the work — but it IS a request that did nothing, and the page must say so instead
|
||||
// of showing the previous run's verdict under a „started" message.
|
||||
// offboxWholeUnitGap is the pseudo-path used to report a WHOLE-unit gap through the mandatory-gap
|
||||
// notification, so a skipped app and a skipped directory reach the operator in one vocabulary.
|
||||
const offboxWholeUnitGap = "(a teljes alkalmazás — nincs helyi mentési egysége)"
|
||||
|
||||
var ErrOffboxRunInFlight = fmt.Errorf("an off-box backup is already running; this request did not start a new one")
|
||||
|
||||
// classifyResticProbe maps a `restic cat config` failure to a repo class. The signatures are the exact
|
||||
// restic stderr matched in the 2026-07-17 diagnosis + restic's no-repo message:
|
||||
// - "orphaned": repo present, wrong key ("wrong password or no key found") — the definitive signal
|
||||
@@ -90,6 +110,126 @@ func classifyResticProbe(out []byte, err error) string {
|
||||
}
|
||||
}
|
||||
|
||||
// ── F-DIAG: four causes, four messages, and no secrets ──────────────────────────────────────────
|
||||
//
|
||||
// The offsite failure notification was a raw passthrough:
|
||||
//
|
||||
// "a NAS-ra mentés hibázott (<dur>): " + err.Error()
|
||||
//
|
||||
// One string for every cause, so an operator could not tell a full disk from a dead network without
|
||||
// reading logs — AND a raw restic/ssh error carries the repo URL, which is built as
|
||||
// `sftp:<user>@<host>:<path>` (offboxBaseArgs). That breaks this project's keys-not-values rule at the
|
||||
// one place the text leaves the box.
|
||||
//
|
||||
// OffsiteFailureClass names the causes that are genuinely DISTINGUISHABLE where the error is produced.
|
||||
// Nothing is invented: each maps to a signal the code already has.
|
||||
type OffsiteFailureClass string
|
||||
|
||||
const (
|
||||
OffsiteFailQuota OffsiteFailureClass = "quota" // the pre-run soft-quota gate refused (offbox.go quota state)
|
||||
OffsiteFailOrphaned OffsiteFailureClass = "orphaned" // ErrOffboxOrphaned — repo keyed under a lost passphrase
|
||||
OffsiteFailNoRepo OffsiteFailureClass = "no_repo" // classifyResticProbe "norepo" — nothing at the location
|
||||
OffsiteFailNoUnits OffsiteFailureClass = "no_units" // apps toggled but no recovery unit found on any drive
|
||||
OffsiteFailTransport OffsiteFailureClass = "transport" // network / SFTP auth / host key / timeout
|
||||
OffsiteFailUnknown OffsiteFailureClass = "unknown" // genuinely unclassified — say so rather than guess
|
||||
)
|
||||
|
||||
// offsiteRepoURLRe matches the `sftp:user@host:/path` repo reference restic echoes back in its errors.
|
||||
// It is the BACKSTOP, not the primary defence — see sanitiseOffsiteErrorFor.
|
||||
var offsiteRepoURLRe = regexp.MustCompile(`sftp:[^\s"']+`)
|
||||
|
||||
// sanitiseOffsiteErrorFor strips anything that could carry a secret or a customer-identifying location
|
||||
// out of an error before it reaches a message, an event or a report.
|
||||
//
|
||||
// IT REDACTS THE KNOWN TARGET VALUES, not a guessed pattern. The first version of this function
|
||||
// regex-matched `sftp:…` and `user@host` and looked complete; its own test caught it leaking on
|
||||
// `ssh: connect to host <host> port 23: Connection refused`, which contains a BARE hostname in neither
|
||||
// shape. Guessing at what a secret looks like fails exactly where it matters — the target's host, user
|
||||
// and repo path are known here, so they are removed literally and the regex stays only as a backstop
|
||||
// for forms built before the target is loaded.
|
||||
//
|
||||
// Whole-token replacement, not masking: a partially-masked host still identifies the customer, and
|
||||
// "it looked masked" is how a leak survives review.
|
||||
func sanitiseOffsiteErrorFor(t *settings.OffboxTarget, err error) string {
|
||||
if err == nil {
|
||||
return ""
|
||||
}
|
||||
out := offsiteRepoURLRe.ReplaceAllString(err.Error(), "<repo>")
|
||||
if t != nil {
|
||||
// Longest first, so the repo path is not half-eaten by the host replacement.
|
||||
for _, v := range []string{t.RepoPath, t.Host, t.User} {
|
||||
if len(strings.TrimSpace(v)) >= 3 {
|
||||
out = strings.ReplaceAll(out, v, "<repo>")
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(out) > 300 {
|
||||
out = out[:300] + "…"
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ClassifyOffsiteFailure maps a run error to its cause.
|
||||
//
|
||||
// Order matters: the explicit sentinels first, then the text signatures. A cause that cannot be told
|
||||
// apart here returns OffsiteFailUnknown rather than being folded into a neighbour — inventing a
|
||||
// precision the code does not have is how a confident-but-wrong diagnosis ships.
|
||||
func ClassifyOffsiteFailure(err error) OffsiteFailureClass {
|
||||
if err == nil {
|
||||
return ""
|
||||
}
|
||||
if errors.Is(err, ErrOffboxOrphaned) {
|
||||
return OffsiteFailOrphaned
|
||||
}
|
||||
s := strings.ToLower(err.Error())
|
||||
switch {
|
||||
case strings.Contains(s, "tárhelykeretet"):
|
||||
return OffsiteFailQuota
|
||||
case strings.Contains(s, "produced no snapshots"):
|
||||
return OffsiteFailNoUnits
|
||||
case strings.Contains(s, "unable to open config file"),
|
||||
strings.Contains(s, "is there a repository at the following location"):
|
||||
return OffsiteFailNoRepo
|
||||
case strings.Contains(s, "connection refused"), strings.Contains(s, "connection reset"),
|
||||
strings.Contains(s, "no route to host"), strings.Contains(s, "i/o timeout"),
|
||||
strings.Contains(s, "timed out"), strings.Contains(s, "permission denied"),
|
||||
strings.Contains(s, "host key"), strings.Contains(s, "handshake"),
|
||||
strings.Contains(s, "could not resolve"), strings.Contains(s, "network is unreachable"):
|
||||
return OffsiteFailTransport
|
||||
default:
|
||||
return OffsiteFailUnknown
|
||||
}
|
||||
}
|
||||
|
||||
// OffsiteFailureMessage returns the operator-facing Hungarian message for a run failure: a distinct
|
||||
// cause line plus the SANITISED detail. The detail is kept because an operator needs something to act
|
||||
// on; it is sanitised because this text leaves the box.
|
||||
//
|
||||
// A method, not a function, so it can reach the target and redact its ACTUAL host/user/path rather
|
||||
// than pattern-matching at what those might look like.
|
||||
func (m *Manager) OffsiteFailureMessage(err error, dur time.Duration) string {
|
||||
var t *settings.OffboxTarget
|
||||
if m != nil && m.settings != nil {
|
||||
t = m.settings.GetOffboxTarget()
|
||||
}
|
||||
return offsiteFailureMessage(t, err, dur)
|
||||
}
|
||||
|
||||
func offsiteFailureMessage(t *settings.OffboxTarget, err error, dur time.Duration) string {
|
||||
head := map[OffsiteFailureClass]string{
|
||||
OffsiteFailQuota: "A távoli mentés nem fért el a tárhelykereten belül",
|
||||
OffsiteFailOrphaned: "A távoli tárhely egy korábbi, már nem elérhető kulccsal készült",
|
||||
OffsiteFailNoRepo: "A távoli tárhelyen nincs mentési adattár",
|
||||
OffsiteFailNoUnits: "Nem volt mit menteni: egyetlen kijelölt alkalmazásnak sem található mentése",
|
||||
OffsiteFailTransport: "A távoli tárhely nem érhető el (hálózat vagy bejelentkezés)",
|
||||
OffsiteFailUnknown: "A távoli mentés ismeretlen okból nem sikerült",
|
||||
}[ClassifyOffsiteFailure(err)]
|
||||
if head == "" {
|
||||
head = "A távoli mentés nem sikerült"
|
||||
}
|
||||
return fmt.Sprintf("%s (%s): %s", head, dur.Round(time.Second), sanitiseOffsiteErrorFor(t, err))
|
||||
}
|
||||
|
||||
func defaultOffboxSSH(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
|
||||
if port == 0 {
|
||||
port = 22
|
||||
@@ -204,7 +344,21 @@ func (m *Manager) ResetOrphanedRepo(ctx context.Context) error {
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
return m.resetOrphanedRepo(ctx, base, env, "operator-confirmed (claimed)")
|
||||
if err := m.resetOrphanedRepo(ctx, base, env, "operator-confirmed (claimed)"); err != nil {
|
||||
return err
|
||||
}
|
||||
// R-241: THE COUNTDOWN STARTS HERE AND NOT IN THE SHARED HELPER, deliberately. The helper is also
|
||||
// the UNCLAIMED auto-reset path (Scenario B in ensureOffboxRepo), where nobody decided anything —
|
||||
// an as-delivered box tidying a stranger's leftover store must not put a customer's 14-day
|
||||
// deletion clock on it. Only the confirmed, claimed choice is a decision.
|
||||
//
|
||||
// The path is read back from OrphanedRenamedTo, which the helper has just written.
|
||||
if cur := m.settings.GetOffboxTarget(); cur != nil && cur.OrphanedRenamedTo != "" {
|
||||
m.startAbandonCountdown(cur.OrphanedRenamedTo)
|
||||
} else {
|
||||
m.logger.Printf("[WARN] [offbox] the reset succeeded but no set-aside path was recorded — no countdown started; the old history stays indefinitely")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// shellQuote single-quotes a path for the remote shell (our repo paths have no single quotes).
|
||||
@@ -217,7 +371,9 @@ func (m *Manager) SetOffboxSizer(fn func(path string) int64) { m.offboxSizer = f
|
||||
func (m *Manager) SetOffboxEnlargeBlockedNotifier(fn func(stack string, estBytes int64, usedGB, quotaGB int)) {
|
||||
m.offboxEnlargeBlockedNotify = fn
|
||||
}
|
||||
func (m *Manager) SetOffboxPlaceCopier(fn func(src, dst string) (int, error)) { m.offboxPlaceCopier = fn }
|
||||
func (m *Manager) SetOffboxPlaceCopier(fn func(src, dst string) (int, error)) {
|
||||
m.offboxPlaceCopier = fn
|
||||
}
|
||||
|
||||
// offboxSize returns the mandatory-set byte estimator (nil seam → the real du -sb dirSizeBytes).
|
||||
func (m *Manager) offboxSize() func(string) int64 {
|
||||
@@ -239,10 +395,46 @@ func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir()
|
||||
func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") }
|
||||
func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") }
|
||||
|
||||
// ErrOffboxSealedPackageHeld is the R-241 mint refusal: this box has no repository password and the
|
||||
// hub is holding a sealed recovery package for it, so minting one would write a key the package does
|
||||
// not cover — orphaning the very history the customer's recovery code protects.
|
||||
//
|
||||
// It is a SENTINEL, not a failure. `ApplyOffsiteTarget` catches it and still configures the transport
|
||||
// (SSH key, known_hosts, host/user/path), because the transport is not the problem and having it is
|
||||
// what lets the recovery screen bring the tier up the moment the key arrives (R-219). What it does
|
||||
// NOT do is let the tier come up under a key nobody escrowed.
|
||||
var ErrOffboxSealedPackageHeld = fmt.Errorf("offbox: the hub holds a sealed recovery package for this box — not minting a repository password over it")
|
||||
|
||||
// ErrOffboxSealedPackageHeld reports whether err is the mint refusal (errors.Is-friendly for callers
|
||||
// that wrap it).
|
||||
func IsOffboxSealedPackageHeld(err error) bool { return errors.Is(err, ErrOffboxSealedPackageHeld) }
|
||||
|
||||
// WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned
|
||||
// known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator
|
||||
// (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty
|
||||
// sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys).
|
||||
// (UI), never logged. Idempotent: an empty sshKey/knownHosts leaves the existing file untouched (a
|
||||
// re-save of just the target shouldn't wipe keys).
|
||||
//
|
||||
// ⚠ R-241 (v0.206.0) — IT NO LONGER MINTS OVER A SEALED PACKAGE, AND THAT IS THE WHOLE FIX.
|
||||
//
|
||||
// Until now the auto-generate branch consulted **one** input: does the file exist. Not the settings,
|
||||
// not the hub's ACK — nothing about whether anything already depended on a different key. Its two
|
||||
// neighbours in this very file, `OffsiteRecoveryOffer` (:1412) and `needsOffsiteCredential` (:1377),
|
||||
// BOTH consult `GetHubEscrowIdentityPresent()`. The same fact was available on three paths and used
|
||||
// on two.
|
||||
//
|
||||
// WHAT THAT COST, measured on the final walk (2026-08-06/07, SPIKE-r241-recovery-offer-2026-08-07):
|
||||
// a rebuilt box's credential self-heal reached here at 03:18:06Z and minted `9b4a9a9d…` over a hub
|
||||
// package sealing `30ef574f…`. The recovery screen then looked, found a key present and no orphan
|
||||
// recorded, and correctly said there was nothing to recover. **The screen was telling the truth; the
|
||||
// lie happened thirty minutes earlier, here.** And the flag was not merely available at that moment —
|
||||
// it was the PRECONDITION of the chain that reached this function: the credential retry only runs
|
||||
// while `needsOffsiteCredential` is true, which requires this exact flag, and the venue logged it at
|
||||
// 02:48:03Z, six ticks before the mint.
|
||||
//
|
||||
// THE GUARD IS DELIBERATELY NARROW — see the Scenario B test. It fires ONLY when a package is held AND
|
||||
// no password exists. A box the hub holds nothing for mints exactly as before, which is every
|
||||
// first-time box in the fleet; widening this to "never mint" would leave a new customer unable to
|
||||
// start, waiting for a package that will never exist.
|
||||
func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
|
||||
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
|
||||
return fmt.Errorf("offbox dir: %w", err)
|
||||
@@ -265,8 +457,15 @@ func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
|
||||
return fmt.Errorf("offbox known_hosts: %w", err)
|
||||
}
|
||||
}
|
||||
// Auto-generate the repo password once (0600), never log it.
|
||||
// Auto-generate the repo password once (0600), never log it — UNLESS the hub is holding a sealed
|
||||
// package for us (R-241). The transport files above are already written and that is deliberate.
|
||||
if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) {
|
||||
if m.sealedPackageHeld() {
|
||||
m.logger.Printf("[WARN] [offbox] NOT minting a repository password: the hub holds a sealed recovery package for this box, " +
|
||||
"and a fresh key would orphan the history that package protects (R-241). The transport is configured; " +
|
||||
"the tier stays down until the customer's recovery code places the escrowed key.")
|
||||
return ErrOffboxSealedPackageHeld
|
||||
}
|
||||
pw, gerr := generateOffboxPassword()
|
||||
if gerr != nil {
|
||||
return gerr
|
||||
@@ -278,6 +477,37 @@ func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// sealedPackageHeld reports the ACK-cached fact that the hub is holding a sealed recovery package for
|
||||
// this box. It is the SAME call `OffsiteRecoveryOffer` and `needsOffsiteCredential` already make —
|
||||
// deliberately, so the three paths can never disagree about it. A nil settings store reads as "no
|
||||
// package": the mint guard must never block a box whose settings could not be read, because that
|
||||
// would turn a transient read failure into a tier that never comes up.
|
||||
func (m *Manager) sealedPackageHeld() bool {
|
||||
return m.settings != nil && m.settings.GetHubEscrowIdentityPresent()
|
||||
}
|
||||
|
||||
// OffboxAwaitingRecoveryKey reports the R-241 holding state: a transport target exists, but no
|
||||
// repository password does, because the hub holds a sealed package and the mint was refused.
|
||||
//
|
||||
// DERIVED, NOT STORED, and that is the §2.1 ruling applied to this field too: a stored flag would be a
|
||||
// second copy of a fact the three inputs already carry, and a second copy is a thing that can drift.
|
||||
// The moment a recovery places the escrowed key, this goes false on its own with nothing to clear.
|
||||
//
|
||||
// ⚠ `t.Enabled` IS LOAD-BEARING, and it was missing in the first draft — caught by the existing
|
||||
// TestOffsiteDeclare_DisabledTargetIsNotStranded rather than by review. A customer who switched
|
||||
// off-site OFF is not awaiting anything, and a box that declares a holding state for a tier nobody
|
||||
// asked for is the R-215 shape (a screen about data the customer did not ask to protect). This is the
|
||||
// SAME Scenario-E carve-out `needsOffsiteCredential` makes two functions above; the two must agree,
|
||||
// and now do.
|
||||
func (m *Manager) OffboxAwaitingRecoveryKey() bool {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || !t.Enabled || !m.sealedPackageHeld() {
|
||||
return false
|
||||
}
|
||||
_, hasPw := m.OffboxRepoPasswordHash()
|
||||
return !hasPw
|
||||
}
|
||||
|
||||
// generateOffboxPassword returns a 256-bit hex repo password.
|
||||
func generateOffboxPassword() (string, error) {
|
||||
b := make([]byte, 32)
|
||||
@@ -347,8 +577,19 @@ var offboxRepoPwPattern = regexp.MustCompile(`^[0-9a-fA-F]{64}$`)
|
||||
// the repo password to the agent for escrow — the SAME fork-4 enable path a manual config takes. `stage` is
|
||||
// the agent escrow-stage push (nil skips it, e.g. when the agent is unreachable — the run gate still holds).
|
||||
func (m *Manager) ApplyOffsiteTarget(ctx context.Context, tgt *settings.OffboxTarget, sshKeyPEM, knownHosts string, stage func(ctx context.Context, pw string) error) error {
|
||||
// R-241: the mint refusal is a HOLDING state, not a failure. The transport files were written
|
||||
// before the refusal, so we still record the target — `OffboxConfigured()` stays false because the
|
||||
// password file is absent, which is what keeps runs gated, and the recovery screen can bring the
|
||||
// tier up the instant the escrowed key is placed (R-219's synchronous tier-up).
|
||||
//
|
||||
// Returning the error here instead would leave `needsOffsiteCredential` true forever, so the hub
|
||||
// would re-stage a credential the box had already consumed, on every cycle, for ever.
|
||||
awaitingKey := false
|
||||
if err := m.WriteOffboxSecrets(sshKeyPEM, knownHosts); err != nil {
|
||||
return fmt.Errorf("apply offsite secrets: %w", err)
|
||||
if !IsOffboxSealedPackageHeld(err) {
|
||||
return fmt.Errorf("apply offsite secrets: %w", err)
|
||||
}
|
||||
awaitingKey = true
|
||||
}
|
||||
// Re-apply (v0.109.1 live finding): the bridge rebuilds the target from the descriptor, but the
|
||||
// EXISTING target's custody + runtime status must carry over — EscrowState tracks the REPO PASSWORD
|
||||
@@ -359,7 +600,12 @@ func (m *Manager) ApplyOffsiteTarget(ctx context.Context, tgt *settings.OffboxTa
|
||||
tgt.EscrowState = cur.EscrowState
|
||||
tgt.LastRun, tgt.LastStatus, tgt.LastError = cur.LastRun, cur.LastStatus, cur.LastError
|
||||
tgt.LastDuration, tgt.LastWarning = cur.LastDuration, cur.LastWarning
|
||||
// R-100: carry the staleness anchor across a hub re-apply, for the same reason as the rest of
|
||||
// this block — a re-apply is not a new tier. Dropping it would reset an established tier to
|
||||
// "never succeeded" every time the hub re-pushes the descriptor.
|
||||
tgt.LastSuccess = cur.LastSuccess
|
||||
tgt.RepoSizeHuman, tgt.RepoSizeBytes, tgt.SnapshotCount = cur.RepoSizeHuman, cur.RepoSizeBytes, cur.SnapshotCount
|
||||
tgt.StatsKnown = cur.StatsKnown // R-225: carry the KNOWN-ness with the numbers
|
||||
}
|
||||
if tgt.EscrowState != "escrowed" {
|
||||
tgt.EscrowState = "pending"
|
||||
@@ -367,6 +613,14 @@ func (m *Manager) ApplyOffsiteTarget(ctx context.Context, tgt *settings.OffboxTa
|
||||
if err := m.settings.SetOffboxTarget(tgt); err != nil {
|
||||
return fmt.Errorf("apply offsite target: %w", err)
|
||||
}
|
||||
// R-241: nothing to stage — there is no repository password, by design. Say so once, plainly, and
|
||||
// return without touching the escrow. `PushOffboxPasswordForEscrow` would fail on the absent file
|
||||
// anyway; naming the situation beats a misleading "agent unreachable?" warning.
|
||||
if awaitingKey {
|
||||
m.logger.Printf("[INFO] [offbox] apply-offsite: transport configured for %s@%s:%s, tier HELD awaiting the escrowed key "+
|
||||
"(the hub holds a sealed package; no key was minted — R-241)", tgt.User, tgt.Host, tgt.RepoPath)
|
||||
return nil
|
||||
}
|
||||
if stage != nil {
|
||||
// Best-effort: the offbox is configured + pending regardless. A stage-push failure (agent momentarily
|
||||
// unreachable) is logged, not fatal — the escrow can be (re-)staged later (operator ceremony / re-enable).
|
||||
@@ -572,6 +826,22 @@ func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) erro
|
||||
// tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A
|
||||
// failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error.
|
||||
func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
return m.runOffboxBackup(ctx, false)
|
||||
}
|
||||
|
||||
// RunOffboxBackupWithProgress is the MANUAL („Távoli mentés most") entry point: identical work, but
|
||||
// with the live progress sink installed so the page can show total bytes, percent and current app
|
||||
// (v0.147.0, 4c). The nightly scheduled run keeps calling RunOffboxBackup and stays silent — nobody
|
||||
// is watching a progress bar at 03:00, and a sink left installed would publish stale percentages
|
||||
// into a page that never asked for them.
|
||||
func (m *Manager) RunOffboxBackupWithProgress(ctx context.Context) error {
|
||||
return m.runOffboxBackup(ctx, true)
|
||||
}
|
||||
|
||||
func (m *Manager) runOffboxBackup(ctx context.Context, withProgress bool) error {
|
||||
if withProgress {
|
||||
defer m.beginManualProgress()()
|
||||
}
|
||||
if !m.OffboxConfigured() {
|
||||
return fmt.Errorf("off-box backup not configured")
|
||||
}
|
||||
@@ -593,11 +863,33 @@ func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
}
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
m.logger.Printf("[INFO] [offbox] skipped — another backup is running")
|
||||
return nil // single-flight: don't race; the next scheduled run retries
|
||||
// R-234 (the MEASURED cause). The nightly path is unchanged: returning nil is right for it —
|
||||
// nobody asked, and the next scheduled run retries.
|
||||
//
|
||||
// The MANUAL path is a different question, and answering it the same way is what produced the
|
||||
// 2026-08-06 sequence. The customer pressed „Távoli mentés most" and was told
|
||||
// „A távoli mentés elindult"; the run was dropped here and returned nil; the card then showed
|
||||
// the PREVIOUS run's „✓ Rendben", which they read as covering the app they had just selected.
|
||||
// It did not — the restore refused for that app minutes later. A request that did nothing must
|
||||
// not be reported as one that started, so the manual caller is told.
|
||||
if withProgress {
|
||||
return ErrOffboxRunInFlight
|
||||
}
|
||||
return nil
|
||||
}
|
||||
defer m.releaseRunning()
|
||||
|
||||
apps := m.settings.GetOffboxApps()
|
||||
// Reserved-name defense in depth (R-7b): an app keyed `_shares` would collide with the shares
|
||||
// leg's restic tag and blocked-set entry. Catalog names cannot realistically produce this, but a
|
||||
// silent collision would corrupt both sources, so it is refused loudly instead.
|
||||
for i, a := range apps {
|
||||
if a == SharesPseudoStack {
|
||||
m.logger.Printf("[ERROR] [offbox] app %q uses the RESERVED shares key — excluded from the run to protect the shares leg", a)
|
||||
apps = append(apps[:i:i], apps[i+1:]...)
|
||||
break
|
||||
}
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
// Edge-trigger for the enlarge-blocked notification: capture the PRIOR blocked set so we notify only
|
||||
@@ -628,9 +920,40 @@ func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
m.offboxRecordStats(ctx, base, env) // the prune may have brought the size back down — refresh
|
||||
runErr = fmt.Errorf("A távoli mentés túllépte a tárhelykeretet (%d/%d GB) — törölj régi mentéseket vagy kérj nagyobb keretet.", usedGB, quota)
|
||||
} else {
|
||||
// R-43/R-44 (v0.148.0) — THE COHERENCE PRE-PHASE. Refresh the DB/volume dumps and the recovery
|
||||
// units BEFORE capturing, so the snapshot restic is about to write is an internally coherent
|
||||
// {DB@T, files@T} pair. Before this, a push shipped live files beside whatever dump the 02:30
|
||||
// local run happened to leave — on 2026-07-19 that was a dump taken four hours before the
|
||||
// customer's account even existed, so the "backup" of the photos contained zero of them
|
||||
// (DIAG-immich-restore-2026-07-19).
|
||||
//
|
||||
// Order matters and is the whole mechanism: dumps FIRST, then files. The gap between the two
|
||||
// can only ADD files the DB does not reference yet (an upload landing mid-run is a harmless
|
||||
// orphan blob), never remove one the DB DOES reference — so the file set is always a superset
|
||||
// of what the restored DB points at. The reverse order would produce dangling rows.
|
||||
//
|
||||
// This runs on the NIGHTLY path too, not just the manual one: "every snapshot is a coherent
|
||||
// pair" is the property that makes retention a history of restorable points rather than a
|
||||
// history of skewed ones. It also makes the nightly ordering structural instead of a
|
||||
// coincidence of two independent scheduler entries at 02:30 and 04:15.
|
||||
endStamp := m.beginOffsiteRunStamp(start.UTC().Format("20060102T150405Z"))
|
||||
if withProgress {
|
||||
m.offboxProgress.setPhase(OffboxPhaseDump)
|
||||
}
|
||||
dumpStart := time.Now()
|
||||
if dErr := m.offsitePreDump(ctx); dErr != nil {
|
||||
// Data-first: a dump failure must NOT abort the push. The files are still worth shipping,
|
||||
// and refusing to ship them would turn a degraded backup into no backup at all. It is a
|
||||
// loud WARN, and the unit manifest simply carries the older dump set — which the restore
|
||||
// confirm then surfaces as a skewed pair (P2) rather than silently pretending otherwise.
|
||||
m.logger.Printf("[WARN] [offbox] pre-push dump leg failed (%v) — continuing with the existing dumps; the snapshot's DB half may be older than its files", dErr)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [offbox] pre-push dump leg completed in %s — snapshot pair is coherent", time.Since(dumpStart).Round(time.Millisecond))
|
||||
}
|
||||
runResult, runErr = m.runOffboxInternal(ctx, apps, base, env, t)
|
||||
backedUp = runResult.backedUp
|
||||
missing = runResult.missing
|
||||
endStamp()
|
||||
}
|
||||
// Sorted names of apps whose enlargement was blocked this run (replaces the persisted set; empty clears).
|
||||
var blockedNames []string
|
||||
@@ -653,6 +976,10 @@ func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
}
|
||||
if perr := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.LastRun = time.Now().UTC().Format(time.RFC3339)
|
||||
// R-100: LastRun above records the ATTEMPT; this records the RESULT. The hub's staleness
|
||||
// verdict counts from the anchor, never from the attempt. INVARIANT: a failed run neither
|
||||
// advances nor clears it — pinned by TestOffboxAnchorAfterRun_* , not asserted in prose.
|
||||
o.LastSuccess = offboxAnchorAfterRun(o.LastSuccess, o.LastRun, runErr)
|
||||
o.LastDuration = dur.Round(time.Second).String()
|
||||
if errors.Is(runErr, ErrOffboxOrphaned) {
|
||||
// First-detection of the orphaned repo: RepoState (set by markOrphaned) drives the orphan
|
||||
@@ -665,27 +992,101 @@ func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
o.LastError = runErr.Error()
|
||||
o.LastWarning = ""
|
||||
} else {
|
||||
o.LastStatus = "ok"
|
||||
// R-203 — THE VERDICT. A run that could not capture a directory the app declares MANDATORY
|
||||
// is not a successful run. Until v0.197.0 it reported `ok` with a warning beside it, and a
|
||||
// warning beside a success is read as a success: that is how calibre-web's declared book
|
||||
// directory stayed out of every off-site snapshot on demo-hp while the card, the counters
|
||||
// and the hub all said the backup worked.
|
||||
//
|
||||
// NOT "error": the rest of the run worked and the data that WAS captured is real. The
|
||||
// snapshot count and the LastSuccess anchor are deliberately left to record it — half a
|
||||
// backup is not no backup, and reporting it as none would be its own lie. `incomplete` is
|
||||
// minted here because the existing vocabulary ("ok" | "error" | "running") has nothing that
|
||||
// means "it ran, and this app is not fully protected".
|
||||
// R-234 EXTENDS THE SAME RULE TO THE BIGGER CASE. Until v0.205.0 the paragraph above was
|
||||
// applied to ONE of the two shapes it describes: an app missing a declared mandatory
|
||||
// FOLDER made the run incomplete, while an app skipped ENTIRELY — no recovery unit, so
|
||||
// nothing of it in the snapshot at all — still reported ok with a warning beside it. The
|
||||
// smaller gap moved the verdict and the bigger one did not. Measured 2026-08-06: a run
|
||||
// reported „✓ Rendben · 1 pillanatkép" and the restore then refused for the app the
|
||||
// customer had just selected.
|
||||
gaps := len(runResult.mandatoryGaps) > 0
|
||||
unprotected := len(runResult.missingUnprotected) > 0
|
||||
if gaps || unprotected {
|
||||
o.LastStatus = "incomplete"
|
||||
if m.offboxGapNotify != nil {
|
||||
// Reuse, not mirror: the operator signal for "this run left an app less protected
|
||||
// than the customer asked for" is the same signal. A skipped app is reported as a
|
||||
// whole-unit gap so one notification shape covers both, and the recipient does not
|
||||
// have to learn a second vocabulary for the worse case.
|
||||
notify := runResult.mandatoryGaps
|
||||
if unprotected {
|
||||
if notify == nil {
|
||||
notify = map[string][]string{}
|
||||
} else {
|
||||
cp := make(map[string][]string, len(notify)+len(runResult.missingUnprotected))
|
||||
for k, v := range notify {
|
||||
cp[k] = v
|
||||
}
|
||||
notify = cp
|
||||
}
|
||||
for _, a := range runResult.missingUnprotected {
|
||||
notify[a] = append(notify[a], offboxWholeUnitGap)
|
||||
}
|
||||
}
|
||||
m.offboxGapNotify(notify)
|
||||
}
|
||||
} else {
|
||||
o.LastStatus = "ok"
|
||||
}
|
||||
o.LastError = ""
|
||||
o.SnapshotCount = snapshots
|
||||
o.StatsKnown = true // R-225: measured, even if the answer is zero
|
||||
o.EnlargedBlocked = blockedNames // replace each run (sorted); empty slice clears it
|
||||
var warns []string
|
||||
// Zero-toggle honesty (take-two obs.): a configured target with NOTHING selected reports
|
||||
// its emptiness instead of a bare success — the customer thinks offsite runs, but nothing
|
||||
// is covered until at least one app is toggled.
|
||||
if len(apps) == 0 {
|
||||
// R-7b: the shares leg counts as coverage — a box whose only cloud content is its shares
|
||||
// must not be told "nothing is selected".
|
||||
if len(apps) == 0 && !runResult.sharesBackedUp {
|
||||
warns = append(warns, "Sikeres — nincs mentésre jelölt alkalmazás")
|
||||
}
|
||||
if len(missing) > 0 {
|
||||
warns = append(warns, fmt.Sprintf("Figyelmeztetés: %d alkalmazásnak nincs elérhető mentése, ezek kimaradtak: %s",
|
||||
len(missing), strings.Join(missing, ", ")))
|
||||
// R-234 §7.4 — WHICH apps, WHY, and WHEN. The old sentence said only that N apps "had no
|
||||
// available backup and were left out", which names a problem with no next step and reads
|
||||
// the same whether the customer must act or simply wait.
|
||||
if len(runResult.missingUnprotected) > 0 {
|
||||
warns = append(warns, fmt.Sprintf(
|
||||
"Ezek az alkalmazások NEM kerültek be a távoli mentésbe, mert még nincs helyi mentési egységük: %s. A következő mentés általában már elkészíti — ha a második futás után is itt szerepelnek, szólj az üzemeltetőnek.",
|
||||
strings.Join(runResult.missingUnprotected, ", ")))
|
||||
}
|
||||
if len(runResult.missingNotDeployed) > 0 {
|
||||
warns = append(warns, fmt.Sprintf(
|
||||
"Ezek az alkalmazások ki vannak jelölve távoli mentésre, de nincsenek telepítve, ezért nem menthetők: %s. Ha már nincs rájuk szükséged, vedd ki a kijelölésüket a Távoli mentés oldalon.",
|
||||
strings.Join(runResult.missingNotDeployed, ", ")))
|
||||
}
|
||||
// 3a: capture-gap warnings (structurally-refused / on-disk-missing mandatory paths, undeployed).
|
||||
warns = append(warns, runResult.warns...)
|
||||
// 3a: the pre-push enlargement gate blocked some apps' userdata — config+DB still saved.
|
||||
if len(blockedNames) > 0 {
|
||||
// R-7b: the shares source is not an "app" and its degraded floor is the DEFINITIONS, not a
|
||||
// recovery unit — so it gets its own sentence and is excluded from the app count. The
|
||||
// persisted EnlargedBlocked set keeps the RAW `_shares` key (it is a lookup key the
|
||||
// templates index by); only this prose maps it through the display vocabulary.
|
||||
var blockedApps []string
|
||||
sharesBlocked := false
|
||||
for _, n := range blockedNames {
|
||||
if n == SharesPseudoStack {
|
||||
sharesBlocked = true
|
||||
continue
|
||||
}
|
||||
blockedApps = append(blockedApps, n)
|
||||
}
|
||||
if len(blockedApps) > 0 {
|
||||
warns = append(warns, fmt.Sprintf("Figyelmeztetés: a tárhelykeret miatt %d alkalmazásnál csak konfiguráció- és adatbázis-mentés készült: %s.",
|
||||
len(blockedNames), strings.Join(blockedNames, ", ")))
|
||||
len(blockedApps), strings.Join(blockedApps, ", ")))
|
||||
}
|
||||
if sharesBlocked {
|
||||
warns = append(warns, sharesBlockedWarning())
|
||||
}
|
||||
// SLICE 4: approaching the soft quota (≥80%, <100%) — warn on an otherwise-OK run.
|
||||
if qw := offboxQuotaWarning(o); qw != "" {
|
||||
@@ -708,7 +1109,10 @@ func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
usedGB := int(t.RepoSizeBytes / offboxGiB)
|
||||
for _, b := range runResult.blocked {
|
||||
if !priorBlocked[b.stack] {
|
||||
m.offboxEnlargeBlockedNotify(b.stack, b.estBytes, usedGB, t.QuotaGB)
|
||||
// DISPLAY BOUNDARY (R-7b): the notification is a customer-facing surface (it becomes a
|
||||
// Hungarian e-mail), so the reserved `_shares` key is mapped here — and ONLY here plus
|
||||
// the warning prose above. The persisted set and the restic tag stay raw.
|
||||
m.offboxEnlargeBlockedNotify(DisplayStackName(b.stack), b.estBytes, usedGB, t.QuotaGB)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -813,6 +1217,47 @@ type offboxRunResult struct {
|
||||
missing []string
|
||||
blocked []offboxBlocked
|
||||
warns []string
|
||||
// sharesBackedUp (R-7b) records that the sibling shares leg produced a snapshot this run. It keeps
|
||||
// the zero-toggle honesty notice honest: a box with no app toggled but shares in the cloud is NOT
|
||||
// "nothing is covered".
|
||||
sharesBackedUp bool
|
||||
// mandatoryGaps (R-203) is app → the relative paths of its declared MANDATORY data directories
|
||||
// that could NOT be captured. It is the STRUCTURED form of the warnings above, and it is what
|
||||
// decides the run's verdict: a run that dropped a mandatory directory is not a successful run.
|
||||
mandatoryGaps map[string][]string
|
||||
// missingUnprotected / missingNotDeployed (R-234) split `missing` by WHY, because only one of the
|
||||
// two may move the verdict. See the classification comment at the skip site: an app the customer
|
||||
// selected and that IS deployed but has no unit is unprotected and counts; an app that is no
|
||||
// longer installed is named but does not, so a removed app cannot leave the box amber forever.
|
||||
missingUnprotected []string
|
||||
missingNotDeployed []string
|
||||
}
|
||||
|
||||
// stackDeployed reports whether the stack is currently deployed on this box. Used only to classify a
|
||||
// skip (R-234) — never to decide whether to back something up.
|
||||
func (m *Manager) stackDeployed(stack string) bool {
|
||||
if m.stackProvider == nil {
|
||||
return false
|
||||
}
|
||||
for _, st := range m.stackProvider.ListDeployedStacks() {
|
||||
if st.Name == stack {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// driveUnavailableFor reports whether the app's drive is disconnected or decommissioned — states that
|
||||
// already have their own customer-facing signal, so a skip caused by them is not re-reported here.
|
||||
func (m *Manager) driveUnavailableFor(stack string) bool {
|
||||
if m.settings == nil {
|
||||
return false
|
||||
}
|
||||
d := m.GetAppDrivePath(stack)
|
||||
if d == "" {
|
||||
return false
|
||||
}
|
||||
return m.settings.IsDisconnected(d) || m.settings.IsDecommissioned(d)
|
||||
}
|
||||
|
||||
// runOffboxInternal does the repo-ensure + per-app DISCOVER → capture-set → gate → multi-path backup +
|
||||
@@ -833,12 +1278,43 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
|
||||
if !ok {
|
||||
m.logger.Printf("[WARN] [offbox] %s: no recovery unit found on any connected drive — skipping", stack)
|
||||
res.missing = append(res.missing, stack)
|
||||
// R-234 §7.2 — WHICH skips make the run not-successful. The list above is prose for the
|
||||
// customer; this classification is what the VERDICT may consult, and the two are not the
|
||||
// same question. Established by measurement on demo-hp 2026-08-06, not assumed:
|
||||
//
|
||||
// * DEPLOYED, no unit — the run's own pre-dump phase (captureAllRecoveryUnits) writes a
|
||||
// unit for every deployed stack before the push, so this state does not normally
|
||||
// survive a run. Reaching here means the capture was refused (the reserve) or failed.
|
||||
// The app the customer selected is NOT protected: it COUNTS.
|
||||
// * NOT DEPLOYED — nothing can protect an app that is not there, and the remedy is to
|
||||
// deselect it. It is NAMED so the customer can act, but it does NOT count: a box left
|
||||
// permanently amber over an app somebody removed is a status that stops being read,
|
||||
// which is how this whole class of defect starts.
|
||||
// * drive disconnected/decommissioned — has its own signal and its own card; not ours to
|
||||
// re-report as a backup gap.
|
||||
switch {
|
||||
case !m.stackDeployed(stack):
|
||||
res.missingNotDeployed = append(res.missingNotDeployed, stack)
|
||||
case m.driveUnavailableFor(stack):
|
||||
// counted as neither: the drive card is the honest surface for this one.
|
||||
default:
|
||||
res.missingUnprotected = append(res.missingUnprotected, stack)
|
||||
}
|
||||
continue
|
||||
}
|
||||
// Task 3-core TierOffsite capture set: mandatory userdata paths added to the unit snapshot,
|
||||
// plus loud warnings for structurally-refused / on-disk-missing mandatory paths (SP-3.4).
|
||||
extra, capWarns := m.offboxCaptureSet(stack)
|
||||
extra, capWarns, capGaps := m.offboxCaptureSet(stack)
|
||||
res.warns = append(res.warns, capWarns...)
|
||||
// R-203: the gaps are recorded STRUCTURALLY, not only as prose, because the run's verdict now
|
||||
// depends on them. A warning standing beside a success is read as a success — which is exactly
|
||||
// how a customer-declared mandatory directory stayed out of the snapshot while the run said ok.
|
||||
if len(capGaps) > 0 {
|
||||
if res.mandatoryGaps == nil {
|
||||
res.mandatoryGaps = map[string][]string{}
|
||||
}
|
||||
res.mandatoryGaps[stack] = append(res.mandatoryGaps[stack], capGaps...)
|
||||
}
|
||||
// Pre-push enlargement gate (§9): if last-known repo raw-data bytes + the mandatory-set estimate
|
||||
// would cross the soft quota, push UNIT-ONLY (protection never regresses) and record the block.
|
||||
if len(extra) > 0 && t != nil && t.QuotaGB > 0 {
|
||||
@@ -855,7 +1331,9 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
|
||||
}
|
||||
args := append([]string{"backup", "--tag", "felhom-offbox", "--tag", stack, src}, extra...)
|
||||
bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||||
out, berr := m.resticStep(bctx, env, base, "backup:"+stack, args...)
|
||||
// Streaming twin (4c): identical to resticStep unless a MANUAL run installed a progress sink,
|
||||
// in which case it adds --json and feeds the parsed status lines to the page's poll.
|
||||
out, berr := m.resticBackupStep(bctx, env, base, "backup:"+stack, stack, args...)
|
||||
cancel()
|
||||
if berr != nil {
|
||||
m.logger.Printf("[ERROR] [offbox] backup %s failed: %v: %s", stack, berr, truncate(out))
|
||||
@@ -867,6 +1345,22 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
|
||||
res.backedUp++
|
||||
m.logger.Printf("[INFO] [offbox] backed up %s (%s, %d mandatory path(s))", stack, src, len(extra))
|
||||
}
|
||||
// R-7b: the SHARES leg runs AFTER the per-app loop and BEFORE retention, so `forget --group-by
|
||||
// host,tags` covers the `_shares` group for free. It is placed BEFORE the firstErr return on
|
||||
// purpose: share protection must not be dropped because some unrelated app failed to push.
|
||||
m.offboxProgress.setPhase(OffboxPhaseShares)
|
||||
sharesRes, sharesErr := m.runOffboxSharesLeg(ctx, base, env, t)
|
||||
m.recordSharesOffsiteStatus(sharesRes)
|
||||
res.warns = append(res.warns, sharesRes.warns...)
|
||||
if sharesRes.blocked {
|
||||
res.blocked = append(res.blocked, offboxBlocked{stack: SharesPseudoStack, estBytes: sharesRes.estBytes})
|
||||
}
|
||||
if sharesRes.ran {
|
||||
res.sharesBackedUp = true
|
||||
}
|
||||
if sharesErr != nil && firstErr == nil {
|
||||
firstErr = sharesErr
|
||||
}
|
||||
if firstErr != nil {
|
||||
return res, firstErr
|
||||
}
|
||||
@@ -874,6 +1368,7 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
|
||||
// unit-only-shape snapshots share a group with its NEW enlarged shape (same <stack> tag) and age
|
||||
// out naturally — the default host,paths grouping would strand old-shape snapshots in their own
|
||||
// permanently-retained group. prune takes an EXCLUSIVE lock (the C2 stale-lock step) → resticStep.
|
||||
m.offboxProgress.setPhase(OffboxPhaseRetention)
|
||||
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||||
defer cancel()
|
||||
if out, ferr := m.resticStep(fctx, env, base, "prune", "forget", "--group-by", "host,tags", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune"); ferr != nil {
|
||||
@@ -889,28 +1384,251 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
|
||||
// multiplied by the retained-snapshot count (SP-1). The displayed size drops one-time after deploy.
|
||||
const offboxGiB = int64(1) << 30
|
||||
|
||||
// offboxAnchorAfterRun returns the last-SUCCESS anchor after a run that finished at `at` with
|
||||
// `runErr`, given the anchor value `prev` from before the run. R-100.
|
||||
//
|
||||
// THE RULE THIS ENCODES: a timestamp recording an ATTEMPT is not evidence of a RESULT. `LastRun` is
|
||||
// written unconditionally at the end of every run, failures included, so "how long since LastRun"
|
||||
// answers "how long since we last TRIED" — and the hub's staleness verdict was asking exactly that of
|
||||
// exactly that field, so a tier failing on every run read as perfectly fresh forever.
|
||||
//
|
||||
// Both directions matter and each is a different bug if got wrong:
|
||||
// - a FAILURE must not ADVANCE it → otherwise the original defect survives;
|
||||
// - a FAILURE must not CLEAR it → otherwise one bad night makes an established tier read as
|
||||
// never-succeeded, which is the mirror-image over-correction (and on the hub, the newborn-box path).
|
||||
//
|
||||
// It is a function rather than two lines inside the status closure so the rule can be red-proofed
|
||||
// directly; the first version of this fix modelled the rule in its own test and was therefore hollow.
|
||||
func offboxAnchorAfterRun(prev, at string, runErr error) string {
|
||||
if runErr != nil {
|
||||
return prev // failures neither advance nor clear the anchor
|
||||
}
|
||||
return at
|
||||
}
|
||||
|
||||
// OffboxReportStatus is the NON-SECRET offsite summary carried on the hub report (SLICE 4) — the input
|
||||
// to the hub's OffsiteChecker (fill + staleness alerts). nil when no offbox target is configured.
|
||||
type OffboxReportStatus struct {
|
||||
Enabled bool `json:"enabled"`
|
||||
EscrowState string `json:"escrow_state"`
|
||||
LastRun string `json:"last_run,omitempty"` // RFC3339
|
||||
LastStatus string `json:"last_status,omitempty"` // "ok" | "error" | "running"
|
||||
Enabled bool `json:"enabled"`
|
||||
EscrowState string `json:"escrow_state"`
|
||||
// State (v0.199.0, R-204 item 4 / R-193) is a DECLARED condition — the box naming its own
|
||||
// situation rather than the hub deducing it from a silence. Empty on every configured box, so a
|
||||
// healthy report's JSON is byte-identical to v0.198.0's.
|
||||
//
|
||||
// WHY A DECLARATION AND NOT AN INFERENCE (the operator ruling, 2026-08-05). An ABSENT off-site
|
||||
// object has FOUR meanings — never configured, mid-restart, a transient config read failure, and
|
||||
// rebuilt-and-stranded — and the hub cannot tell them apart. The BOX can, from two local facts it
|
||||
// holds with certainty. So it says so.
|
||||
State string `json:"state,omitempty"`
|
||||
// AbandonPurgeRequested (v0.206.0, R-241) — the customer's abandonment countdown has run out, the
|
||||
// set-aside off-site history HAS been deleted, and the hub is asked to drop the sealed package
|
||||
// that protected it so the two halves go together (Scenario F).
|
||||
//
|
||||
// It is a DECLARATION, on the same principle as State: the box knows it has deleted the store; the
|
||||
// hub cannot see that and must not infer it. It keeps being sent until the ACK stops reporting a
|
||||
// superseded package, so a lost request retries by itself rather than leaving the pair half-gone.
|
||||
// Absent/false on every other box, so a healthy report is byte-identical to v0.205.0's.
|
||||
AbandonPurgeRequested bool `json:"abandon_purge_requested,omitempty"`
|
||||
LastRun string `json:"last_run,omitempty"` // RFC3339
|
||||
LastStatus string `json:"last_status,omitempty"` // "ok" | "incomplete" (R-203) | "error" | "running"
|
||||
// LastSuccess (R-100) is the last run that SUCCEEDED — the hub's staleness anchor. Absent on a
|
||||
// pre-v0.181.0 controller, which the hub must degrade on explicitly rather than by accident:
|
||||
// treating absence as failure alarms every un-upgraded box, treating it as success keeps the bug.
|
||||
LastSuccess string `json:"last_success,omitempty"` // RFC3339
|
||||
SnapshotCount int `json:"snapshot_count"`
|
||||
RepoSizeBytes int64 `json:"repo_size_bytes"`
|
||||
QuotaGB int `json:"quota_gb"`
|
||||
}
|
||||
|
||||
// OffboxReportStatus returns the offsite summary for the hub report (nil = not configured; the hub's
|
||||
// checker treats absence as "nothing to watch" — pre-v0.109 reports look the same).
|
||||
// OffsiteStateNeedsCredential is the ONE declared state (v0.199.0, R-204 item 4 / R-193): this box
|
||||
// has no off-site tier, holds no repository password, and the hub says it is keeping a sealed
|
||||
// recovery package for it — i.e. it is a REBUILT box whose predecessor spent the one-time provider
|
||||
// password, and it cannot configure its off-site tier without a credential it has no way to obtain.
|
||||
// That was the last of the four manual interventions the 2026-08-04 drill needed.
|
||||
const OffsiteStateNeedsCredential = "needs_credential"
|
||||
|
||||
// OffsiteStateAwaitingRecoveryKey (v0.206.0, R-241) is the declared HOLDING state: the transport is
|
||||
// configured, but no repository password exists because the hub holds a sealed package and minting
|
||||
// one would orphan the history it protects. The box is not stranded (it has its credential) and not
|
||||
// healthy (it cannot run) — it is waiting for a person with a recovery code.
|
||||
//
|
||||
// WHY IT IS INERT TO EVERY EXISTING HUB READER, established from their code rather than assumed —
|
||||
// the same discipline `OffsiteStateNeedsCredential`'s own note applies:
|
||||
//
|
||||
// - `offsiteheal` acts on EXACTLY ONE string, `needs_credential` ("Everything else … is a no-op"),
|
||||
// so it will not re-stage a credential this box already has;
|
||||
// - `monitor.OffsiteChecker.isStale` returns false unless `Enabled && EscrowState == "escrowed"`,
|
||||
// and this object carries Enabled=false;
|
||||
// - `monitor/offsite_delivery.go` keys on the delivery shape, which is `applied` here (the secret
|
||||
// WAS consumed), and that branch is skipped;
|
||||
// - an unknown `state` string is ignored by encoding/json on an older hub.
|
||||
//
|
||||
// So this needs NO hub change to be safe. It does mean a held box raises no alarm — which is R-243,
|
||||
// filed and deliberately not widened here; the difference from R-241 is that this state is now
|
||||
// VISIBLE to the customer instead of silent.
|
||||
const OffsiteStateAwaitingRecoveryKey = "awaiting_recovery_key"
|
||||
|
||||
// needsOffsiteCredential is the stranded-rebuild predicate. BOTH facts are required and neither is
|
||||
// sufficient on its own — this is the whole correctness of the feature:
|
||||
//
|
||||
// - the data area is FRESH (no repository password on disk). Alone, this is simply a box that never
|
||||
// had off-site backups, and declaring on it would make every un-configured box in the fleet ask
|
||||
// for a credential.
|
||||
// - the HUB holds a sealed recovery package (the ACK's identity_blob_present, cached in settings).
|
||||
// Alone, this is a healthy box that has run its ceremony.
|
||||
//
|
||||
// Only together do they mean "this box HAD an off-site tier, and no longer has what it needs to use
|
||||
// it". A target that exists but is DISABLED is not stranded either — the customer switched it off —
|
||||
// so the caller only consults this when there is no enabled target, and a non-nil disabled target
|
||||
// short-circuits to false here.
|
||||
//
|
||||
// ⚠ THE DECLARATION STOPS WHEN THE TIER WORKS, NOT WHEN A KEY EXISTS (R-218, v0.201.0).
|
||||
//
|
||||
// This used to carry a third condition: hold a repository password ⇒ not stranded, stop declaring.
|
||||
// It reads as a sound freshness test and it is the exact opposite on the one path that matters,
|
||||
// because installing a repository password is the RECOVERY SCREEN'S WHOLE JOB. Measured live on
|
||||
// 2026-08-05 (CAMPAIGN-11 Phase 1):
|
||||
//
|
||||
// 13:39:54 needs_credential the box asks
|
||||
// 13:42:43 needs_credential the hub's 2-report debounce is satisfied
|
||||
// 13:47:03 hub re-stages the credential — "the box re-consumes on its next cycle"
|
||||
// 13:47:35 the customer's unlock succeeds and places the recovered key
|
||||
// 13:53:42 (silence) …and never asks again
|
||||
//
|
||||
// Thirty-two seconds after the remedy fired, the customer's own success switched off the mechanism
|
||||
// that would have delivered the coordinates for the key they had just recovered. The hub held an
|
||||
// unconsumed credential the box had no reason to collect; the box held a correct key and nowhere to
|
||||
// use it; the screen said "a few minutes"; nothing was ever going to happen. Two features, each
|
||||
// correct alone, cancelling on the path both were built for.
|
||||
//
|
||||
// The two remaining conditions are the honest ones: no target at all, and the hub is keeping a sealed
|
||||
// package for us. Both stay true exactly until the tier is configured — which is the moment the box
|
||||
// genuinely no longer needs a credential — and `OffboxReportStatus` stops consulting this predicate
|
||||
// the instant a target exists. **Scenario E is unaffected and is pinned by its own test**: a disabled
|
||||
// target is non-nil and still short-circuits at the first line, so a box whose customer switched
|
||||
// off-site off stays silent.
|
||||
func (m *Manager) needsOffsiteCredential(t *settings.OffboxTarget) bool {
|
||||
if t != nil {
|
||||
return false // a target exists (merely disabled) — the customer's own choice, not a rebuild
|
||||
}
|
||||
if m.settings == nil || !m.settings.GetHubEscrowIdentityPresent() {
|
||||
return false // the hub holds nothing for us: never had off-site backups
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// OffsiteRecoveryOffer reports whether the customer should be OFFERED the recovery screen (R-193):
|
||||
// the hub is holding a sealed recovery package for this box, and this box cannot open what that
|
||||
// package protects.
|
||||
//
|
||||
// TWO FACTS, BOTH REQUIRED — and this is the whole correctness of the screen:
|
||||
//
|
||||
// 1. **the hub holds a sealed package** (the ACK's identity_blob_present, cached in settings). Without
|
||||
// it there is nothing to recover, and a box that never had off-site backups must never be greeted
|
||||
// by a recovery screen for data it never had. Dropping this condition is the plausible wrong fix.
|
||||
// 2. **this box cannot open the history that package protects** — see the two shapes below.
|
||||
//
|
||||
// ⚠ WHY SHAPE (b) EXISTS, recorded because the task specified only shape (a) and the difference is
|
||||
// load-bearing. The literal reading of "the data area is fresh" is *no repository password on disk*,
|
||||
// which is true of a rebuilt box — but only until it re-applies its off-site target, because
|
||||
// `WriteOffboxSecrets` AUTO-GENERATES a repository password when none is present. That is precisely
|
||||
// R-193's orphaning mechanism, and since hub v0.96.0's credential self-heal the re-apply happens by
|
||||
// itself within ~15–30 minutes. So shape (a) alone would make this screen appear only inside a
|
||||
// half-hour window that closes on its own, and the customer who logs in the next morning — the actual
|
||||
// customer — would never see it. Shape (b) is the state they are in: a repository password exists, but
|
||||
// it is a NEW one and the inherited history cannot be opened with it, which the box has already
|
||||
// measured and recorded as `RepoState == "orphaned"`.
|
||||
//
|
||||
// Shape (b) also happens to be the state the existing move-aside requires (`ResetOrphanedRepo` refuses
|
||||
// unless orphaned), which is what lets "I do not want the old data" reach the shipped handler rather
|
||||
// than needing a new one.
|
||||
//
|
||||
// Scenario B still holds exactly: a healthy box has its own password and is not orphaned; a box that
|
||||
// never had off-site backups fails fact 1; an unclaimed box never reaches an authenticated page.
|
||||
// ── SHAPE (c), v0.206.0, R-241 — THE DISCRIMINATOR THAT ANSWERS THE REAL QUESTION ───────────────
|
||||
//
|
||||
// Shapes (a) and (b) are both PROXIES for one question — *does the hub hold a package for a key other
|
||||
// than the one I am using?* — and both have now been wrong, in opposite directions:
|
||||
//
|
||||
// - (a) "no repository password" went false the moment anything minted one. Before v0.206.0's mint
|
||||
// guard that happened by itself, ~30 minutes after a rebuild, and the customer who logged in the
|
||||
// next morning never saw the screen. That is R-241.
|
||||
// - (b) "a run proved the repo will not open" is unreachable on exactly that box: the only producer
|
||||
// of RepoState=="orphaned" is ensureOffboxRepo, which is downstream of the escrow gate in
|
||||
// runOffboxBackup, and the escrow can never confirm while the hub's package covers a different
|
||||
// key. Self-locking.
|
||||
//
|
||||
// (c) asks the question directly, from two facts the box already holds: the hash the hub's package
|
||||
// covers (ACK-cached) and the hash of the key on disk. **This comparison was already computed on every
|
||||
// ACK and thrown away** — see settings.HubEscrowKeySHA256.
|
||||
//
|
||||
// ⚠ §7.2 — WHAT A STALE OR ABSENT READING RESOLVES TO, decided deliberately rather than by default:
|
||||
//
|
||||
// - **A KNOWN DIFFERENCE OFFERS, however old the reading.** Age is not gated on. Both sides of the
|
||||
// comparison are local; only the hub's half can be stale, and what the hub holds does not change
|
||||
// without a ceremony THIS BOX runs — which refreshes the hash on the next ACK. Gating on age would
|
||||
// add a second failure mode (a box offline from the hub silently stops offering) to fix a window
|
||||
// that closes itself. `HubEscrowKeyCheckedAt` is persisted for diagnosis, not as a gate.
|
||||
// - **AN ABSENT HASH FALLS BACK TO (a)/(b), it does not offer.** "" is what the hub sends for a
|
||||
// legacy hash-less package — one that provably seals no repository password. There is nothing for
|
||||
// (c) to compare, and offering on it would put a permanent screen in front of every legacy box.
|
||||
// This is the one place where "not knowing" resolves to silence, and it does so because an empty
|
||||
// hash is not an unknown: it is the hub positively saying the package covers no key.
|
||||
//
|
||||
// So: fail-closed (offer) on a known difference; fall back on a hash never learned. Pinned by
|
||||
// TestR241_ScenarioD_* and TestR241_StaleComparison_*.
|
||||
func (m *Manager) OffsiteRecoveryOffer() bool {
|
||||
if m.settings == nil || !m.settings.GetHubEscrowIdentityPresent() {
|
||||
return false // the hub holds nothing for us — nothing to recover
|
||||
}
|
||||
localHash, hasLocal := m.OffboxRepoPasswordHash()
|
||||
if !hasLocal {
|
||||
return true // (a) no repository password at all — the pristine rebuilt box
|
||||
}
|
||||
if hubHash, _ := m.settings.GetHubEscrowKeySHA256(); hubHash != "" && hubHash != localHash {
|
||||
return true // (c) the hub's package covers a DIFFERENT key than the one we are using
|
||||
}
|
||||
return m.OffboxOrphaned() // (b) a password exists but the inherited history will not open under it
|
||||
}
|
||||
|
||||
// OffboxReportStatus returns the offsite summary for the hub report.
|
||||
//
|
||||
// nil = nothing to say (not configured, and nothing to ask for) — the hub's checker treats absence as
|
||||
// "nothing to watch"; pre-v0.109 reports look the same.
|
||||
//
|
||||
// v0.199.0: there is now ONE case where an UNCONFIGURED box still reports an object — the stranded
|
||||
// rebuild, which DECLARES OffsiteStateNeedsCredential rather than leaving the hub to deduce it from a
|
||||
// silence. An absent object has FOUR meanings (never configured / mid-restart / a transient config
|
||||
// read failure / rebuilt-and-stranded) and the hub cannot tell them apart; the box can.
|
||||
//
|
||||
// WHY THE DECLARATION IS INERT TO EVERY EXISTING READER, established from their code rather than
|
||||
// assumed: it carries Enabled=false and zero quota/size, and the hub's OffsiteChecker gates
|
||||
// `isStale` on `!off.Enabled` (returns false) and `fillBand` on a zero quota/size (returns OK). So it
|
||||
// raises no staleness and no fill alarm on a new hub OR an old one, and an unknown `state` string is
|
||||
// ignored by encoding/json. The ONE reader that would have misread it is the store's
|
||||
// `reportHasOffsite` ("presence == applied-on-the-box"), which hub v0.96.0 tightens to require
|
||||
// enabled=true — provably a no-op for every report shape that exists today, because this function has
|
||||
// never emitted a disabled object before.
|
||||
func (m *Manager) OffboxReportStatus() *OffboxReportStatus {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
// R-241: the HOLDING state is declared before the enabled/disabled split, because a held target IS
|
||||
// enabled — the customer wants off-site backups; what is missing is the key. Reported with
|
||||
// Enabled=false so every existing hub reader treats it exactly as the stranded declaration (see
|
||||
// OffsiteStateAwaitingRecoveryKey), while the string names the difference for anything that looks.
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
return &OffboxReportStatus{Enabled: false, State: OffsiteStateAwaitingRecoveryKey, EscrowState: t.EscrowState}
|
||||
}
|
||||
if t == nil || !t.Enabled {
|
||||
if m.needsOffsiteCredential(t) {
|
||||
return &OffboxReportStatus{Enabled: false, State: OffsiteStateNeedsCredential}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
return &OffboxReportStatus{
|
||||
Enabled: true, EscrowState: t.EscrowState, LastRun: t.LastRun, LastStatus: t.LastStatus,
|
||||
LastSuccess: t.LastSuccess,
|
||||
SnapshotCount: t.SnapshotCount, RepoSizeBytes: t.RepoSizeBytes, QuotaGB: t.QuotaGB,
|
||||
AbandonPurgeRequested: t.AbandonPurgeRequested, // R-241: declared until the hub drops the package
|
||||
}
|
||||
}
|
||||
|
||||
@@ -998,6 +1716,7 @@ func (m *Manager) offboxRecordStats(ctx context.Context, base, env []string) int
|
||||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.RepoSizeHuman = human
|
||||
o.RepoSizeBytes = st.TotalSize
|
||||
o.StatsKnown = true // R-225
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -23,19 +23,38 @@ type offbox3aProvider struct {
|
||||
hdd map[string]string
|
||||
binds map[string][]ClassifiedBind
|
||||
has map[string]bool
|
||||
// deployed is OPT-IN and defaults to nil, so every existing fixture keeps ListDeployedStacks()
|
||||
// returning nil and nothing about their behaviour moves. R-234's classification is the only
|
||||
// thing that needs a real deployed set.
|
||||
deployed map[string]bool
|
||||
}
|
||||
|
||||
func (p *offbox3aProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *offbox3aProvider) ListDeployedStacks() []StackSummary { return nil }
|
||||
func (p *offbox3aProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *offbox3aProvider) GetStackHDDPath(n string) string { return p.hdd[n] }
|
||||
func (p *offbox3aProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *offbox3aProvider) StopStack(string) error { return nil }
|
||||
func (p *offbox3aProvider) StartStack(string) error { return nil }
|
||||
func (p *offbox3aProvider) RefreshAndIsRunning(string) bool { return false }
|
||||
func (p *offbox3aProvider) GetStackRecoveryInfo(string) (RecoveryInfo, bool) { return RecoveryInfo{}, false }
|
||||
func (p *offbox3aProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *offbox3aProvider) ListDeployedStacks() []StackSummary {
|
||||
if len(p.deployed) == 0 {
|
||||
return nil
|
||||
}
|
||||
out := make([]StackSummary, 0, len(p.deployed))
|
||||
for n := range p.deployed {
|
||||
out = append(out, StackSummary{Name: n})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *offbox3aProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *offbox3aProvider) GetStackHDDPath(n string) string { return p.hdd[n] }
|
||||
func (p *offbox3aProvider) GetImportRoot() string { return "" } // R-75: no import binds in this fixture
|
||||
func (p *offbox3aProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *offbox3aProvider) StopStack(string) error { return nil }
|
||||
func (p *offbox3aProvider) StartStack(string) error { return nil }
|
||||
func (p *offbox3aProvider) RefreshAndIsRunning(string) bool { return false }
|
||||
func (p *offbox3aProvider) GetStackRecoveryInfo(string) (RecoveryInfo, bool) {
|
||||
return RecoveryInfo{}, false
|
||||
}
|
||||
func (p *offbox3aProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *offbox3aProvider) RecreateStackFromUnit(_, _ string, _ map[string]string) error { return nil }
|
||||
func (p *offbox3aProvider) RecreateStackDefinitionFromUnit(_, _ string, _ map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *offbox3aProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *offbox3aProvider) GetStackClassifiedBinds(n string) ([]ClassifiedBind, bool) {
|
||||
return p.binds[n], p.has[n]
|
||||
}
|
||||
|
||||
@@ -0,0 +1,299 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// ABANDONMENT — deciding to give up the old off-site history is a finishable thing (R-241, v0.206.0).
|
||||
//
|
||||
// THE PROBLEM THIS SOLVES. `resetOrphanedRepo` renamed the remote store aside and touched neither the
|
||||
// escrow nor the key, so the hub went on holding a sealed package for a key the box no longer used.
|
||||
// Shape (c) compares those two, finds them different, and offers recovery — correctly, and for ever.
|
||||
// A customer who has already said "I do not want the old data" would be asked again at every login.
|
||||
//
|
||||
// THE OPERATOR'S RULING (2026-08-07) is that the answer is NOT a "they decided" flag. A flag would
|
||||
// leave the box in a state that is genuinely wrong (the hub holding a package for a key nobody uses)
|
||||
// and paper over it. Instead the decision starts a **14-day countdown**, at the end of which the
|
||||
// set-aside store and the sealed package that protects it are removed TOGETHER — after which there is
|
||||
// nothing left to compare and nothing left to ask about. **Fix the state, do not remember that it is
|
||||
// wrong.**
|
||||
//
|
||||
// THE GRACE IS REAL, NOT DECORATIVE. The recovery offer stays reachable for the whole window; that is
|
||||
// the change-of-mind path (Scenario G). A grace period during which recovery is impossible would be
|
||||
// theatre.
|
||||
|
||||
// abandonGraceDays is the countdown the operator set. Reminders fire at 5, 3 and 1 days (see
|
||||
// AbandonRemindAtDays) — visible, reversible, and running out in public.
|
||||
const abandonGraceDays = 14
|
||||
|
||||
// AbandonGraceDays is the exported grace, for the customer-facing copy. The confirmation screen must
|
||||
// state the SAME number the countdown uses — a literal typed into prose is how a promise drifts away
|
||||
// from the code that keeps it.
|
||||
const AbandonGraceDays = abandonGraceDays
|
||||
|
||||
// AbandonRemindAtDays are the remaining-day marks at which the abandoning box reminds the customer.
|
||||
// Descending, so the surface can pick the first one that has been reached.
|
||||
var AbandonRemindAtDays = []int{5, 3, 1}
|
||||
|
||||
// abandonNow is the countdown's clock seam. Tests inject; nil → time.Now. It exists so the terminal
|
||||
// step can be driven deterministically — §7.4 forbids shortening a live timer to watch it fire,
|
||||
// because that is how an irreversible step gets tested once and regretted once.
|
||||
func (m *Manager) abandonNow() time.Time {
|
||||
if m.offboxNow != nil {
|
||||
return m.offboxNow()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
// SetOffboxClock injects the abandonment clock (tests only).
|
||||
func (m *Manager) SetOffboxClock(fn func() time.Time) { m.offboxNow = fn }
|
||||
|
||||
// startAbandonCountdown records the decision and the date the terminal step will run. Called by
|
||||
// resetOrphanedRepo AFTER the move-aside has succeeded — a countdown started before the store has
|
||||
// actually moved would count down to deleting a path that does not exist.
|
||||
func (m *Manager) startAbandonCountdown(setAsidePath string) {
|
||||
now := m.abandonNow().UTC()
|
||||
due := now.AddDate(0, 0, abandonGraceDays)
|
||||
// R-302: pin the hub's escrow key fingerprint HERE, at the decision — the one moment it is a fact
|
||||
// rather than something inferred later from an adjacent value. From now on the banner asks exactly
|
||||
// one question, "is the hub still holding that same package?", instead of guessing which key is
|
||||
// which. Written once and never refreshed: a field re-read at render answers a different question
|
||||
// and would silently restore the defect this replaces.
|
||||
pinned := ""
|
||||
if m.settings != nil {
|
||||
pinned, _ = m.settings.GetHubEscrowKeySHA256()
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonStartedAt = now.Format(time.RFC3339)
|
||||
o.AbandonAt = due.Format(time.RFC3339)
|
||||
o.AbandonRepoPath = setAsidePath
|
||||
o.AbandonPurgeRequested = false
|
||||
o.AbandonPinnedEscrowKeySHA256 = pinned
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not record the abandonment countdown: %v", err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment countdown started: the set-aside history at %s and the hub's sealed package "+
|
||||
"are removed together on %s (%d days). The recovery screen stays reachable until then.",
|
||||
setAsidePath, due.Format("2006-01-02"), abandonGraceDays)
|
||||
}
|
||||
|
||||
// AbandonState is the surface's read model. Zero value = nothing in progress.
|
||||
type AbandonState struct {
|
||||
Active bool // a countdown is running
|
||||
StartedAt time.Time //
|
||||
DueAt time.Time // when the terminal step runs
|
||||
DaysLeft int // ceiling, so "0 days left" only ever means "today"
|
||||
RepoPath string // the set-aside store awaiting deletion
|
||||
PurgeRequested bool // the store is gone; awaiting the hub to drop the sealed package
|
||||
// RetrievalStillOffered (R-302) — may the banner still say the set-aside copies can be retrieved
|
||||
// with the recovery code? TRUE only while the hub is holding the SAME sealed package it held when
|
||||
// the customer decided. Derived here, once, so the banner and anything else asking cannot disagree.
|
||||
//
|
||||
// FALSE covers: the package was replaced after the decision (a fresh escrow ceremony — the act that
|
||||
// cost both demo boxes their history); the hub reports an empty hash (a legacy package sealing no
|
||||
// repository password); and a countdown started before R-302, which carries no pin. All three are
|
||||
// "we cannot see that this is still true", and all three must read as such rather than as a promise.
|
||||
RetrievalStillOffered bool
|
||||
}
|
||||
|
||||
// AbandonStatus reports the countdown for the UI and the report. It never mutates.
|
||||
func (m *Manager) AbandonStatus() AbandonState {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil {
|
||||
return AbandonState{}
|
||||
}
|
||||
st := AbandonState{RepoPath: t.AbandonRepoPath, PurgeRequested: t.AbandonPurgeRequested}
|
||||
if t.AbandonAt == "" {
|
||||
return st
|
||||
}
|
||||
due, err := time.Parse(time.RFC3339, t.AbandonAt)
|
||||
if err != nil {
|
||||
// A malformed stamp must not silently mean "never due" — that would strand the store for ever
|
||||
// with a countdown the customer can see and nothing behind it.
|
||||
m.logger.Printf("[WARN] [offbox] abandonment due-date is unparseable (%q) — treating the countdown as NOT running: %v", t.AbandonAt, err)
|
||||
return st
|
||||
}
|
||||
st.Active, st.DueAt = true, due
|
||||
if s, serr := time.Parse(time.RFC3339, t.AbandonStartedAt); serr == nil {
|
||||
st.StartedAt = s
|
||||
}
|
||||
// R-302: the pinned fingerprint vs what the hub reports NOW. Both must be non-empty and equal.
|
||||
// Empty on either side is "we could not see", never "they match" — the settings comment on
|
||||
// HubEscrowKeySHA256 establishes that the hub sends "" for a package sealing no repo password.
|
||||
if cur, _ := m.settings.GetHubEscrowKeySHA256(); cur != "" &&
|
||||
t.AbandonPinnedEscrowKeySHA256 != "" && cur == t.AbandonPinnedEscrowKeySHA256 {
|
||||
st.RetrievalStillOffered = true
|
||||
}
|
||||
// Ceiling: a countdown with 30 minutes left says "1 day", never "0". Zero is reserved for due.
|
||||
remaining := due.Sub(m.abandonNow())
|
||||
if remaining <= 0 {
|
||||
st.DaysLeft = 0
|
||||
} else {
|
||||
st.DaysLeft = int((remaining + 24*time.Hour - time.Nanosecond) / (24 * time.Hour))
|
||||
}
|
||||
return st
|
||||
}
|
||||
|
||||
// CancelAbandon stops a running countdown — the change-of-mind path (Scenario G). Called when a
|
||||
// recovery succeeds: the customer has their code after all, and the history they were about to give
|
||||
// up is exactly what the code opens.
|
||||
//
|
||||
// It clears the schedule but KEEPS AbandonRepoPath, so the set-aside store remains nameable on the
|
||||
// backups page. Nothing has been deleted at this point by construction — the terminal step is the
|
||||
// only thing that deletes, and it has not run.
|
||||
func (m *Manager) CancelAbandon(reason string) {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || (t.AbandonAt == "" && !t.AbandonPurgeRequested) {
|
||||
return // nothing running — silent, so a healthy recovery does not log about a countdown
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonStartedAt, o.AbandonAt = "", ""
|
||||
o.AbandonPurgeRequested = false
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not cancel the abandonment countdown: %v", err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment countdown CANCELLED (%s) — the set-aside history at %s is kept and nothing was deleted", reason, t.AbandonRepoPath)
|
||||
}
|
||||
|
||||
// AbandonSweep is the daily terminal step. It is the ONLY thing in the product that deletes a
|
||||
// customer's off-site history, and it does so on a date the customer was shown.
|
||||
//
|
||||
// ⚠ IT REMOVES BOTH HALVES OR NEITHER — Scenario F. The set-aside store and the sealed package that
|
||||
// protects it are the two halves of one thing; removing only the store leaves the hub holding a
|
||||
// package for a key that opens nothing, and removing only the package leaves ciphertext nobody can
|
||||
// ever decrypt. Either is a state that asks a question nobody can answer.
|
||||
//
|
||||
// The two halves cannot be made atomic across two machines, so this is a two-phase commit with the
|
||||
// STORE FIRST and a durable marker: delete the remote store, record AbandonPurgeRequested, and keep
|
||||
// declaring it in the report until the hub's ACK stops reporting a superseded package. A crash
|
||||
// between the two leaves the marker set and the next sweep re-declares — it never leaves the pair
|
||||
// half-removed and silent.
|
||||
//
|
||||
// Returns (deleted, err). deleted=false with err=nil is the normal "nothing due" case.
|
||||
func (m *Manager) AbandonSweep(ctx context.Context) (bool, error) {
|
||||
st := m.AbandonStatus()
|
||||
// Phase 2 outstanding: the store is gone, the hub has not confirmed. Re-declare and wait.
|
||||
if st.PurgeRequested {
|
||||
m.logger.Printf("[DEBUG] [offbox] abandonment: the set-aside store is deleted; awaiting the hub to drop the sealed package")
|
||||
return false, nil
|
||||
}
|
||||
if !st.Active || st.DueAt.After(m.abandonNow()) {
|
||||
return false, nil // not due — quiet by construction on every healthy box
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || t.AbandonRepoPath == "" {
|
||||
m.logger.Printf("[WARN] [offbox] abandonment is due but no set-aside path is recorded — nothing deleted; clearing the countdown so it does not retry for ever")
|
||||
m.CancelAbandon("no set-aside path recorded")
|
||||
return false, fmt.Errorf("abandonment due with no recorded path")
|
||||
}
|
||||
port := t.Port
|
||||
if port == 0 {
|
||||
port = 22
|
||||
}
|
||||
m.logger.Printf("[WARN] [offbox] abandonment DUE — deleting the set-aside off-site history at %s (chosen by the customer on %s; this is irreversible)",
|
||||
t.AbandonRepoPath, st.StartedAt.Format("2006-01-02"))
|
||||
out, err := m.sshRunner()(ctx, t.Host, t.User, port, m.offboxKeyPath(), m.offboxKnownHosts(),
|
||||
"rm -rf "+shellQuote(t.AbandonRepoPath))
|
||||
if err != nil {
|
||||
// NOT cleared: a transport failure must retry tomorrow, not silently abandon the abandonment.
|
||||
m.logger.Printf("[ERROR] [offbox] abandonment: deleting the set-aside history failed — the countdown stays due and retries: %v: %s", err, truncate(out))
|
||||
return false, fmt.Errorf("delete set-aside history: %w", err)
|
||||
}
|
||||
// Phase 1 done. Record it durably BEFORE anything else, so a crash here re-declares rather than
|
||||
// forgetting that the store is already gone.
|
||||
if uerr := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPurgeRequested = true
|
||||
o.AbandonAt = "" // the schedule has fired; the marker now drives the rest
|
||||
}); uerr != nil {
|
||||
m.logger.Printf("[ERROR] [offbox] abandonment: the store was deleted but the marker could not be saved — the hub's package may outlive it: %v", uerr)
|
||||
return true, uerr
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment: set-aside history deleted; requesting the hub to drop the sealed package that protected it")
|
||||
if m.offboxOrphanEvent != nil {
|
||||
m.offboxOrphanEvent("offbox_abandon_completed", t.AbandonRepoPath)
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// ClearAbandonPurgeIfConfirmed closes the two-phase commit: once the hub's ACK stops reporting a
|
||||
// superseded package, both halves are gone and the abandonment is finished. Called from the ACK path.
|
||||
//
|
||||
// This is what makes §2.1 work without a "they decided" flag: afterwards the hub holds a package for
|
||||
// the key the box is actually using (or none at all), shape (c) has nothing to compare, and the
|
||||
// recovery offer falls silent on its own — because the state is right, not because something is
|
||||
// remembering that it once was not.
|
||||
func (m *Manager) ClearAbandonPurgeIfConfirmed(supersededPresent bool) {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || !t.AbandonPurgeRequested || supersededPresent {
|
||||
return
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPurgeRequested = false
|
||||
o.AbandonRepoPath = ""
|
||||
o.AbandonStartedAt = ""
|
||||
o.OrphanedRenamedTo = ""
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not close out the abandonment: %v", err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] abandonment COMPLETE — the set-aside history and the sealed package that protected it are both gone; nothing further to ask about")
|
||||
}
|
||||
|
||||
// ── OPERATOR CONTROL (§7.5) ─────────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// The automatic 30-day abandonment is deliberately NOT built (see R-245). What IS built is the path
|
||||
// that actually happens: **the customer gets in touch.** Someone who cannot find their recovery code
|
||||
// rings support, and support needs something to press — either "give them longer" or "stop it".
|
||||
//
|
||||
// Both live on the controller CLI rather than in the customer UI, deliberately: extending a deletion
|
||||
// the customer asked for is an operator judgement, not a self-service button, and a customer who
|
||||
// wants it stopped already has the self-service route — they recover with their code, which cancels
|
||||
// it (Scenario G).
|
||||
|
||||
// ExtendAbandon pushes the terminal step out by `days` from NOW. Returns the new due date.
|
||||
//
|
||||
// It refuses when no countdown is running: extending nothing would print a reassuring date for a
|
||||
// deletion that was never scheduled, which is the kind of comfort this project keeps removing.
|
||||
func (m *Manager) ExtendAbandon(days int) (time.Time, error) {
|
||||
if days <= 0 {
|
||||
return time.Time{}, fmt.Errorf("the extension must be a positive number of days")
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
if st.PurgeRequested {
|
||||
return time.Time{}, fmt.Errorf("too late: the set-aside history has already been deleted and only the sealed package is still being removed")
|
||||
}
|
||||
return time.Time{}, fmt.Errorf("no abandonment countdown is running on this box — nothing to extend")
|
||||
}
|
||||
due := m.abandonNow().UTC().AddDate(0, 0, days)
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonAt = due.Format(time.RFC3339)
|
||||
}); err != nil {
|
||||
return time.Time{}, fmt.Errorf("record the extension: %w", err)
|
||||
}
|
||||
m.logger.Printf("[WARN] [offbox] abandonment EXTENDED by an operator: the set-aside history at %s is now deleted on %s (was %s)",
|
||||
st.RepoPath, due.Format("2006-01-02"), st.DueAt.Format("2006-01-02"))
|
||||
return due, nil
|
||||
}
|
||||
|
||||
// StopAbandon cancels the countdown outright — the operator's version of Scenario G, for the
|
||||
// customer who telephoned instead of finding their code. The set-aside history is kept and nothing
|
||||
// is deleted; it is `CancelAbandon` with an operator's reason and a refusal when nothing is running,
|
||||
// so an operator never gets a silent no-op they might read as success.
|
||||
func (m *Manager) StopAbandon() error {
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
if st.PurgeRequested {
|
||||
return fmt.Errorf("too late: the set-aside history has already been deleted")
|
||||
}
|
||||
return fmt.Errorf("no abandonment countdown is running on this box — nothing to stop")
|
||||
}
|
||||
m.CancelAbandon("stopped by an operator")
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,208 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// ── R-302 — THE BANNER PROMISES ONLY WHAT THE BOX CAN STILL SEE IS TRUE ─────────────────────────
|
||||
//
|
||||
// The abandon banner said "until then you can still retrieve them with your recovery code",
|
||||
// unconditionally, on every page. Yesterday's reading proved that false on a reachable state.
|
||||
//
|
||||
// THE CONDITION IS A PIN, NOT A COMPARISON AGAINST THE CURRENT KEY, and the difference is the whole
|
||||
// design. The obvious proxy — "does the hub hold a key different from the one I use?" — asks about the
|
||||
// wrong key: the set-aside copies were written under an OLDER key the box no longer has, which is why
|
||||
// they were set aside. On a twice-rebuilt box the proxy answers "yes, promise it" about copies no key
|
||||
// on file can open. The pin instead records the package the hub held AT THE DECISION and asks only
|
||||
// "is the hub still holding that same one?".
|
||||
//
|
||||
// ⚠ THE PIN IS A RECORDED ASSUMPTION. It presumes the package held at the decision is the one that
|
||||
// opens the set-aside copies. Nothing on the box records which key wrote them. See the field comment
|
||||
// on settings.AbandonPinnedEscrowKeySHA256.
|
||||
//
|
||||
// The countdown is never started, shortened or triggered on a real machine — the clock is injected.
|
||||
|
||||
const pinnedHubKey = "1111111111111111111111111111111111111111111111111111111111111111"
|
||||
const replacedHubKey = "2222222222222222222222222222222222222222222222222222222222222222"
|
||||
|
||||
// startedCountdown drives the PRODUCTION path (ResetOrphanedRepo → resetOrphanedRepo →
|
||||
// startAbandonCountdown) so the pin cannot be written by tests alone while the live path never sets
|
||||
// it — the inert-seam shape that has shipped here before, fully green.
|
||||
func startedCountdown(t *testing.T, hubKeyAtDecision string) (*Manager, *settings.Settings, time.Time) {
|
||||
t.Helper()
|
||||
start := time.Date(2026, 8, 12, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, _ := abandonFixture(t, start)
|
||||
if err := sett.SetHubEscrowKeySHA256(hubKeyAtDecision, start.Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatalf("the production reset path failed: %v", err)
|
||||
}
|
||||
return m, sett, start
|
||||
}
|
||||
|
||||
// PRODUCTION WIRING: the live decision path writes the pin. If this fails, every render test below is
|
||||
// testing a field nothing sets.
|
||||
func TestR302_ProductionResetPathWritesThePin(t *testing.T) {
|
||||
_, sett, _ := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
got := sett.GetOffboxTarget().AbandonPinnedEscrowKeySHA256
|
||||
if got != pinnedHubKey {
|
||||
t.Fatalf("pinned fingerprint = %q, want the hub key cached at the decision (%q). The whole "+
|
||||
"design is that this is recorded when it is a fact; if the live path does not write it, "+
|
||||
"the banner falls to the cautious branch for ever and the grace period becomes theatre",
|
||||
got, pinnedHubKey)
|
||||
}
|
||||
if sett.GetOffboxTarget().AbandonAt == "" {
|
||||
t.Error("no countdown recorded — the fixture is not exercising the path it claims to")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO A — package unchanged since the decision → the promise stands ──────────────────────
|
||||
//
|
||||
// RED-PROOF: force the condition false (drop the `cur == t.AbandonPinnedEscrowKeySHA256` arm) and this
|
||||
// fails — a customer who can genuinely still change their mind loses the clause, which the code says
|
||||
// explicitly must not happen ("a grace period during which recovery is impossible would be theatre").
|
||||
func TestR302_ScenarioA_PackageUnchanged_RetrievalStillOffered(t *testing.T) {
|
||||
m, _, _ := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("countdown not active")
|
||||
}
|
||||
if !st.RetrievalStillOffered {
|
||||
t.Error("the hub still holds the same package it held at the decision, so the customer really " +
|
||||
"can still change their mind — the promise must stand")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO B — the package was REPLACED after the decision → promise withdrawn ────────────────
|
||||
//
|
||||
// This is the act that cost both demo boxes their history on 2026-08-04: a fresh escrow ceremony
|
||||
// supersedes the package, and the old key it covered is unreachable (superseded packages grant no
|
||||
// read path — hub store.go's own comment).
|
||||
//
|
||||
// RED-PROOF: re-read the pin at render (compare `cur` against itself, i.e. use the CURRENT cached
|
||||
// value on both sides) and this fails — the promise returns, which is today's defect.
|
||||
func TestR302_ScenarioB_PackageReplaced_PromiseWithdrawn(t *testing.T) {
|
||||
m, sett, start := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
// A fresh ceremony after the decision.
|
||||
if err := sett.SetHubEscrowKeySHA256(replacedHubKey, start.Add(48*time.Hour).Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := m.AbandonStatus(); st.RetrievalStillOffered {
|
||||
t.Error("the hub's package was replaced after the customer decided, so the key that opened the " +
|
||||
"set-aside copies is no longer served — the banner must stop promising retrieval")
|
||||
}
|
||||
// The pin itself must NOT have moved: it is written once, at the decision.
|
||||
if got := sett.GetOffboxTarget().AbandonPinnedEscrowKeySHA256; got != pinnedHubKey {
|
||||
t.Errorf("the pin was refreshed to %q — a field re-read later answers a different question and "+
|
||||
"silently restores the defect this replaces", got)
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO D — a countdown started BEFORE this shipped carries no pin ─────────────────────────
|
||||
//
|
||||
// RED-PROOF: backfill the pin from the current cached value when it is empty and this fails — a legacy
|
||||
// countdown gets promised at, asserting as recorded-at-the-decision something read long afterwards.
|
||||
func TestR302_ScenarioD_LegacyCountdownWithoutAPin_TakesTheCautiousBranch(t *testing.T) {
|
||||
m, sett, _ := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
// Model the pre-R-302 on-disk shape: a live countdown, no pin.
|
||||
if err := sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPinnedEscrowKeySHA256 = ""
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("countdown should still be running")
|
||||
}
|
||||
if st.RetrievalStillOffered {
|
||||
t.Error("a countdown with no pin was promised at. There is no honest way to know whether the " +
|
||||
"hub's package is still the one from the decision, and the cautious answer is the only one " +
|
||||
"available")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO E — pinned present, hub's cached value EMPTY → cautious ────────────────────────────
|
||||
//
|
||||
// The hub sends "" for a legacy package that provably seals no repository password. Empty is a
|
||||
// measurement, not a match.
|
||||
//
|
||||
// RED-PROOF: treat empty as equal (drop the `cur != ""` arm) and this fails.
|
||||
func TestR302_ScenarioE_EmptyHubHash_IsNotAMatch(t *testing.T) {
|
||||
m, sett, start := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
if err := sett.SetHubEscrowKeySHA256("", start.Add(time.Hour).Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := m.AbandonStatus(); st.RetrievalStillOffered {
|
||||
t.Error("an EMPTY hub hash was read as a match. It means the hub holds a package that seals no " +
|
||||
"repository password — the opposite of evidence that retrieval works")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO F — no countdown → nothing about retrieval is claimed at all ───────────────────────
|
||||
func TestR302_ScenarioF_NoCountdown_NoClaim(t *testing.T) {
|
||||
start := time.Date(2026, 8, 12, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
|
||||
st := m.AbandonStatus()
|
||||
if st.Active {
|
||||
t.Fatal("no countdown was started, yet one is reported active")
|
||||
}
|
||||
if st.RetrievalStillOffered {
|
||||
t.Error("retrieval was offered with no countdown running — the flag must be meaningless " +
|
||||
"outside an abandonment, not default-true")
|
||||
}
|
||||
}
|
||||
|
||||
// The pin is a hash of a secret. It must never reach a customer-facing surface or the report; this
|
||||
// pins that it is not accidentally exported through the read model.
|
||||
func TestR302_PinIsNotExposedThroughTheReadModel(t *testing.T) {
|
||||
m, _, _ := startedCountdown(t, pinnedHubKey)
|
||||
st := m.AbandonStatus()
|
||||
if st.RepoPath == pinnedHubKey {
|
||||
t.Fatal("the pin leaked into RepoPath")
|
||||
}
|
||||
// AbandonState carries a BOOLEAN verdict, never the fingerprint itself.
|
||||
if got := st.RetrievalStillOffered; got != true && got != false {
|
||||
t.Fatal("unreachable")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO E, the case that actually bites — BOTH sides empty ─────────────────────────────────
|
||||
//
|
||||
// A legacy countdown (no pin) on a box whose hub reports an empty hash (a package sealing no repo
|
||||
// password). "" == "" is the equality that would quietly become a promise, and it is the ONLY state
|
||||
// where dropping the emptiness guards changes the answer — TestR302_ScenarioE above passes even with
|
||||
// them removed, because its pin is non-empty so the equality fails on its own. That test guards the
|
||||
// sentence; this one guards the claim.
|
||||
//
|
||||
// RED-PROOF: drop either `cur != ""` or `t.AbandonPinnedEscrowKeySHA256 != ""` and this fails.
|
||||
func TestR302_ScenarioE2_BothSidesEmpty_IsNotAMatch(t *testing.T) {
|
||||
m, sett, start := startedCountdown(t, pinnedHubKey)
|
||||
|
||||
if err := sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.AbandonPinnedEscrowKeySHA256 = "" // legacy countdown, no pin
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetHubEscrowKeySHA256("", start.Add(time.Hour).Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("countdown should still be running")
|
||||
}
|
||||
if st.RetrievalStillOffered {
|
||||
t.Error("two absences compared equal and became a promise. Empty means we could not see; two " +
|
||||
"things we could not see are not a match")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,335 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-241 — abandoning starts a countdown that ENDS THE QUESTION (Scenarios E, F, G).
|
||||
//
|
||||
// The countdown is driven by an injected clock throughout. §7.4 forbids shortening a live timer to
|
||||
// watch the terminal step fire: it is the only thing in the product that deletes a customer's
|
||||
// off-site history, and a step tested once on real data is a step regretted once.
|
||||
|
||||
// abandonFixture: an orphaned, configured box holding a key, with the hub holding a package for a
|
||||
// DIFFERENT key — i.e. shape (c) is live and the customer is being offered recovery.
|
||||
// Returns the manager and a recorder of every remote shell command issued.
|
||||
type sshRecorder struct{ cmds []string }
|
||||
|
||||
func (r *sshRecorder) run(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
|
||||
r.cmds = append(r.cmds, remoteCmd)
|
||||
return []byte(""), nil
|
||||
}
|
||||
|
||||
func abandonFixture(t *testing.T, now time.Time) (*Manager, *settings.Settings, *sshRecorder) {
|
||||
t.Helper()
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, now.Format(time.RFC3339)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.EscrowState = "escrowed"
|
||||
o.RepoState = "orphaned"
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rec := &sshRecorder{}
|
||||
m.SetOffboxSSH(rec.run)
|
||||
m.SetOffboxRunner(func(ctx context.Context, env []string, args ...string) ([]byte, error) { return []byte(""), nil })
|
||||
m.SetOffboxClock(func() time.Time { return now })
|
||||
return m, sett, rec
|
||||
}
|
||||
|
||||
// ── SCENARIO E — abandoning sets aside, keeps the package, starts a countdown, stays reversible ──
|
||||
func TestR241_ScenarioE_AbandonStartsAReversibleCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, rec := abandonFixture(t, start)
|
||||
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatalf("abandon: %v", err)
|
||||
}
|
||||
// The store was MOVED, not deleted — no rm anywhere in this phase.
|
||||
joined := strings.Join(rec.cmds, " | ")
|
||||
if !strings.Contains(joined, "mv ") {
|
||||
t.Errorf("the old store must be moved aside; commands were: %s", joined)
|
||||
}
|
||||
if strings.Contains(joined, "rm -rf") {
|
||||
t.Fatalf("NOTHING may be deleted when the customer abandons — only at the end of the grace. Commands: %s", joined)
|
||||
}
|
||||
st := m.AbandonStatus()
|
||||
if !st.Active {
|
||||
t.Fatal("a countdown must be running after an abandonment")
|
||||
}
|
||||
if got := st.DueAt.Sub(start); got != abandonGraceDays*24*time.Hour {
|
||||
t.Errorf("countdown length = %v, want %d days", got, abandonGraceDays)
|
||||
}
|
||||
if st.DaysLeft != abandonGraceDays {
|
||||
t.Errorf("DaysLeft = %d, want %d", st.DaysLeft, abandonGraceDays)
|
||||
}
|
||||
if st.RepoPath == "" {
|
||||
t.Error("the set-aside path must be recorded, or the terminal step has nothing to delete")
|
||||
}
|
||||
// THE GRACE IS REAL: the recovery offer stays reachable for the whole window. A grace in which
|
||||
// recovery is impossible would be decorative.
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("the recovery offer MUST stay reachable during the grace — that is the change-of-mind path")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO G — changing your mind inside the window ───────────────────────────────────────────
|
||||
//
|
||||
// RED-PROOF: make the countdown uncancellable (delete the body of CancelAbandon). The countdown then
|
||||
// survives a successful recovery and this test fails — a customer who proved they hold their code
|
||||
// would still have the history deleted under them.
|
||||
func TestR241_ScenarioG_RecoveryInsideTheWindowCancelsTheCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
day6 := start.AddDate(0, 0, 6)
|
||||
m.SetOffboxClock(func() time.Time { return day6 })
|
||||
if st := m.AbandonStatus(); !st.Active || st.DaysLeft != 8 {
|
||||
t.Fatalf("precondition: day 6 of 14 should leave 8 days, got %+v", st)
|
||||
}
|
||||
pathBefore := m.AbandonStatus().RepoPath
|
||||
|
||||
m.CancelAbandon("the customer recovered with their code")
|
||||
|
||||
st := m.AbandonStatus()
|
||||
if st.Active {
|
||||
t.Fatal("a countdown must be cancellable — the customer found their code")
|
||||
}
|
||||
if st.RepoPath != pathBefore {
|
||||
t.Errorf("the set-aside store must stay NAMEABLE after a cancel: got %q want %q", st.RepoPath, pathBefore)
|
||||
}
|
||||
// And a sweep now deletes nothing, on any later date.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 90) })
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if err != nil || deleted {
|
||||
t.Fatalf("a cancelled countdown must never delete: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO F — the countdown ends the question, and removes BOTH halves ────────────────────────
|
||||
//
|
||||
// RED-PROOF (store half): make AbandonSweep skip the rm. The first assertion fails.
|
||||
// RED-PROOF (package half): drop AbandonPurgeRequested from OffboxReportStatus. The declaration
|
||||
// assertion fails — the hub is never asked and the package outlives the store for ever.
|
||||
func TestR241_ScenarioF_TerminalStepRemovesBothHalvesTogether(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, rec := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
setAside := m.AbandonStatus().RepoPath
|
||||
|
||||
// Not due yet — nothing happens, quietly.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 13) })
|
||||
if deleted, err := m.AbandonSweep(context.Background()); deleted || err != nil {
|
||||
t.Fatalf("day 13 must not delete: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
|
||||
// Due.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 14).Add(time.Minute) })
|
||||
rec.cmds = nil
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("terminal step: %v", err)
|
||||
}
|
||||
if !deleted {
|
||||
t.Fatal("the terminal step must delete when due")
|
||||
}
|
||||
// HALF 1: the store is gone.
|
||||
joined := strings.Join(rec.cmds, " | ")
|
||||
if !strings.Contains(joined, "rm -rf") || !strings.Contains(joined, setAside) {
|
||||
t.Fatalf("the set-aside store at %s must be deleted; commands: %s", setAside, joined)
|
||||
}
|
||||
// HALF 2: the hub is ASKED for the package, and keeps being asked until it confirms.
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil || !st.AbandonPurgeRequested {
|
||||
t.Fatalf("the report must declare abandon_purge_requested until the hub drops the package, got %+v", st)
|
||||
}
|
||||
// It repeats — a lost request must retry rather than leave the pair half-removed.
|
||||
if d2, err2 := m.AbandonSweep(context.Background()); d2 || err2 != nil {
|
||||
t.Fatalf("a second sweep must be a quiet no-op while awaiting the hub: deleted=%v err=%v", d2, err2)
|
||||
}
|
||||
if st2 := m.OffboxReportStatus(); st2 == nil || !st2.AbandonPurgeRequested {
|
||||
t.Fatal("the declaration must persist across sweeps until confirmed")
|
||||
}
|
||||
|
||||
// The hub confirms by no longer reporting a superseded package → the question is over.
|
||||
m.ClearAbandonPurgeIfConfirmed(false)
|
||||
if got := sett.GetOffboxTarget(); got.AbandonPurgeRequested || got.AbandonRepoPath != "" || got.AbandonAt != "" {
|
||||
t.Errorf("the abandonment must be fully closed out, got %+v", got)
|
||||
}
|
||||
if st3 := m.OffboxReportStatus(); st3 != nil && st3.AbandonPurgeRequested {
|
||||
t.Error("the declaration must stop once the hub has confirmed")
|
||||
}
|
||||
}
|
||||
|
||||
// While the hub STILL reports a superseded package, the close-out must not fire — otherwise the box
|
||||
// stops asking and the package outlives the store silently, which is exactly half of Scenario F.
|
||||
func TestR241_PurgeIsNotClosedOutWhileThePackageRemains(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
if _, err := m.AbandonSweep(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.ClearAbandonPurgeIfConfirmed(true) // the hub STILL holds a retained package
|
||||
if !sett.GetOffboxTarget().AbandonPurgeRequested {
|
||||
t.Fatal("the request must stand while the hub still reports a superseded package")
|
||||
}
|
||||
}
|
||||
|
||||
// A transport failure during the terminal step must NOT clear the countdown — it retries tomorrow.
|
||||
// Silently abandoning the abandonment would leave the store for ever with nothing counting down.
|
||||
func TestR241_TerminalStepFailureKeepsTheCountdownDue(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, sett, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetOffboxSSH(func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
|
||||
return []byte("ssh: connect to host nas.local port 22: No route to host"), context.DeadlineExceeded
|
||||
})
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if deleted || err == nil {
|
||||
t.Fatalf("a failed deletion must be reported, not swallowed: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.AbandonAt == "" || got.AbandonPurgeRequested {
|
||||
t.Fatalf("a failed terminal step must leave the countdown DUE and unrequested, got %+v", got)
|
||||
}
|
||||
if !m.AbandonStatus().Active {
|
||||
t.Error("the countdown must still be active so tomorrow's sweep retries")
|
||||
}
|
||||
}
|
||||
|
||||
// Quiet by construction: a box with no countdown does no work and says nothing (§ the daily job's
|
||||
// own contract). Asserted, because "it probably does nothing" is how a sweep with a bug hides.
|
||||
func TestR241_Sweep_QuietWhenNothingDue(t *testing.T) {
|
||||
m, _, rec := abandonFixture(t, time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC))
|
||||
deleted, err := m.AbandonSweep(context.Background())
|
||||
if deleted || err != nil {
|
||||
t.Fatalf("a box with no countdown must be a pure no-op: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
if len(rec.cmds) != 0 {
|
||||
t.Fatalf("a no-op sweep must issue no remote commands, got %v", rec.cmds)
|
||||
}
|
||||
if m.AbandonStatus().Active {
|
||||
t.Error("no countdown should be reported")
|
||||
}
|
||||
}
|
||||
|
||||
// The UNCLAIMED auto-reset must NOT start a customer countdown — nobody decided anything there.
|
||||
// An as-delivered box tidying a stranger's leftover store must not put a 14-day deletion clock on it.
|
||||
func TestR241_UnclaimedAutoResetStartsNoCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
t2 := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t2)
|
||||
if err := m.resetOrphanedRepo(context.Background(), base, env, "auto (unclaimed)"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.AbandonStatus().Active {
|
||||
t.Fatal("the unclaimed auto-reset must not start a customer abandonment countdown")
|
||||
}
|
||||
}
|
||||
|
||||
// ── §7.5 — THE OPERATOR LEVERS ──────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// The automatic 30-day ending is deliberately NOT built (R-245). These are what IS built: the path
|
||||
// that actually happens is the customer telephoning, and support needs something to press.
|
||||
func TestR241_OperatorCanExtendARunningCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, rec := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
day10 := start.AddDate(0, 0, 10)
|
||||
m.SetOffboxClock(func() time.Time { return day10 })
|
||||
|
||||
due, err := m.ExtendAbandon(30)
|
||||
if err != nil {
|
||||
t.Fatalf("extend: %v", err)
|
||||
}
|
||||
if want := day10.AddDate(0, 0, 30); !due.Equal(want) {
|
||||
t.Errorf("new due = %v, want %v (from NOW, not from the old date)", due, want)
|
||||
}
|
||||
// The original date has passed and nothing is deleted, because the extension moved it.
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
rec.cmds = nil
|
||||
if deleted, serr := m.AbandonSweep(context.Background()); deleted || serr != nil {
|
||||
t.Fatalf("an extended countdown must not fire on the old date: deleted=%v err=%v", deleted, serr)
|
||||
}
|
||||
if len(rec.cmds) != 0 {
|
||||
t.Fatalf("nothing may be deleted after an extension, got %v", rec.cmds)
|
||||
}
|
||||
}
|
||||
|
||||
func TestR241_OperatorCanStopARunningCountdown(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, rec := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.StopAbandon(); err != nil {
|
||||
t.Fatalf("stop: %v", err)
|
||||
}
|
||||
if m.AbandonStatus().Active {
|
||||
t.Fatal("the countdown must be stopped")
|
||||
}
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 90) })
|
||||
rec.cmds = nil
|
||||
if deleted, err := m.AbandonSweep(context.Background()); deleted || err != nil {
|
||||
t.Fatalf("a stopped countdown must never delete: deleted=%v err=%v", deleted, err)
|
||||
}
|
||||
if len(rec.cmds) != 0 {
|
||||
t.Fatalf("a stopped countdown must issue no remote commands, got %v", rec.cmds)
|
||||
}
|
||||
}
|
||||
|
||||
// Both levers REFUSE when nothing is running. A silent no-op is the thing an operator most easily
|
||||
// mistakes for success — they would tell the customer it was handled.
|
||||
func TestR241_OperatorLeversRefuseWhenNothingIsRunning(t *testing.T) {
|
||||
m, _, _ := abandonFixture(t, time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC))
|
||||
if _, err := m.ExtendAbandon(30); err == nil {
|
||||
t.Error("extending a countdown that is not running must be an error, never a quiet success")
|
||||
}
|
||||
if err := m.StopAbandon(); err == nil {
|
||||
t.Error("stopping a countdown that is not running must be an error, never a quiet success")
|
||||
}
|
||||
if _, err := m.ExtendAbandon(0); err == nil {
|
||||
t.Error("a non-positive extension must be refused")
|
||||
}
|
||||
}
|
||||
|
||||
// Once the store is deleted there is nothing left to extend or stop, and saying otherwise would be
|
||||
// the worst kind of reassurance: an operator telling a customer their data is safe when it is gone.
|
||||
func TestR241_OperatorLeversRefuseAfterTheDeletion(t *testing.T) {
|
||||
start := time.Date(2026, 8, 7, 12, 0, 0, 0, time.UTC)
|
||||
m, _, _ := abandonFixture(t, start)
|
||||
if err := m.ResetOrphanedRepo(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetOffboxClock(func() time.Time { return start.AddDate(0, 0, 15) })
|
||||
if _, err := m.AbandonSweep(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := m.ExtendAbandon(30); err == nil {
|
||||
t.Error("extending after the deletion must be refused — there is nothing left to save")
|
||||
}
|
||||
if err := m.StopAbandon(); err == nil {
|
||||
t.Error("stopping after the deletion must be refused — there is nothing left to save")
|
||||
}
|
||||
}
|
||||
@@ -29,13 +29,13 @@ type offboxBlocked struct {
|
||||
// snapshot, plus any Hungarian warnings for capture gaps. It never returns optional/excluded paths
|
||||
// (the TierOffsite filter drops them — §2). Returns (nil, nil) for the legacy / no-provider / no-block
|
||||
// world: offsite stays UNIT-ONLY, byte-identical to pre-v0.134.0 (the SQ5 cost-regression guard).
|
||||
func (m *Manager) offboxCaptureSet(stack string) (extra []string, warns []string) {
|
||||
func (m *Manager) offboxCaptureSet(stack string) (extra []string, warns []string, gaps []string) {
|
||||
if m.stackProvider == nil {
|
||||
return nil, nil // no provider wired → legacy world → unit only
|
||||
return nil, nil, nil // no provider wired → legacy world → unit only
|
||||
}
|
||||
binds, has := m.stackProvider.GetStackClassifiedBinds(stack)
|
||||
if !has {
|
||||
return nil, nil // no backup block → legacy → unit only
|
||||
return nil, nil, nil // no backup block → legacy → unit only
|
||||
}
|
||||
// Resolve against the app's LIVE HDD_PATH (raw — NOT GetAppDrivePath, whose systemDataPath fallback
|
||||
// would resolve userdata onto the wrong drive). Empty ⇒ undeployed / no HDD (decision §2.4):
|
||||
@@ -43,12 +43,11 @@ func (m *Manager) offboxCaptureSet(stack string) (extra []string, warns []string
|
||||
hdd := strings.TrimSpace(m.stackProvider.GetStackHDDPath(stack))
|
||||
if hdd == "" {
|
||||
m.logger.Printf("[WARN] [offbox] %s: not deployed — offsite push is unit-only (mandatory userdata not resolvable)", stack)
|
||||
return nil, []string{fmt.Sprintf("Figyelmeztetés: a(z) %s nincs telepítve — csak a mentési egység került a távoli mentésbe.", stack)}
|
||||
return nil, []string{fmt.Sprintf("Figyelmeztetés: a(z) %s nincs telepítve — csak a mentési egység került a távoli mentésbe.", stack)}, nil
|
||||
}
|
||||
nsRoot := m.namespaceRoot(hdd)
|
||||
cs := appbackup.ComputeCaptureSet(binds, has, appbackup.TierOffsite, nsRoot)
|
||||
cs := appbackup.ComputeCaptureSet(binds, has, appbackup.TierOffsite, nsRoot, m.stackProvider.GetImportRoot())
|
||||
|
||||
var gaps []string
|
||||
// Structurally-refused MANDATORY paths (traversal / bare drive-root / reserved backups/ zone) are
|
||||
// loud ERROR gaps — the path the customer thinks is protected is not in the snapshot.
|
||||
for _, sk := range cs.Skipped {
|
||||
@@ -60,17 +59,26 @@ func (m *Manager) offboxCaptureSet(stack string) (extra []string, warns []string
|
||||
}
|
||||
// Stat-filter (§2.5): a declared mandatory path absent on disk. restic would skip it SILENTLY
|
||||
// (SP-3.4), so drop it from argv AND warn — never a silent "looks backed up but isn't".
|
||||
//
|
||||
// R-203: the class check mirrors tier2_capture.go's ("optional-missing is silent"). It is a NO-OP
|
||||
// today — TierOffsite's tierKeeps() already admits ClassMandatory only, so cs.Paths cannot contain
|
||||
// an optional path here — and it is written anyway so the two tiers read the same and so the
|
||||
// verdict below can never be flipped by an unused optional folder if that filter ever widens.
|
||||
for _, p := range cs.Paths {
|
||||
if _, err := os.Stat(p.Abs); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: mandatory data path missing on disk, skipped from offsite: %s", stack, p.Abs)
|
||||
gaps = append(gaps, p.RelPath)
|
||||
continue
|
||||
if p.Class == appbackup.ClassMandatory {
|
||||
m.logger.Printf("[WARN] [offbox] %s: mandatory data path missing on disk, skipped from offsite: %s", stack, p.Abs)
|
||||
gaps = append(gaps, p.RelPath)
|
||||
}
|
||||
continue // optional-missing is silent (not a gap) — parity with Tier 2
|
||||
}
|
||||
extra = append(extra, p.Abs)
|
||||
}
|
||||
if len(gaps) > 0 {
|
||||
warns = append(warns, fmt.Sprintf("Figyelmeztetés: a(z) %s alkalmazás egyes adatmappái nem kerültek a távoli mentésbe: %s.",
|
||||
// R-234 §7.4: this sits beside the whole-app gap message on the same card, and both now drive
|
||||
// the same `incomplete` verdict — so it says what to do, not only what happened.
|
||||
warns = append(warns, fmt.Sprintf("Figyelmeztetés: a(z) %s alkalmazás egyes adatmappái nem kerültek a távoli mentésbe: %s. Ellenőrizd, hogy a mappák megvannak-e a meghajtón; ha igen és ez a következő mentés után is látszik, szólj az üzemeltetőnek.",
|
||||
stack, strings.Join(gaps, ", ")))
|
||||
}
|
||||
return extra, warns
|
||||
return extra, warns, gaps
|
||||
}
|
||||
|
||||
@@ -0,0 +1,193 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-204 item 4 / R-193 — a REBUILT box declares that it needs an off-site credential, instead of
|
||||
// reporting an absence the hub cannot interpret.
|
||||
//
|
||||
// THE POINT OF THESE TESTS is the conjunction. An absent off-site object has FOUR meanings (never
|
||||
// configured / mid-restart / a transient read failure / rebuilt-and-stranded). The declaration has
|
||||
// one, and it is only sound because BOTH halves are required: a fresh data area AND a hub-held
|
||||
// recovery package. Scenario B is the one that matters most — drop the escrow half and every
|
||||
// un-configured box in the fleet starts asking for a credential.
|
||||
|
||||
// bareManager builds a Manager with NO off-site target and NO repository password — the shape of a
|
||||
// freshly rebuilt box before anything is configured.
|
||||
func bareManager(t *testing.T) (*Manager, *settings.Settings) {
|
||||
t.Helper()
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
sett, err := settings.Load(filepath.Join(dataDir, "settings.json"), lg)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
cfg.Paths.SystemDataPath = filepath.Join(dataDir, "sys")
|
||||
return NewManager(cfg, sett, lg), sett
|
||||
}
|
||||
|
||||
// SCENARIO A — a rebuilt box (fresh data area + a hub-held escrow) DECLARES the state.
|
||||
//
|
||||
// RED-PROOF: remove the `GetHubEscrowIdentityPresent()` condition from needsOffsiteCredential —
|
||||
// Scenario A still passes (it has an escrow), and SCENARIO B FAILS, which is the point: the plausible
|
||||
// wrong fix is to declare on freshness alone, and that would make every un-configured box in the
|
||||
// fleet ask for a credential.
|
||||
func TestOffsiteDeclare_RebuiltBoxDeclaresNeedsCredential(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("a rebuilt box reported NO off-site object — the hub cannot distinguish it from a box that never had off-site backups (this is the defect)")
|
||||
}
|
||||
if st.State != OffsiteStateNeedsCredential {
|
||||
t.Fatalf("declared state = %q, want %q", st.State, OffsiteStateNeedsCredential)
|
||||
}
|
||||
// Enabled MUST be false and the sizes zero — that is what makes the declaration inert to the
|
||||
// hub's existing fill and staleness checkers (and to a pre-upgrade hub).
|
||||
if st.Enabled {
|
||||
t.Error("a declaration must not claim the tier is enabled — the hub's staleness check keys on it")
|
||||
}
|
||||
if st.QuotaGB != 0 || st.RepoSizeBytes != 0 || st.SnapshotCount != 0 {
|
||||
t.Errorf("a declaration must carry zero sizes (fill band keys on them): %+v", st)
|
||||
}
|
||||
// And it must be on the off-site object, not a new top-level field.
|
||||
b, err := json.Marshal(st)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(b), `"state":"needs_credential"`) {
|
||||
t.Fatalf("declared state absent from the marshalled off-site object: %s", b)
|
||||
}
|
||||
if !strings.Contains(string(b), `"enabled":false`) {
|
||||
t.Fatalf("marshalled object must carry enabled:false: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO B — a box that never had off-site backups says NOTHING. This is the guard on the
|
||||
// conjunction; without it the feature churns credentials fleet-wide.
|
||||
func TestOffsiteDeclare_NeverHadOffsiteSaysNothing(t *testing.T) {
|
||||
m, _ := bareManager(t) // fresh data area, but NO hub-held escrow
|
||||
|
||||
if st := m.OffboxReportStatus(); st != nil {
|
||||
t.Fatalf("a box that never had off-site backups DECLARED a need: %+v — every un-configured box in the fleet would now ask for a credential", st)
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO D (R-218) — THE DECLARATION STOPS WHEN THE TIER WORKS, NOT WHEN A KEY EXISTS.
|
||||
//
|
||||
// ⚠ THIS TEST ASSERTED THE OPPOSITE until v0.201.0, and it was green the whole time. It required a
|
||||
// box holding a repository password to stay SILENT — which reads as a sound freshness test and is the
|
||||
// exact opposite on the one path that matters, because installing a repository password is the
|
||||
// RECOVERY SCREEN'S WHOLE JOB. Measured live 2026-08-05 (CAMPAIGN-11 Phase 1): 32 seconds after the
|
||||
// hub re-staged the credential, the customer's successful unlock switched off the mechanism that
|
||||
// would have delivered the coordinates for the key they had just recovered. Deadlock, both halves.
|
||||
//
|
||||
// RED-PROOF: restore the `if _, ok := m.OffboxRepoPasswordHash(); ok { return false }` short-circuit
|
||||
// in needsOffsiteCredential and this test FAILS — the box goes silent again with no target, which is
|
||||
// the deadlock. Demonstrated failing before this test was kept.
|
||||
func TestOffsiteDeclare_StillDeclaresAfterARecoveredKeyIsPlaced(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The post-unlock shape: the recovered repository password is on disk, and there is STILL no
|
||||
// off-site target — so the box cannot use what it just recovered.
|
||||
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(m.offboxPwPath(), []byte("a-recovered-repository-password"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("R-218: the box went SILENT after recovering its key while still having no off-site target — the hub's staged credential is never collected and nothing ever asks again")
|
||||
}
|
||||
if st.State != OffsiteStateNeedsCredential {
|
||||
t.Fatalf("declared state = %q, want %q", st.State, OffsiteStateNeedsCredential)
|
||||
}
|
||||
}
|
||||
|
||||
// SCENARIO E — and once the tier ACTUALLY WORKS the box goes quiet. This is the condition that
|
||||
// replaces the deleted one, and the pair above/below is what makes the deletion safe.
|
||||
func TestOffsiteDeclare_ConfiguredTierIsSilent(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
Schedule: "daily", EscrowState: "escrowed",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("a configured tier must still report its ordinary off-site object")
|
||||
}
|
||||
if st.State == OffsiteStateNeedsCredential {
|
||||
t.Fatal("a box whose tier is configured must not keep asking for a credential")
|
||||
}
|
||||
}
|
||||
|
||||
// A DISABLED target is the customer's own choice, not a rebuild — it must not declare either.
|
||||
func TestOffsiteDeclare_DisabledTargetIsNotStranded(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: false, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if st := m.OffboxReportStatus(); st != nil {
|
||||
t.Fatalf("a deliberately DISABLED target declared a need: %+v", st)
|
||||
}
|
||||
}
|
||||
|
||||
// A CONFIGURED box's report object must be byte-identical to v0.198.0's — no `state` key at all.
|
||||
// This is what lets a pre-upgrade hub and every existing checker read the fleet unchanged.
|
||||
func TestOffsiteDeclare_ConfiguredBoxJSONIsUnchanged(t *testing.T) {
|
||||
m, sett := bareManager(t)
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
Schedule: "daily", EscrowState: "escrowed", LastStatus: "ok",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil {
|
||||
t.Fatal("a configured box must still report an off-site object")
|
||||
}
|
||||
if st.State != "" {
|
||||
t.Errorf("a configured box must declare NO state, got %q", st.State)
|
||||
}
|
||||
b, err := json.Marshal(st)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Contains(string(b), `"state"`) {
|
||||
t.Fatalf("a healthy report's JSON gained a `state` key — it must stay byte-compatible: %s", b)
|
||||
}
|
||||
if !strings.Contains(string(b), `"enabled":true`) {
|
||||
t.Fatalf("a configured box must report enabled:true: %s", b)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-193 Part 3 — WHAT IS IN THERE. After a successful unlock the customer is shown the contents of the
|
||||
// repository they just opened: which apps, from when, how big.
|
||||
//
|
||||
// READ-ONLY, AND THAT IS THE POINT. This restores nothing, puts nothing back, and compares nothing
|
||||
// against live data. Unlocking and restoring are separate (operator ruling, 2026-08-05): restore is
|
||||
// already per-app and already lives in the backups area, and a screen that unlocks and then offers to
|
||||
// overwrite is two decisions wearing one button.
|
||||
//
|
||||
// WHY A LISTING AT ALL, rather than a success message: "unlocked" with nothing shown is
|
||||
// indistinguishable from having unlocked an EMPTY store, and the customer has no way to tell whether
|
||||
// what came back is the right thing. Seeing their own app names and dates is how they know.
|
||||
|
||||
// errNoOffsiteTarget is returned when the repository cannot even be addressed — no off-site target is
|
||||
// configured on this box yet. Distinguished from a read failure because the remedy differs: this one
|
||||
// resolves by itself once the tier is re-applied.
|
||||
var errNoOffsiteTarget = errors.New("no off-site target is configured on this box yet")
|
||||
|
||||
// ErrNoOffsiteTarget reports whether err is the not-yet-configured case, so a caller can say the right
|
||||
// thing rather than showing a generic failure.
|
||||
func ErrNoOffsiteTarget(err error) bool { return errors.Is(err, errNoOffsiteTarget) }
|
||||
|
||||
// ErrNoOffsiteTargetSentinel exposes the sentinel itself so other packages — and their tests — can
|
||||
// construct the not-yet-configured case. Added for R-237, whose restore list must distinguish
|
||||
// "no target yet" (resolves by itself) from "could not read" (does not), and must be able to pin
|
||||
// both in a table test.
|
||||
func ErrNoOffsiteTargetSentinel() error { return errNoOffsiteTarget }
|
||||
|
||||
// OffsiteInventoryApp is one app's presence in the opened repository. Non-secret throughout.
|
||||
type OffsiteInventoryApp struct {
|
||||
App string // the restic tag == the stack name
|
||||
LatestAt time.Time // the newest snapshot's time for this app
|
||||
SizeBytes int64 // restore size of that newest snapshot (0 = could not be determined)
|
||||
}
|
||||
|
||||
// OffsiteInventory is the whole answer, including the EMPTY case stated explicitly.
|
||||
type OffsiteInventory struct {
|
||||
Apps []OffsiteInventoryApp
|
||||
// Empty is true when the repository opened cleanly and holds no snapshots. It is a real and
|
||||
// confusing outcome — a bare list there reads as a broken page — so it is named rather than
|
||||
// inferred from len(Apps)==0, which is also what a failed read looks like.
|
||||
Empty bool
|
||||
}
|
||||
|
||||
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
|
||||
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
|
||||
//
|
||||
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
|
||||
// app is in there matters more than knowing how big it is, and dropping it would under-report the
|
||||
// customer's own data.
|
||||
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
|
||||
var inv OffsiteInventory
|
||||
// A box can hold a recovered key and still have no off-site COORDINATES — the pristine rebuilt
|
||||
// shape, before its target is re-applied. Reading the repository is impossible then, and saying so
|
||||
// is the honest answer; without this guard offboxBaseArgs nil-derefs on the missing target.
|
||||
if !m.OffboxConfigured() {
|
||||
return inv, errNoOffsiteTarget
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||||
defer cancel()
|
||||
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
||||
if err != nil {
|
||||
return inv, err
|
||||
}
|
||||
var snaps []struct {
|
||||
ShortID string `json:"short_id"`
|
||||
ID string `json:"id"`
|
||||
Time time.Time `json:"time"`
|
||||
Tags []string `json:"tags"`
|
||||
}
|
||||
if uerr := json.Unmarshal(out, &snaps); uerr != nil {
|
||||
return inv, uerr
|
||||
}
|
||||
if len(snaps) == 0 {
|
||||
inv.Empty = true
|
||||
return inv, nil
|
||||
}
|
||||
// Newest snapshot per tag. A snapshot may carry several tags; each names an app it belongs to.
|
||||
newest := map[string]struct {
|
||||
id string
|
||||
at time.Time
|
||||
}{}
|
||||
for _, s := range snaps {
|
||||
id := s.ShortID
|
||||
if id == "" {
|
||||
id = s.ID
|
||||
}
|
||||
for _, tag := range s.Tags {
|
||||
if tag == "" {
|
||||
continue
|
||||
}
|
||||
if cur, ok := newest[tag]; !ok || s.Time.After(cur.at) {
|
||||
newest[tag] = struct {
|
||||
id string
|
||||
at time.Time
|
||||
}{id: id, at: s.Time}
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(newest) == 0 {
|
||||
// Snapshots exist but carry no tags — not "empty", and saying so would be a lie. Report an
|
||||
// empty app list without the Empty flag; the page renders the honest in-between wording.
|
||||
return inv, nil
|
||||
}
|
||||
for tag, n := range newest {
|
||||
app := OffsiteInventoryApp{App: tag, LatestAt: n.at}
|
||||
if size, serr := m.offboxSnapshotSize(ctx, n.id); serr == nil {
|
||||
app.SizeBytes = size
|
||||
} else {
|
||||
m.logger.Printf("[WARN] [offbox] inventory: size of %s's newest snapshot unknown: %v (listing it anyway)", tag, serr)
|
||||
}
|
||||
inv.Apps = append(inv.Apps, app)
|
||||
}
|
||||
sort.Slice(inv.Apps, func(i, j int) bool { return inv.Apps[i].App < inv.Apps[j].App })
|
||||
return inv, nil
|
||||
}
|
||||
|
||||
// HumanizeBytes exposes the shared byte formatter to the web layer so the recovery page renders sizes
|
||||
// the same way every other surface does.
|
||||
func HumanizeBytes(n int64) string { return humanizeBytes(n) }
|
||||
@@ -0,0 +1,132 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
func newTestSettings(t *testing.T) *settings.Settings {
|
||||
t.Helper()
|
||||
sett, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatalf("settings.Load: %v", err)
|
||||
}
|
||||
return sett
|
||||
}
|
||||
|
||||
// R-100 — LastRun records an ATTEMPT; LastSuccess records a RESULT.
|
||||
//
|
||||
// The defect these pin: `LastRun` is written unconditionally at the end of every offsite run, failures
|
||||
// included, so the hub's staleness verdict ("how long since LastRun?") was really asking "how long
|
||||
// since we last TRIED?" — and a tier failing on every single run read as perfectly fresh forever.
|
||||
//
|
||||
// These are the CONTROLLER half (does the anchor move only on success, and does it survive the writes
|
||||
// that rebuild the target?). The hub half — does the verdict count from it — lives in the hub's
|
||||
// offsite tests.
|
||||
|
||||
// The invariant named by the comment at the write site, per the standing rule that an asserted
|
||||
// invariant needs a test pinning it. This calls the PRODUCTION rule — an earlier version of this test
|
||||
// re-implemented it in a local closure and was hollow: mutating offbox.go left it green.
|
||||
//
|
||||
// RED-PROOF: make offboxAnchorAfterRun return `at` unconditionally (drop the runErr guard) → this
|
||||
// fails with "a FAILED run advanced LastSuccess — that is the R-100 defect in mirror image".
|
||||
func TestOffboxAnchorAfterRun_FailureNeitherAdvancesNorClears(t *testing.T) {
|
||||
const monday = "2026-07-20T02:15:00Z"
|
||||
boom := errors.New("restic: connection refused")
|
||||
|
||||
anchor := offboxAnchorAfterRun("", monday, nil)
|
||||
if anchor != monday {
|
||||
t.Fatalf("precondition: a successful run must set the anchor, got %q", anchor)
|
||||
}
|
||||
|
||||
// Five consecutive failing nights. The attempt clock moves; the anchor must not.
|
||||
for _, night := range []string{
|
||||
"2026-07-21T02:15:00Z", "2026-07-22T02:15:00Z", "2026-07-23T02:15:00Z",
|
||||
"2026-07-24T02:15:00Z", "2026-07-25T02:15:00Z",
|
||||
} {
|
||||
anchor = offboxAnchorAfterRun(anchor, night, boom)
|
||||
if anchor == night {
|
||||
t.Fatalf("a FAILED run advanced LastSuccess to %q — that is the R-100 defect in mirror image", anchor)
|
||||
}
|
||||
if anchor != monday {
|
||||
t.Fatalf("a FAILED run CLEARED or moved the anchor (got %q, want %q) — one bad night must not make an established tier read as never-succeeded", anchor, monday)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Recovery: a later success moves it forward, or a tier would stay permanently stale after one good
|
||||
// night.
|
||||
//
|
||||
// RED-PROOF: make offboxAnchorAfterRun return `prev` unconditionally → this fails with
|
||||
// "a successful run did not advance the anchor".
|
||||
func TestOffboxAnchorAfterRun_SuccessAdvances(t *testing.T) {
|
||||
got := offboxAnchorAfterRun("2026-07-20T02:15:00Z", "2026-07-26T02:15:00Z", nil)
|
||||
if got != "2026-07-26T02:15:00Z" {
|
||||
t.Errorf("a successful run did not advance the anchor: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A never-run tier stays empty on failure — it must not acquire a fabricated anchor, because "" is the
|
||||
// signal the hub's newborn-box path keys on.
|
||||
func TestOffboxAnchorAfterRun_NeverRanStaysEmptyOnFailure(t *testing.T) {
|
||||
if got := offboxAnchorAfterRun("", "2026-07-21T02:15:00Z", errors.New("boom")); got != "" {
|
||||
t.Errorf("a failed first run fabricated an anchor (%q) — the newborn-box path keys on empty", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The wire carries it. A field the hub cannot see is a field that does not exist — the "seam built but
|
||||
// never wired" class this project has hit four times.
|
||||
//
|
||||
// RED-PROOF: drop `LastSuccess: t.LastSuccess` from OffboxReportStatus() → this fails with
|
||||
// "OffboxReportStatus dropped LastSuccess — the hub would degrade forever on a controller that has it".
|
||||
func TestOffboxReportStatus_CarriesLastSuccess(t *testing.T) {
|
||||
m := &Manager{settings: newTestSettings(t)}
|
||||
if err := m.settings.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true,
|
||||
Host: "nas.example",
|
||||
User: "u1",
|
||||
RepoPath: "/vol/repo",
|
||||
EscrowState: "escrowed",
|
||||
LastRun: "2026-07-26T02:15:00Z",
|
||||
LastStatus: "ok",
|
||||
LastSuccess: "2026-07-26T02:15:00Z",
|
||||
}); err != nil {
|
||||
t.Fatalf("seed: %v", err)
|
||||
}
|
||||
got := m.OffboxReportStatus()
|
||||
if got == nil {
|
||||
t.Fatal("OffboxReportStatus returned nil for an enabled target")
|
||||
}
|
||||
if got.LastSuccess != "2026-07-26T02:15:00Z" {
|
||||
t.Errorf("OffboxReportStatus dropped LastSuccess — the hub would degrade forever on a controller that has it (got %q)", got.LastSuccess)
|
||||
}
|
||||
}
|
||||
|
||||
// A re-apply from the hub is not a new tier. Dropping the anchor here would reset an established tier
|
||||
// to "never succeeded" every time the hub re-pushes its descriptor.
|
||||
//
|
||||
// RED-PROOF: remove `tgt.LastSuccess = cur.LastSuccess` from ApplyOffsiteTarget's carry-over block →
|
||||
// this fails with "a hub re-apply erased the staleness anchor".
|
||||
func TestApplyOffsiteTarget_PreservesLastSuccess(t *testing.T) {
|
||||
m := &Manager{settings: newTestSettings(t)}
|
||||
if err := m.settings.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.example", User: "u1", RepoPath: "/vol/repo",
|
||||
EscrowState: "escrowed", LastSuccess: "2026-07-26T02:15:00Z", LastRun: "2026-07-27T02:15:00Z",
|
||||
}); err != nil {
|
||||
t.Fatalf("seed: %v", err)
|
||||
}
|
||||
cur := m.settings.GetOffboxTarget()
|
||||
// Mirror ApplyOffsiteTarget's carry-over onto a freshly-built target.
|
||||
tgt := &settings.OffboxTarget{Enabled: true, Host: "nas.example", User: "u1", RepoPath: "/vol/repo", Schedule: "daily"}
|
||||
tgt.EscrowState = cur.EscrowState
|
||||
tgt.LastRun, tgt.LastStatus, tgt.LastError = cur.LastRun, cur.LastStatus, cur.LastError
|
||||
tgt.LastSuccess = cur.LastSuccess
|
||||
if tgt.LastSuccess != "2026-07-26T02:15:00Z" {
|
||||
t.Errorf("a hub re-apply erased the staleness anchor (got %q)", tgt.LastSuccess)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-241 — THE MINT GUARD. This file is the session's headline test.
|
||||
//
|
||||
// The defect, measured on the final walk (SPIKE-r241-recovery-offer-2026-08-07): a rebuilt box's
|
||||
// credential self-heal reached WriteOffboxSecrets at 03:18:06Z and minted a fresh repository password
|
||||
// over a hub package sealing a DIFFERENT key. The recovery screen then correctly reported that there
|
||||
// was nothing recoverable under the key the box held. The screen was honest; the minting was not.
|
||||
//
|
||||
// Scenario A asserts the key is NOT written. Scenario B asserts the guard is narrow enough that a
|
||||
// first-time box still starts — the guard's own failure mode, and the one an over-broad fix produces.
|
||||
|
||||
// mintGuardManager builds a Manager with NO offbox secrets written, so the mint branch is live.
|
||||
// hubHoldsPackage sets the ACK-cached fact the guard consults.
|
||||
func mintGuardManager(t *testing.T, hubHoldsPackage bool) (*Manager, *settings.Settings, string) {
|
||||
t.Helper()
|
||||
logger := log.New(os.Stderr, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
sett, err := settings.Load(filepath.Join(dataDir, "settings.json"), logger)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
cfg.Paths.SystemDataPath = filepath.Join(dataDir, "sys")
|
||||
m := NewManager(cfg, sett, logger)
|
||||
if err := sett.SetHubEscrowIdentityPresent(hubHoldsPackage); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return m, sett, filepath.Join(dataDir, "offbox", "repo_password")
|
||||
}
|
||||
|
||||
// ── SCENARIO A — the box does not mint over a sealed package ────────────────────────────────────
|
||||
//
|
||||
// RED-PROOF: delete the `if m.sealedPackageHeld()` block in WriteOffboxSecrets. The password file
|
||||
// then exists and this test fails on the first assertion — which is exactly the 03:18:06Z event.
|
||||
func TestR241_ScenarioA_NoMintWhenHubHoldsSealedPackage(t *testing.T) {
|
||||
m, _, pwPath := mintGuardManager(t, true)
|
||||
|
||||
err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey")
|
||||
|
||||
if !IsOffboxSealedPackageHeld(err) {
|
||||
t.Fatalf("want the sealed-package refusal sentinel, got %v", err)
|
||||
}
|
||||
// THE ASSERTION THAT IS THE WHOLE SESSION: no key on disk.
|
||||
if _, serr := os.Stat(pwPath); !os.IsNotExist(serr) {
|
||||
t.Fatalf("R-241 REGRESSION: a repository password was minted over the hub's sealed package (stat err=%v)", serr)
|
||||
}
|
||||
// The transport IS still written — the refusal is a holding state, not a failure. Without this the
|
||||
// recovery screen could not bring the tier up when the key arrives (R-219).
|
||||
for _, f := range []string{"ssh_key", "known_hosts"} {
|
||||
if _, serr := os.Stat(filepath.Join(filepath.Dir(pwPath), f)); serr != nil {
|
||||
t.Errorf("transport file %s should still be written on the refusal path: %v", f, serr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario A at the APPLY level — the path the self-heal actually takes. ApplyOffsiteTarget must
|
||||
// swallow the sentinel, record the target, and NOT stage an escrow.
|
||||
func TestR241_ScenarioA_ApplyOffsiteTargetHoldsInsteadOfMinting(t *testing.T) {
|
||||
m, sett, pwPath := mintGuardManager(t, true)
|
||||
|
||||
staged := 0
|
||||
stage := func(ctx context.Context, pw string) error { staged++; return nil }
|
||||
|
||||
tgt := &settings.OffboxTarget{Enabled: true, Host: "box.example", Port: 23, User: "u1", RepoPath: "/home/felhom-repo"}
|
||||
if err := m.ApplyOffsiteTarget(context.Background(), tgt, "KEYMATERIAL", "box.example ssh-ed25519 HOSTKEY", stage); err != nil {
|
||||
t.Fatalf("apply should SUCCEED into the holding state, not fail: %v", err)
|
||||
}
|
||||
if _, serr := os.Stat(pwPath); !os.IsNotExist(serr) {
|
||||
t.Fatalf("R-241 REGRESSION: apply minted a repository password over the sealed package")
|
||||
}
|
||||
if staged != 0 {
|
||||
t.Errorf("nothing may be staged for escrow — there is no key to escrow; staged=%d", staged)
|
||||
}
|
||||
// The target is recorded, so the box stops declaring needs_credential and the hub stops re-staging.
|
||||
if got := sett.GetOffboxTarget(); got == nil {
|
||||
t.Fatal("the transport target must be recorded, or the hub re-stages a consumed credential forever")
|
||||
}
|
||||
// Runs stay gated: no password file ⇒ not configured.
|
||||
if m.OffboxConfigured() {
|
||||
t.Error("OffboxConfigured must be false while the key is awaited — runs must not proceed")
|
||||
}
|
||||
// And the box says so, in the state the hub reads.
|
||||
if !m.OffboxAwaitingRecoveryKey() {
|
||||
t.Error("OffboxAwaitingRecoveryKey should be true in the holding state")
|
||||
}
|
||||
st := m.OffboxReportStatus()
|
||||
if st == nil || st.State != OffsiteStateAwaitingRecoveryKey {
|
||||
t.Fatalf("want declared state %q, got %+v", OffsiteStateAwaitingRecoveryKey, st)
|
||||
}
|
||||
if st.Enabled {
|
||||
t.Error("the declared holding object must carry Enabled=false so existing hub readers stay inert")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO B — a box the hub holds nothing for still mints, exactly as today ───────────────────
|
||||
//
|
||||
// RED-PROOF: widen the guard to `if true` (or drop the GetHubEscrowIdentityPresent() conjunct in
|
||||
// sealedPackageHeld). A first-time box then cannot start, and this test fails — the failure mode an
|
||||
// over-broad fix produces, which is why the guard is written as a conjunction.
|
||||
func TestR241_ScenarioB_FirstTimeBoxStillMints(t *testing.T) {
|
||||
m, _, pwPath := mintGuardManager(t, false) // the hub holds nothing for us
|
||||
|
||||
if err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey"); err != nil {
|
||||
t.Fatalf("a first-time box must mint exactly as before, got %v", err)
|
||||
}
|
||||
pw, rerr := os.ReadFile(pwPath)
|
||||
if rerr != nil {
|
||||
t.Fatalf("a first-time box must get a repository password: %v", rerr)
|
||||
}
|
||||
if !offboxRepoPwPattern.Match(pw) {
|
||||
t.Errorf("minted password is not the expected 64-hex shape")
|
||||
}
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
t.Error("a box with no sealed package is not awaiting anything")
|
||||
}
|
||||
}
|
||||
|
||||
// The guard must not fire once a key EXISTS — a healthy box re-applying its target (a quota bump,
|
||||
// a hub re-push) must be untouched, package or no package. This is the idempotency half.
|
||||
func TestR241_ExistingKeyIsNeverDisturbed(t *testing.T) {
|
||||
m, _, pwPath := mintGuardManager(t, false)
|
||||
if err := m.WriteOffboxSecrets("K", "kh"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before, err := os.ReadFile(pwPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Now the hub starts holding a package (the ceremony ran) and the target is re-applied.
|
||||
if err := m.settings.SetHubEscrowIdentityPresent(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.WriteOffboxSecrets("K2", "kh2"); err != nil {
|
||||
t.Fatalf("a re-apply on a box that already has a key must not be refused: %v", err)
|
||||
}
|
||||
after, err := os.ReadFile(pwPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if string(before) != string(after) {
|
||||
t.Error("the existing repository password must never be rotated by an apply")
|
||||
}
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
t.Error("a box holding its key is not awaiting one")
|
||||
}
|
||||
}
|
||||
|
||||
// Fail-safe: an unreadable settings store must not block a tier. A transient read failure turning
|
||||
// into a permanently-held tier is a worse defect than the one being fixed.
|
||||
func TestR241_NilSettingsDoesNotBlockTheMint(t *testing.T) {
|
||||
logger := log.New(os.Stderr, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
m := NewManager(cfg, nil, logger)
|
||||
if m.sealedPackageHeld() {
|
||||
t.Fatal("a nil settings store must read as 'no package held' — fail toward letting the box work")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario E's carve-out, pinned for the HOLDING state too. A customer who switched off-site off is
|
||||
// not awaiting a recovery key, and must not declare one. The first draft of
|
||||
// OffboxAwaitingRecoveryKey omitted `t.Enabled` and TestOffsiteDeclare_DisabledTargetIsNotStranded
|
||||
// caught it; this test pins the same invariant from the new predicate's own side, so a future edit
|
||||
// to THIS function fails here rather than in a neighbouring file.
|
||||
func TestR241_DisabledTargetIsNotAwaitingAnything(t *testing.T) {
|
||||
m, sett, _ := mintGuardManager(t, true) // the hub holds a package, and there is no key
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: false, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffboxAwaitingRecoveryKey() {
|
||||
t.Fatal("a deliberately DISABLED target must never declare the holding state (Scenario E)")
|
||||
}
|
||||
if st := m.OffboxReportStatus(); st != nil {
|
||||
t.Fatalf("a disabled target must stay silent in the report, got %+v", st)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-241 shape (c) — the recovery offer is driven by the comparison the box already makes.
|
||||
//
|
||||
// Scenarios C and D from the task, plus §7.2's two staleness cases. The point of shape (c) is that
|
||||
// it asks the real question — *does the hub hold a package for a key other than the one I am
|
||||
// using?* — rather than the two proxies that have each now been wrong in opposite directions.
|
||||
|
||||
// offerFixture builds a manager holding a repository password, with the hub's cached facts settable.
|
||||
// Returns the local key's hash so a test can make the hub's hash match or differ deliberately.
|
||||
func offerFixture(t *testing.T, hubHoldsPackage bool) (*Manager, *settings.Settings, string) {
|
||||
t.Helper()
|
||||
m, sett, pwPath := mintGuardManager(t, false) // mint freely first
|
||||
if err := sett.SetOffboxTarget(&settings.OffboxTarget{
|
||||
Enabled: true, Host: "nas.local", Port: 22, User: "felhom", RepoPath: "/srv/repo", Schedule: "daily",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.WriteOffboxSecrets("KEYMATERIAL", "nas.local ssh-ed25519 HOSTKEY"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := os.Stat(pwPath); err != nil {
|
||||
t.Fatalf("fixture should hold a repository password: %v", err)
|
||||
}
|
||||
local, ok := m.OffboxRepoPasswordHash()
|
||||
if !ok {
|
||||
t.Fatal("fixture should be able to hash its own key")
|
||||
}
|
||||
if err := sett.SetHubEscrowIdentityPresent(hubHoldsPackage); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return m, sett, local
|
||||
}
|
||||
|
||||
const otherKeyHash = "9b4a9a9dcec7898e7544f35b18470aac77c3d9064e5d3a302897617fa62edd65"
|
||||
|
||||
// ── SCENARIO C — a differing key offers recovery, whatever the reason for the difference ────────
|
||||
//
|
||||
// This is the venue's exact state on 2026-08-07: a key present, no orphan recorded, escrow stuck
|
||||
// pending — and before shape (c), silence.
|
||||
func TestR241_ScenarioC_DifferingKeyOffersRecovery(t *testing.T) {
|
||||
m, sett, local := offerFixture(t, true)
|
||||
if local == otherKeyHash {
|
||||
t.Fatal("fixture precondition: the local key must differ from the hub's")
|
||||
}
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2026-08-07T03:28:03Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Neither proxy fires: a key EXISTS (so not shape (a)) and nothing is orphaned (so not shape (b)).
|
||||
if _, ok := m.OffboxRepoPasswordHash(); !ok {
|
||||
t.Fatal("precondition: shape (a) must be false")
|
||||
}
|
||||
if m.OffboxOrphaned() {
|
||||
t.Fatal("precondition: shape (b) must be false")
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("R-241: the hub holds a package for a DIFFERENT key and the screen was not offered — this is the defect")
|
||||
}
|
||||
}
|
||||
|
||||
// ── SCENARIO D — a healthy box is never offered recovery ────────────────────────────────────────
|
||||
//
|
||||
// RED-PROOF: drop the `hubHash != localHash` conjunct in shape (c) (make it `hubHash != ""`). A
|
||||
// healthy box is then offered recovery forever, and this test fails — which is how a screen stops
|
||||
// being read.
|
||||
func TestR241_ScenarioD_MatchingKeyOffersNothing(t *testing.T) {
|
||||
m, sett, local := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256(local, "2026-08-07T09:00:00Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a box whose key the hub's package covers must never be offered recovery")
|
||||
}
|
||||
}
|
||||
|
||||
// A box the hub holds nothing for is never offered, even if a stale hash lingers in settings. Fact 1
|
||||
// stays required — the spike's comment block calls dropping it "the plausible wrong fix".
|
||||
func TestR241_ShapeC_NeverHadOffsiteIsStillSilent(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, false) // the hub holds NOTHING
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2026-08-07T09:00:00Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a box that never had off-site backups must never be greeted by a recovery screen")
|
||||
}
|
||||
}
|
||||
|
||||
// ── §7.2 — the staleness decision, both halves ──────────────────────────────────────────────────
|
||||
|
||||
// A KNOWN DIFFERENCE OFFERS, however old the reading. Age is deliberately not gated on: gating would
|
||||
// make a box offline from the hub silently stop offering, which is the failure this session exists
|
||||
// to remove.
|
||||
func TestR241_StaleComparison_KnownDifferenceStillOffers(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2020-01-01T00:00:00Z"); err != nil { // ancient
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a known difference must offer regardless of how old the reading is (§7.2)")
|
||||
}
|
||||
}
|
||||
|
||||
// AN ABSENT HASH FALLS BACK TO (a)/(b) — it does not offer. "" is the hub positively saying its
|
||||
// package seals no repository password (legacy hash-less escrow); there is nothing to compare, and
|
||||
// offering would put a permanent screen in front of every legacy box.
|
||||
func TestR241_StaleComparison_AbsentHashFallsBackAndDoesNotOffer(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := sett.SetHubEscrowKeySHA256("", ""); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("a hash never learned must fall back to (a)/(b), not offer (§7.2)")
|
||||
}
|
||||
// ...and the fallback still works: mark the repo orphaned and shape (b) fires as before.
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoState = "orphaned" }); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("shape (b) must still work when the hub's hash was never learned")
|
||||
}
|
||||
}
|
||||
|
||||
// Shape (a) is untouched: a box with no key at all is still offered, which is the pristine rebuild.
|
||||
func TestR241_ShapeAStillWorks(t *testing.T) {
|
||||
m, sett, _ := offerFixture(t, true)
|
||||
if err := os.Remove(filepath.Join(m.cfg.Paths.DataDir, "offbox", "repo_password")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetHubEscrowKeySHA256(otherKeyHash, "2026-08-07T09:00:00Z"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.OffsiteRecoveryOffer() {
|
||||
t.Fatal("shape (a) — no repository password at all — must still offer")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,345 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Offsite backup progress (v0.147.0, feedback slice 4c).
|
||||
//
|
||||
// THE PROBLEM: „Távoli mentés most" started a background restic run and redirected with „A távoli
|
||||
// mentés elindult". After that the page polled a status field whose only values were running / ok /
|
||||
// error. For a first offsite push of tens of gigabytes over SFTP that is 20+ minutes of a spinner
|
||||
// with no total, no percentage and no indication of WHICH app is being pushed — indistinguishable
|
||||
// from a hang.
|
||||
//
|
||||
// restic already reports all of it: `backup --json` writes newline-delimited status objects to
|
||||
// stdout. We only had to stop throwing them away — the existing runner seam uses CombinedOutput(),
|
||||
// which buffers everything until exit.
|
||||
//
|
||||
// SCOPE: the MANUAL trigger only. The nightly scheduled run stays silent (nobody is watching a
|
||||
// progress bar at 03:00, and a sink left installed would keep publishing stale percentages into a
|
||||
// page that never asked). The sink is installed for the duration of a manual run and cleared after.
|
||||
|
||||
// OffboxProgress is a snapshot of an in-flight manual offsite backup.
|
||||
type OffboxProgress struct {
|
||||
Active bool `json:"active"`
|
||||
CurrentApp string `json:"current_app"`
|
||||
Percent float64 `json:"percent"` // 0..100, restic's byte-based percent_done
|
||||
BytesDone int64 `json:"bytes_done"`
|
||||
TotalBytes int64 `json:"total_bytes"`
|
||||
DoneHuman string `json:"done_human"`
|
||||
TotalHuman string `json:"total_human"`
|
||||
// FilesDone/TotalFiles matter more than they look. On an INCREMENTAL run where nothing changed,
|
||||
// restic transfers no new bytes: bytes_done stays 0 (it is `omitempty`, so it is not even in the
|
||||
// JSON) and percent_done stays 0 for the whole run, while restic still walks every file. Measured
|
||||
// on the demo box: a 430MB immich push sat at 0% for 40+ seconds and then completed. A byte-only
|
||||
// bar is therefore indistinguishable from a hang precisely in the COMMON case. File counts move
|
||||
// in that case, so the page falls back to them.
|
||||
FilesDone int64 `json:"files_done"`
|
||||
TotalFiles int64 `json:"total_files"`
|
||||
// CurrentFile/ElapsedSec are the last resort, and on real data the most important fields here.
|
||||
// restic 0.14 only counts a file into files_done/bytes_done when it COMPLETES, so a single
|
||||
// dominant file freezes both counters: measured on the demo box, immich sat at files_done 1 of 46
|
||||
// and bytes_done 0 for 42 seconds while restic worked on one ~430MB volume tar. No percentage can
|
||||
// move during that window. What CAN be shown truthfully is which file is being processed and how
|
||||
// long it has been going — "working on X, 42s" is a completely different message from "0%".
|
||||
CurrentFile string `json:"current_file"`
|
||||
ElapsedSec int64 `json:"elapsed_sec"`
|
||||
// Phase names the part of the run in progress. A run is NOT just the per-app loop: after the last
|
||||
// app come the shares leg and `forget --prune`, which on the demo box took 40 of a 57-second run.
|
||||
// Without this the card froze on the last app's finished counters for that whole tail — the same
|
||||
// silence the slice exists to remove, just relocated. "" = per-app backup.
|
||||
Phase string `json:"phase"`
|
||||
}
|
||||
|
||||
// resticStatusLine is the subset of restic's `--json` status object we consume. restic emits several
|
||||
// message_types (status, summary, error, verbose_status); anything that is not "status" is ignored
|
||||
// here rather than treated as garbage, because restic adds new types between versions and an unknown
|
||||
// type must never break the run. Schema captured from
|
||||
// restic 0.14.0 (the version in the controller image) via a live `backup --dry-run --json`:
|
||||
//
|
||||
// {"message_type":"status","percent_done":0,"total_files":1,"total_bytes":112}
|
||||
// {"message_type":"status","percent_done":0.558,"total_files":173,"files_done":87,
|
||||
// "total_bytes":166878,"bytes_done":93161,"current_files":[...]}
|
||||
//
|
||||
// Note every numeric field except percent_done is `omitempty` on restic's side: a zero simply is not
|
||||
// in the JSON. That is why an incremental run reports no bytes_done at all rather than an explicit 0.
|
||||
type resticStatusLine struct {
|
||||
MessageType string `json:"message_type"`
|
||||
PercentDone float64 `json:"percent_done"` // 0..1
|
||||
TotalBytes int64 `json:"total_bytes"`
|
||||
BytesDone int64 `json:"bytes_done"`
|
||||
TotalFiles int64 `json:"total_files"`
|
||||
FilesDone int64 `json:"files_done"`
|
||||
CurrentFiles []string `json:"current_files"`
|
||||
SecondsElapsed int64 `json:"seconds_elapsed"`
|
||||
}
|
||||
|
||||
// resticProgress is one parsed status line.
|
||||
type resticProgress struct {
|
||||
Percent float64 // 0..100
|
||||
BytesDone int64
|
||||
TotalBytes int64
|
||||
FilesDone int64
|
||||
TotalFiles int64
|
||||
CurrentFile string
|
||||
ElapsedSec int64
|
||||
}
|
||||
|
||||
// parseResticStatus parses ONE line of restic --json output.
|
||||
//
|
||||
// Kept as a pure function precisely so it can be tested without restic, a network, or a repo — the
|
||||
// parser is the part that silently rots when restic changes its output, and a progress bar that
|
||||
// quietly stops moving is worse than no progress bar at all.
|
||||
func parseResticStatus(line string) (resticProgress, bool) {
|
||||
var s resticStatusLine
|
||||
if err := json.Unmarshal([]byte(line), &s); err != nil {
|
||||
return resticProgress{}, false
|
||||
}
|
||||
if s.MessageType != "status" {
|
||||
return resticProgress{}, false
|
||||
}
|
||||
pct := s.PercentDone * 100
|
||||
// restic revises its total as the scan proceeds, so percent_done legitimately moves backwards
|
||||
// mid-run and has been seen slightly above 1 near completion. Clamp — a bar wider than its track
|
||||
// is a visible bug.
|
||||
if pct < 0 {
|
||||
pct = 0
|
||||
}
|
||||
if pct > 100 {
|
||||
pct = 100
|
||||
}
|
||||
cur := ""
|
||||
if len(s.CurrentFiles) > 0 {
|
||||
cur = s.CurrentFiles[0]
|
||||
}
|
||||
return resticProgress{
|
||||
Percent: pct, BytesDone: s.BytesDone, TotalBytes: s.TotalBytes,
|
||||
FilesDone: s.FilesDone, TotalFiles: s.TotalFiles,
|
||||
CurrentFile: cur, ElapsedSec: s.SecondsElapsed,
|
||||
}, true
|
||||
}
|
||||
|
||||
// offboxProgressState is the published snapshot, guarded independently of the Manager mutex so a
|
||||
// poll never blocks behind the running backup.
|
||||
type offboxProgressState struct {
|
||||
mu sync.Mutex
|
||||
cur OffboxProgress
|
||||
live bool
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) begin() {
|
||||
p.mu.Lock()
|
||||
p.cur = OffboxProgress{Active: true}
|
||||
p.live = true
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) end() {
|
||||
p.mu.Lock()
|
||||
p.cur = OffboxProgress{}
|
||||
p.live = false
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) setApp(app string) {
|
||||
p.mu.Lock()
|
||||
if p.live {
|
||||
// A new app resets the byte counters: restic's percentages are per-invocation, and carrying
|
||||
// the previous app's 100% into the next app's start would show a bar that jumps backwards.
|
||||
p.cur.CurrentApp = app
|
||||
p.cur.Phase = ""
|
||||
p.cur.Percent, p.cur.BytesDone, p.cur.TotalBytes = 0, 0, 0
|
||||
p.cur.FilesDone, p.cur.TotalFiles = 0, 0
|
||||
p.cur.CurrentFile, p.cur.ElapsedSec = "", 0
|
||||
p.cur.DoneHuman, p.cur.TotalHuman = "", ""
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// setPhase marks a non-per-app stage of the run and clears the app-scoped counters, so the card
|
||||
// stops showing the last app's finished numbers against work that is no longer about that app.
|
||||
func (p *offboxProgressState) setPhase(phase string) {
|
||||
p.mu.Lock()
|
||||
if p.live {
|
||||
p.cur.Phase = phase
|
||||
p.cur.CurrentApp = ""
|
||||
p.cur.Percent, p.cur.BytesDone, p.cur.TotalBytes = 0, 0, 0
|
||||
p.cur.FilesDone, p.cur.TotalFiles = 0, 0
|
||||
p.cur.CurrentFile, p.cur.ElapsedSec = "", 0
|
||||
p.cur.DoneHuman, p.cur.TotalHuman = "", ""
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// OffboxPhaseDump is the PRE-app-loop stage (R-44, v0.148.0); OffboxPhaseShares /
|
||||
// OffboxPhaseRetention are the post-app-loop stages.
|
||||
const (
|
||||
OffboxPhaseDump = "dump"
|
||||
OffboxPhaseShares = "shares"
|
||||
OffboxPhaseRetention = "retention"
|
||||
)
|
||||
|
||||
func (p *offboxProgressState) update(r resticProgress) {
|
||||
p.mu.Lock()
|
||||
if p.live {
|
||||
p.cur.Percent, p.cur.BytesDone, p.cur.TotalBytes = r.Percent, r.BytesDone, r.TotalBytes
|
||||
p.cur.FilesDone, p.cur.TotalFiles = r.FilesDone, r.TotalFiles
|
||||
p.cur.ElapsedSec = r.ElapsedSec
|
||||
// Keep the last KNOWN current file: restic omits current_files on some status ticks, and
|
||||
// blanking the label every other second is its own kind of flicker.
|
||||
if r.CurrentFile != "" {
|
||||
p.cur.CurrentFile = r.CurrentFile
|
||||
}
|
||||
p.cur.DoneHuman, p.cur.TotalHuman = humanizeBytes(r.BytesDone), humanizeBytes(r.TotalBytes)
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
func (p *offboxProgressState) snapshot() OffboxProgress {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
return p.cur
|
||||
}
|
||||
|
||||
// OffboxProgressSnapshot is the poll surface for the „Távoli mentés" page.
|
||||
func (m *Manager) OffboxProgressSnapshot() OffboxProgress { return m.offboxProgress.snapshot() }
|
||||
|
||||
// offboxStreamRunner is the streaming restic-exec seam: like offboxRunner, but calls onLine for each
|
||||
// stdout line AS IT ARRIVES instead of only returning the buffered output at exit. Tests inject a
|
||||
// fake that emits canned `--json` status lines, so the whole progress path is exercised without
|
||||
// restic, a network or a repo.
|
||||
type offboxStreamRunner func(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error)
|
||||
|
||||
// SetOffboxStreamRunner installs the streaming seam (nil → the real streaming exec).
|
||||
func (m *Manager) SetOffboxStreamRunner(r offboxStreamRunner) { m.offboxStreamRunner = r }
|
||||
|
||||
func (m *Manager) streamRunner() offboxStreamRunner {
|
||||
if m.offboxStreamRunner != nil {
|
||||
return m.offboxStreamRunner
|
||||
}
|
||||
return defaultOffboxStreamRunner
|
||||
}
|
||||
|
||||
// defaultOffboxStreamRunner runs restic with stdout scanned line-by-line. stderr is captured whole
|
||||
// (restic's --json progress goes to stdout; errors go to stderr) and appended to the returned output
|
||||
// so callers keep the same error-diagnosis material CombinedOutput gave them.
|
||||
func defaultOffboxStreamRunner(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error) {
|
||||
cmd := exec.CommandContext(ctx, "restic", args...)
|
||||
cmd.Env = append(os.Environ(), env...)
|
||||
|
||||
stdout, err := cmd.StdoutPipe()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var stderr syncBuf
|
||||
cmd.Stderr = &stderr
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
var tail lineTail
|
||||
scanner := bufio.NewScanner(stdout)
|
||||
// restic status lines are small, but a --json summary listing many paths can exceed the 64KB
|
||||
// default; a scanner that dies mid-run would silently freeze the progress bar.
|
||||
scanner.Buffer(make([]byte, 0, 64*1024), 4*1024*1024)
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
tail.add(line)
|
||||
if onLine != nil {
|
||||
onLine(line)
|
||||
}
|
||||
}
|
||||
_, _ = io.Copy(io.Discard, stdout)
|
||||
|
||||
werr := cmd.Wait()
|
||||
// Keep only the tail of stdout: the full --json stream of a large backup is megabytes of status
|
||||
// spam, and every caller uses this output for error diagnosis (and lock-pattern matching) only.
|
||||
out := append(tail.bytes(), stderr.bytes()...)
|
||||
return out, werr
|
||||
}
|
||||
|
||||
// lineTail keeps the last N lines seen, so error diagnosis has context without buffering the whole
|
||||
// --json stream.
|
||||
type lineTail struct {
|
||||
lines []string
|
||||
}
|
||||
|
||||
func (t *lineTail) add(s string) {
|
||||
const keep = 40
|
||||
t.lines = append(t.lines, s)
|
||||
if len(t.lines) > keep {
|
||||
t.lines = t.lines[len(t.lines)-keep:]
|
||||
}
|
||||
}
|
||||
|
||||
func (t *lineTail) bytes() []byte {
|
||||
var b []byte
|
||||
for _, l := range t.lines {
|
||||
b = append(b, l...)
|
||||
b = append(b, '\n')
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
type syncBuf struct {
|
||||
mu sync.Mutex
|
||||
b []byte
|
||||
}
|
||||
|
||||
func (s *syncBuf) Write(p []byte) (int, error) {
|
||||
s.mu.Lock()
|
||||
s.b = append(s.b, p...)
|
||||
s.mu.Unlock()
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
func (s *syncBuf) bytes() []byte {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return append([]byte{}, s.b...)
|
||||
}
|
||||
|
||||
// resticBackupStep is resticStep's streaming twin, used ONLY by the app-backup leg when a manual run
|
||||
// has a progress sink installed. It keeps resticStep's crash-lock self-heal semantics by delegating
|
||||
// the retry path to resticStep (a retry after an unlock is rare and does not need progress).
|
||||
func (m *Manager) resticBackupStep(ctx context.Context, env, base []string, label, app string, args ...string) ([]byte, error) {
|
||||
if !m.offboxProgress.snapshot().Active {
|
||||
return m.resticStep(ctx, env, base, label, args...) // nightly / no watcher: unchanged path
|
||||
}
|
||||
m.offboxProgress.setApp(app)
|
||||
full := append(append([]string{}, base...), args...)
|
||||
// --json turns on the machine-readable progress stream. It is added ONLY here, so the nightly
|
||||
// run's output format (and everything that greps it) is untouched.
|
||||
full = append(full, "--json")
|
||||
out, err := m.streamRunner()(ctx, env, func(line string) {
|
||||
if r, ok := parseResticStatus(line); ok {
|
||||
m.offboxProgress.update(r)
|
||||
}
|
||||
}, full...)
|
||||
if err == nil || !offboxLockRe.Match(out) {
|
||||
return out, err
|
||||
}
|
||||
// Lock collision: fall back to the non-streaming step, which owns the unlock --remove-all
|
||||
// self-heal. Progress stalls for that one retry; correctness beats a moving bar.
|
||||
m.logger.Printf("[WARN] [offbox] %s hit a lock during a manual run — retrying via the self-healing step", label)
|
||||
return m.resticStep(ctx, env, base, label, args...)
|
||||
}
|
||||
|
||||
// beginManualProgress installs the progress sink for a manual run and returns the cleanup func.
|
||||
func (m *Manager) beginManualProgress() func() {
|
||||
m.offboxProgress.begin()
|
||||
started := time.Now()
|
||||
return func() {
|
||||
m.logger.Printf("[INFO] [offbox] manual run progress reporting ended after %s", time.Since(started).Round(time.Second))
|
||||
m.offboxProgress.end()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// v0.147.0 slice 4c — live progress for a MANUAL offsite run.
|
||||
//
|
||||
// These tests exercise the WHOLE path a real run takes: a fake restic emits canned `--json` status
|
||||
// lines through the streaming seam, and the assertion is on what the poll surface
|
||||
// (OffboxProgressSnapshot) reports — not on the parser in isolation. A parser that works but is
|
||||
// never wired to the snapshot would leave the customer looking at the same silent spinner, which is
|
||||
// the bug being fixed.
|
||||
//
|
||||
// RED-PROOF (run manually, both confirmed to fail):
|
||||
// 1. break the parser — flip `s.MessageType != "status"` to `== "status"` in parseResticStatus:
|
||||
// TestManualRunReportsParsedProgress fails ("percent = 0, want 42").
|
||||
// 2. drop the wiring — make resticBackupStep always delegate to resticStep:
|
||||
// the same test fails (no --json, no stream, no snapshot).
|
||||
|
||||
// jsonStatus is one restic --json status line.
|
||||
func jsonStatus(pct float64, done, total int64) string {
|
||||
return `{"message_type":"status","percent_done":` + ftoa(pct) + `,"total_bytes":` + itoa(total) + `,"bytes_done":` + itoa(done) + `}`
|
||||
}
|
||||
|
||||
func ftoa(f float64) string {
|
||||
// small helper — avoids strconv import noise in the canned lines
|
||||
switch f {
|
||||
case 0:
|
||||
return "0"
|
||||
case 0.42:
|
||||
return "0.42"
|
||||
case 1:
|
||||
return "1"
|
||||
}
|
||||
return "0.5"
|
||||
}
|
||||
|
||||
func itoa(i int64) string {
|
||||
if i == 0 {
|
||||
return "0"
|
||||
}
|
||||
var b []byte
|
||||
neg := i < 0
|
||||
if neg {
|
||||
i = -i
|
||||
}
|
||||
for i > 0 {
|
||||
b = append([]byte{byte('0' + i%10)}, b...)
|
||||
i /= 10
|
||||
}
|
||||
if neg {
|
||||
return "-" + string(b)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// TestManualRunReportsParsedProgress is the headline: a manual run driven by a fake restic that emits
|
||||
// --json status lines must make OffboxProgressSnapshot report the parsed percentage, byte counts and
|
||||
// the app currently being pushed.
|
||||
func TestManualRunReportsParsedProgress(t *testing.T) {
|
||||
e := newSharesOffboxEnv(t, "immich")
|
||||
if err := e.sett.SetSMBEnabled(false); err != nil { // keep this test to the app leg only
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var seen []OffboxProgress
|
||||
var sawJSONFlag bool
|
||||
e.m.SetOffboxStreamRunner(func(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error) {
|
||||
for _, a := range args {
|
||||
if a == "--json" {
|
||||
sawJSONFlag = true
|
||||
}
|
||||
}
|
||||
// Emit a scan phase (total not yet known), then real progress, sampling the published
|
||||
// snapshot after each line exactly as the polling page would.
|
||||
onLine(jsonStatus(0, 0, 0))
|
||||
seen = append(seen, e.m.OffboxProgressSnapshot())
|
||||
onLine(`{"message_type":"verbose_status","action":"unchanged"}`) // must be ignored, not fatal
|
||||
onLine(jsonStatus(0.42, 4200, 10000))
|
||||
seen = append(seen, e.m.OffboxProgressSnapshot())
|
||||
return []byte(`{"message_type":"summary","snapshot_id":"abc"}`), nil
|
||||
})
|
||||
// The non-streaming seam still serves every other restic call (cat config, forget, snapshots…).
|
||||
e.m.SetOffboxRunner(func(ctx context.Context, env []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[]`), nil
|
||||
}
|
||||
return []byte(""), nil
|
||||
})
|
||||
|
||||
if err := e.m.RunOffboxBackupWithProgress(context.Background()); err != nil {
|
||||
t.Fatalf("manual run: %v", err)
|
||||
}
|
||||
|
||||
if !sawJSONFlag {
|
||||
t.Fatal("the manual backup leg did not pass --json to restic — nothing could ever be parsed")
|
||||
}
|
||||
if len(seen) != 2 {
|
||||
t.Fatalf("expected 2 sampled snapshots, got %d", len(seen))
|
||||
}
|
||||
|
||||
// While restic is still scanning, total is unknown: report 0 rather than inventing a percentage.
|
||||
if seen[0].TotalBytes != 0 || seen[0].Percent != 0 {
|
||||
t.Errorf("scan phase: got %+v, want zeroed counters", seen[0])
|
||||
}
|
||||
if seen[0].CurrentApp != "immich" {
|
||||
t.Errorf("scan phase: current_app = %q, want %q", seen[0].CurrentApp, "immich")
|
||||
}
|
||||
|
||||
got := seen[1]
|
||||
if got.Percent != 42 {
|
||||
t.Errorf("percent = %v, want 42", got.Percent)
|
||||
}
|
||||
if got.BytesDone != 4200 || got.TotalBytes != 10000 {
|
||||
t.Errorf("bytes = %d/%d, want 4200/10000", got.BytesDone, got.TotalBytes)
|
||||
}
|
||||
if got.CurrentApp != "immich" {
|
||||
t.Errorf("current_app = %q, want %q", got.CurrentApp, "immich")
|
||||
}
|
||||
if got.TotalHuman == "" || got.DoneHuman == "" {
|
||||
t.Errorf("humanized byte strings not populated: %+v", got)
|
||||
}
|
||||
if !got.Active {
|
||||
t.Error("progress reported inactive during a manual run")
|
||||
}
|
||||
|
||||
// After the run the sink must be torn down, or the page would keep rendering a stale bar.
|
||||
if after := e.m.OffboxProgressSnapshot(); after.Active || after.Percent != 0 {
|
||||
t.Errorf("progress still published after the run: %+v", after)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNightlyRunStaysSilent pins the scope decision: the scheduled run must neither install the sink
|
||||
// nor pass --json, so its output format (and everything that greps it) is untouched.
|
||||
func TestNightlyRunStaysSilent(t *testing.T) {
|
||||
e := newSharesOffboxEnv(t, "immich")
|
||||
if err := e.sett.SetSMBEnabled(false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
e.m.SetOffboxStreamRunner(func(ctx context.Context, env []string, onLine func(string), args ...string) ([]byte, error) {
|
||||
t.Fatal("the nightly run used the streaming runner — progress must be manual-only")
|
||||
return nil, nil
|
||||
})
|
||||
var backupArgs []string
|
||||
e.m.SetOffboxRunner(func(ctx context.Context, env []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "backup") {
|
||||
backupArgs = append([]string{}, args...)
|
||||
}
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[]`), nil
|
||||
}
|
||||
return []byte(""), nil
|
||||
})
|
||||
|
||||
if err := e.m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("nightly run: %v", err)
|
||||
}
|
||||
if backupArgs == nil {
|
||||
t.Fatal("no backup call was made")
|
||||
}
|
||||
if contains(backupArgs, "--json") {
|
||||
t.Errorf("the nightly run passed --json: %v", backupArgs)
|
||||
}
|
||||
if p := e.m.OffboxProgressSnapshot(); p.Active {
|
||||
t.Errorf("the nightly run published progress: %+v", p)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusIgnoresNonStatus covers the lines restic actually interleaves with status
|
||||
// output. An unknown message_type must be ignored, never mistaken for progress — restic adds new
|
||||
// types between versions.
|
||||
func TestParseResticStatusIgnoresNonStatus(t *testing.T) {
|
||||
for _, line := range []string{
|
||||
`{"message_type":"summary","total_bytes_processed":123}`,
|
||||
`{"message_type":"verbose_status","action":"new"}`,
|
||||
`{"message_type":"error","error":{"message":"boom"}}`,
|
||||
`not json at all`,
|
||||
``,
|
||||
`{}`,
|
||||
} {
|
||||
if _, ok := parseResticStatus(line); ok {
|
||||
t.Errorf("parsed a non-status line as progress: %q", line)
|
||||
}
|
||||
}
|
||||
r, ok := parseResticStatus(jsonStatus(0.42, 4200, 10000))
|
||||
if !ok {
|
||||
t.Fatal("a real status line was not parsed")
|
||||
}
|
||||
if r.Percent != 42 || r.BytesDone != 4200 || r.TotalBytes != 10000 {
|
||||
t.Errorf("got %v %d %d, want 42 4200 10000", r.Percent, r.BytesDone, r.TotalBytes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusIncrementalRun is the case that MATTERS and the one a synthetic test suite
|
||||
// would never think to write. It is a real status line shape from restic 0.14 on the demo box:
|
||||
// an incremental push where nothing changed transfers no new bytes, so restic omits bytes_done
|
||||
// entirely (`omitempty`) and percent_done stays 0 — while files_done climbs steadily.
|
||||
//
|
||||
// Measured live: a 430MB immich push reported 0% for 40+ seconds and then completed. A byte-only
|
||||
// progress bar is therefore indistinguishable from a hang in the COMMON case, which is the exact
|
||||
// failure 4c exists to remove. The file counters must survive parsing so the page can fall back to
|
||||
// them.
|
||||
func TestParseResticStatusIncrementalRun(t *testing.T) {
|
||||
// bytes_done and files_done absent — restic's scan-start state.
|
||||
r, ok := parseResticStatus(`{"message_type":"status","percent_done":0,"total_files":1,"total_bytes":112}`)
|
||||
if !ok {
|
||||
t.Fatal("scan-start status line was not parsed")
|
||||
}
|
||||
if r.BytesDone != 0 || r.TotalBytes != 112 || r.TotalFiles != 1 {
|
||||
t.Errorf("scan-start: got %+v", r)
|
||||
}
|
||||
|
||||
// The incremental steady state: no bytes moving, files moving.
|
||||
r, ok = parseResticStatus(`{"message_type":"status","percent_done":0,"total_files":8123,"files_done":4110,"total_bytes":451130451}`)
|
||||
if !ok {
|
||||
t.Fatal("incremental status line was not parsed")
|
||||
}
|
||||
if r.BytesDone != 0 {
|
||||
t.Errorf("bytes_done = %d, want 0 (absent in the JSON)", r.BytesDone)
|
||||
}
|
||||
if r.FilesDone != 4110 || r.TotalFiles != 8123 {
|
||||
t.Errorf("file counters lost: got %d/%d, want 4110/8123 — the page has nothing left to move",
|
||||
r.FilesDone, r.TotalFiles)
|
||||
}
|
||||
if r.TotalBytes != 451130451 {
|
||||
t.Errorf("total_bytes = %d, want 451130451", r.TotalBytes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusKeepsCurrentFileAndElapsed — the case where NO counter can move: restic 0.14
|
||||
// only counts a file when it completes, so an app dominated by one big archive freezes bytes_done
|
||||
// AND files_done. Measured on the demo box: immich at 1 of 46 files, 0 bytes, for 42 seconds while
|
||||
// restic worked through a single ~430MB volume tar. current_files + seconds_elapsed are then the only
|
||||
// honest signals of liveness left, so losing them in parsing would put the bar back to looking hung.
|
||||
func TestParseResticStatusKeepsCurrentFileAndElapsed(t *testing.T) {
|
||||
line := `{"message_type":"status","seconds_elapsed":42,"percent_done":0,"total_files":46,` +
|
||||
`"files_done":1,"total_bytes":451130451,"current_files":["/mnt/hdd/felhom-data/backups/primary/immich/volumes/immich_upload.tar"]}`
|
||||
r, ok := parseResticStatus(line)
|
||||
if !ok {
|
||||
t.Fatal("status line was not parsed")
|
||||
}
|
||||
if r.ElapsedSec != 42 {
|
||||
t.Errorf("elapsed = %d, want 42", r.ElapsedSec)
|
||||
}
|
||||
if r.CurrentFile == "" {
|
||||
t.Fatal("current_file lost — with no counter moving this is the only liveness signal left")
|
||||
}
|
||||
if want := "immich_upload.tar"; !strings.HasSuffix(r.CurrentFile, want) {
|
||||
t.Errorf("current_file = %q, want it to end in %q", r.CurrentFile, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProgressKeepsLastKnownCurrentFile — restic omits current_files on some status ticks. Blanking
|
||||
// the label every other second is its own kind of flicker, so the last known value must persist.
|
||||
func TestProgressKeepsLastKnownCurrentFile(t *testing.T) {
|
||||
var st offboxProgressState
|
||||
st.begin()
|
||||
st.setApp("immich")
|
||||
st.update(resticProgress{CurrentFile: "/data/big.tar", ElapsedSec: 5, TotalFiles: 46, FilesDone: 1})
|
||||
st.update(resticProgress{CurrentFile: "", ElapsedSec: 7, TotalFiles: 46, FilesDone: 1}) // tick without current_files
|
||||
if got := st.snapshot().CurrentFile; got != "/data/big.tar" {
|
||||
t.Errorf("current_file = %q after a tick that omitted it, want the last known value", got)
|
||||
}
|
||||
if got := st.snapshot().ElapsedSec; got != 7 {
|
||||
t.Errorf("elapsed = %d, want it to keep advancing (7)", got)
|
||||
}
|
||||
// A new app must clear it — otherwise the previous app's file is shown against the next one.
|
||||
st.setApp("nextcloud")
|
||||
if got := st.snapshot().CurrentFile; got != "" {
|
||||
t.Errorf("current_file = %q after switching app, want cleared", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSetPhaseClearsAppScopedCounters — a run is not only the per-app loop. The shares leg and
|
||||
// forget --prune follow it and took 40 of a 57-second run on the demo box. Without a phase the card
|
||||
// kept showing the last app's finished counters ("calibre-web 8 / 8") for that whole tail, which is
|
||||
// the same frozen-looking silence 4c exists to remove, just relocated to the end of the run.
|
||||
func TestSetPhaseClearsAppScopedCounters(t *testing.T) {
|
||||
var st offboxProgressState
|
||||
st.begin()
|
||||
st.setApp("calibre-web")
|
||||
st.update(resticProgress{Percent: 100, BytesDone: 850000, TotalBytes: 850000, FilesDone: 8, TotalFiles: 8, CurrentFile: "/data/x"})
|
||||
|
||||
st.setPhase(OffboxPhaseRetention)
|
||||
got := st.snapshot()
|
||||
if got.Phase != OffboxPhaseRetention {
|
||||
t.Errorf("phase = %q, want %q", got.Phase, OffboxPhaseRetention)
|
||||
}
|
||||
if got.CurrentApp != "" || got.FilesDone != 0 || got.TotalFiles != 0 || got.BytesDone != 0 || got.CurrentFile != "" {
|
||||
t.Errorf("app-scoped counters survived the phase switch: %+v — the card would show the last "+
|
||||
"app's finished numbers against retention work", got)
|
||||
}
|
||||
// Starting another app must clear the phase again, or the card would stay on „Karbantartás".
|
||||
st.setApp("immich")
|
||||
if p := st.snapshot(); p.Phase != "" || p.CurrentApp != "immich" {
|
||||
t.Errorf("after setApp: phase=%q app=%q, want phase cleared and app set", p.Phase, p.CurrentApp)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseResticStatusClampsPercent — restic has been seen to report percent_done slightly above 1
|
||||
// near completion. A bar wider than its track is a visible bug.
|
||||
func TestParseResticStatusClampsPercent(t *testing.T) {
|
||||
r, ok := parseResticStatus(`{"message_type":"status","percent_done":1.04}`)
|
||||
if !ok || r.Percent != 100 {
|
||||
t.Errorf("percent = %v (ok=%v), want clamped to 100", r.Percent, ok)
|
||||
}
|
||||
r, ok = parseResticStatus(`{"message_type":"status","percent_done":-0.2}`)
|
||||
if !ok || r.Percent != 0 {
|
||||
t.Errorf("percent = %v (ok=%v), want clamped to 0", r.Percent, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLineTailKeepsOnlyTheTail — the --json stream of a large backup is megabytes of status spam, and
|
||||
// every caller uses this output for error diagnosis and lock-pattern matching. Buffering all of it
|
||||
// would be a memory leak proportional to backup size.
|
||||
func TestLineTailKeepsOnlyTheTail(t *testing.T) {
|
||||
var tl lineTail
|
||||
for i := 0; i < 500; i++ {
|
||||
tl.add("line-" + itoa(int64(i)))
|
||||
}
|
||||
out := string(tl.bytes())
|
||||
if strings.Contains(out, "line-0\n") {
|
||||
t.Error("the oldest line survived — the tail is unbounded")
|
||||
}
|
||||
if !strings.Contains(out, "line-499") {
|
||||
t.Error("the newest line was dropped")
|
||||
}
|
||||
if n := strings.Count(out, "\n"); n > 64 {
|
||||
t.Errorf("tail kept %d lines, want a small bounded number", n)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,455 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Offsite reconstitution (R-43, v0.148.0) — the leg that was missing.
|
||||
//
|
||||
// Until v0.148.0 NO offsite path could restore a database. The two „visszaállítás" buttons staged
|
||||
// files into a scratch folder and never touched postgres; the place-to-live button merged only the
|
||||
// files MISSING from the live tree (`rsync --ignore-existing`) and never replayed a dump. For a
|
||||
// DB-indexed app — most of the catalog — that combination cannot bring content back: the bytes
|
||||
// return and the application still cannot see them, because its index lives in the database.
|
||||
// Measured live on 2026-07-19 (DIAG-immich-restore-2026-07-19): 11 photos, files intact on disk,
|
||||
// timeline empty, two "successful" restores that merged 0 files.
|
||||
//
|
||||
// ReconstituteFromOffsite is the honest version of that operation: it takes the CHOSEN snapshot's
|
||||
// coherent pair and makes the live app equal to it — files overwritten to the snapshot's version,
|
||||
// database replayed from the same snapshot's dump, app restarted. It is deliberately a different
|
||||
// function from PlaceOffsiteRestore rather than a flag on it, because the two have opposite file
|
||||
// semantics and conflating them is exactly how the missing-only merge came to be presented as a
|
||||
// restore.
|
||||
//
|
||||
// Two invariants hold throughout:
|
||||
//
|
||||
// - NOTHING IS EVER DELETED. The file copy overwrites and adds; it never carries `--delete`. A
|
||||
// file the customer created after the snapshot survives the restore as an extra. That is the
|
||||
// house boundary — a restore that silently removed newer work would be a data-loss event
|
||||
// wearing a recovery button's label.
|
||||
// - THE UNDO EXISTS BEFORE THE ACT. A safety dump of the live database is written, and verified
|
||||
// present on disk, BEFORE anything is stopped, overwritten or replayed. If that dump cannot be
|
||||
// taken, the whole operation refuses with zero changes — a replay whose previous state was not
|
||||
// captured is not a restore, it is an overwrite with no way back.
|
||||
|
||||
// offsitePreDump runs the coherence pre-phase's dump leg (nil seam → runDBDumpsInternal, which also
|
||||
// refreshes the recovery units so the manifests enumerate the dumps just written). Extracted as a
|
||||
// seam because the ORDER — dumps strictly before the restic capture — is the entire mechanism of
|
||||
// R-44, and an ordering guarantee that no test can observe is one refactor away from silently
|
||||
// reverting to the behaviour that produced DIAG-immich-restore-2026-07-19.
|
||||
func (m *Manager) offsitePreDump(ctx context.Context) error {
|
||||
if m.offsitePreDumpFn != nil {
|
||||
return m.offsitePreDumpFn(ctx)
|
||||
}
|
||||
return m.runDBDumpsInternal(ctx)
|
||||
}
|
||||
|
||||
// SetOffsitePreDumpFn overrides the offsite dump pre-phase (tests; no Docker needed).
|
||||
func (m *Manager) SetOffsitePreDumpFn(fn func(ctx context.Context) error) { m.offsitePreDumpFn = fn }
|
||||
|
||||
// preRestoreDumpPrefix marks the safety dumps taken immediately before a reconstitution. They live
|
||||
// in the app's own unit db-dumps dir so `ListDumpFiles` surfaces them beside the regular dumps —
|
||||
// they ARE the undo, and an undo the customer cannot see is not much of one. The regular replay
|
||||
// loop matches `<stack>-<dbtype>.sql` exactly, so a prefixed file is never mistaken for a source.
|
||||
const preRestoreDumpPrefix = "pre-restore-"
|
||||
|
||||
// OffsiteReconstituteResult reports what a reconstitution actually did, so the flash can state an
|
||||
// OUTCOME instead of a mechanism. Every field here exists because the v0.147 flash could not say it.
|
||||
type OffsiteReconstituteResult struct {
|
||||
SnapshotID string
|
||||
FilesPlaced int
|
||||
DBsReplayed int
|
||||
SafetyDump string // path of the pre-restore dump (the undo), "" when the app has no DB
|
||||
DumpsAt time.Time // when the snapshot's DB half was taken (zero = unknown/legacy unit)
|
||||
OffsiteRunID string // "" for a pre-v0.148 snapshot — an unverified pair
|
||||
Skewed bool // the snapshot carries no coherence stamp: files and DB may differ in age
|
||||
LooksEmpty bool // R-44 sniff on the dump about to be replayed
|
||||
}
|
||||
|
||||
// fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite).
|
||||
// Deliberately NOT placeCopier(): that one is `--ignore-existing`, whose whole purpose is to leave
|
||||
// live files alone, which is precisely what a full restore must not do.
|
||||
func (m *Manager) fullPlaceCopier() func(src, dst string) (int, error) {
|
||||
if m.offboxFullPlaceCopier != nil {
|
||||
return m.offboxFullPlaceCopier
|
||||
}
|
||||
return rsyncRestoreOverwrite
|
||||
}
|
||||
|
||||
// rsyncRestoreOverwrite copies src over dst: `rsync -a --itemize-changes`, with NO
|
||||
// `--ignore-existing` (a changed file becomes the snapshot's version) and NO `--delete` (an extra
|
||||
// file at dst survives). Returns the number of regular files transferred.
|
||||
func rsyncRestoreOverwrite(src, dst string) (int, error) {
|
||||
if err := os.MkdirAll(dst, 0755); err != nil {
|
||||
return 0, fmt.Errorf("mkdir %s: %w", dst, err)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
|
||||
defer cancel()
|
||||
cmd := exec.CommandContext(ctx, "rsync", "-a", "--itemize-changes",
|
||||
strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/")
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out)))
|
||||
}
|
||||
return countRestoredFiles(string(out)), nil
|
||||
}
|
||||
|
||||
// writeSafetyDump dumps every live database of stack into the app's unit db-dumps dir under the
|
||||
// `pre-restore-` prefix, and returns the first dump's path. Returns ("", nil) when the app has no
|
||||
// database at all — a no-DB app has nothing to undo and must flow exactly as it did before
|
||||
// v0.148.0 (no dump, no replay, no behaviour change).
|
||||
//
|
||||
// A discovered database that CANNOT be dumped is a hard error: it means the undo would not exist.
|
||||
func (m *Manager) writeSafetyDump(ctx context.Context, stackName, nsRoot string) (string, error) {
|
||||
discover := m.discoverDBs
|
||||
if discover == nil {
|
||||
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
||||
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||||
}
|
||||
}
|
||||
dbs, err := discover(ctx)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("a biztonsági mentés előtt nem sikerült felderíteni az adatbázisokat: %w", err)
|
||||
}
|
||||
var mine []DiscoveredDB
|
||||
for _, db := range dbs {
|
||||
if db.StackName == stackName {
|
||||
mine = append(mine, db)
|
||||
}
|
||||
}
|
||||
if len(mine) == 0 {
|
||||
return "", nil // no DB → nothing to undo → scenario E flows unchanged
|
||||
}
|
||||
|
||||
dumpDir := AppDBDumpPath(nsRoot, stackName)
|
||||
if err := os.MkdirAll(dumpDir, 0755); err != nil {
|
||||
return "", fmt.Errorf("a biztonsági mentés könyvtára nem hozható létre: %w", err)
|
||||
}
|
||||
stamp := time.Now().UTC().Format("20060102T150405Z")
|
||||
first := ""
|
||||
for _, db := range mine {
|
||||
res := m.dumpForSafety(ctx, db, dumpDir)
|
||||
if res.Error != nil {
|
||||
return "", fmt.Errorf("a jelenlegi adatbázis biztonsági mentése sikertelen (%s): %w — a visszaállítás nem indult el", db.ContainerName, res.Error)
|
||||
}
|
||||
// DumpOne writes `<stack>-<dbtype>.sql`; rename it under the safety prefix so it can never be
|
||||
// picked up as a replay SOURCE and can never overwrite the app's real dump.
|
||||
safe := filepath.Join(dumpDir, fmt.Sprintf("%s%s-%s-%s.sql", preRestoreDumpPrefix, stamp, stackName, db.DBType))
|
||||
if res.FilePath != safe {
|
||||
if err := os.Rename(res.FilePath, safe); err != nil {
|
||||
return "", fmt.Errorf("a biztonsági mentés véglegesítése sikertelen: %w", err)
|
||||
}
|
||||
}
|
||||
if first == "" {
|
||||
first = safe
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] %s: pre-restore safety dump written → %s (%s)", stackName, filepath.Base(safe), humanizeBytes(res.Size))
|
||||
}
|
||||
return first, nil
|
||||
}
|
||||
|
||||
// dumpForSafety is the DumpOne seam for the safety dump (tests inject; nil → the real DumpOne).
|
||||
func (m *Manager) dumpForSafety(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
|
||||
if m.safetyDumpFn != nil {
|
||||
return m.safetyDumpFn(ctx, db, dumpDir)
|
||||
}
|
||||
return DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
|
||||
}
|
||||
|
||||
// ReconstituteFromOffsite makes the live app equal to a restored full-scratch snapshot: files
|
||||
// overwritten to the snapshot's version (extras survive, nothing deleted), then the snapshot's own
|
||||
// DB dump replayed, with a safety dump of the current database taken first. Requires a completed
|
||||
// FULL scratch restore (RestoreOffboxScratch with full=true). Single-flight.
|
||||
func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string) (OffsiteReconstituteResult, error) {
|
||||
var res OffsiteReconstituteResult
|
||||
if !m.OffboxConfigured() {
|
||||
return res, fmt.Errorf("off-box backup not configured")
|
||||
}
|
||||
if !isSafeStackName(stack) {
|
||||
return res, fmt.Errorf("invalid stack name")
|
||||
}
|
||||
if m.stackProvider == nil {
|
||||
return res, fmt.Errorf("stack provider not configured")
|
||||
}
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
return res, fmt.Errorf("egy másik mentési/visszaállítási művelet már fut")
|
||||
}
|
||||
defer m.releaseRunning()
|
||||
|
||||
scratch, _, err := m.offboxRestoreScratchDir(stack)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
if _, sErr := os.Stat(scratch); sErr != nil {
|
||||
return res, fmt.Errorf("nincs előkészített teljes visszaállítás — futtass előbb egy teljes visszaállítást")
|
||||
}
|
||||
id, paths, err := m.offboxLatestSnapshot(ctx, stack)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.SnapshotID = id
|
||||
|
||||
hdd := strings.TrimSpace(m.stackProvider.GetStackHDDPath(stack))
|
||||
if hdd == "" {
|
||||
// R-253: the same sentence the restore page now shows, so the page and the refusal cannot
|
||||
// drift apart again. It is a REFUSAL, not a failure — the data is untouched and the customer
|
||||
// has one step to take. The restore deliberately does NOT deploy the app itself: the
|
||||
// destination is the app's own HDD path, which is a drive the CUSTOMER chooses at deploy
|
||||
// time, and picking it for them is the decision this whole recovery path exists to leave
|
||||
// with them.
|
||||
return res, fmt.Errorf("a(z) %s nincs telepítve, ezért nincs hová visszaállítani az adatait — "+
|
||||
"telepítsd újra az alkalmazást (Alkalmazások), utána ez a visszaállítás működni fog", stack)
|
||||
}
|
||||
liveNs := m.namespaceRoot(hdd)
|
||||
|
||||
placements, err := mapOffsiteRestorePaths(paths, stack, scratch, liveNs)
|
||||
if err != nil {
|
||||
return res, err // whole-placement refusal (no partial writes)
|
||||
}
|
||||
// Stat pre-pass over EVERY placement before the first copy — an incomplete scratch (e.g. only a
|
||||
// unit-only restore was run) refuses with ZERO copies.
|
||||
for _, pl := range placements {
|
||||
if _, sErr := os.Stat(pl.src); sErr != nil {
|
||||
return res, fmt.Errorf("a teljes visszaállítás hiányos (%s nincs meg) — futtass előbb egy teljes visszaállítást", filepath.Base(pl.src))
|
||||
}
|
||||
}
|
||||
|
||||
// The snapshot's coherence stamp, read from the RESTORED unit manifest (not the live one).
|
||||
scratchUnit := ""
|
||||
for _, pl := range placements {
|
||||
if pl.isUnit {
|
||||
scratchUnit = pl.src
|
||||
break
|
||||
}
|
||||
}
|
||||
if scratchUnit == "" {
|
||||
return res, fmt.Errorf("a pillanatképben nincs mentési egység — a visszaállítás nem indítható")
|
||||
}
|
||||
scratchDumpDir := filepath.Join(scratchUnit, "db-dumps")
|
||||
if man := readManifest(filepath.Join(scratchUnit, "manifest.json")); man != nil {
|
||||
res.OffsiteRunID = man.OffsiteRunID
|
||||
if man.DumpsAt != "" {
|
||||
if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil {
|
||||
res.DumpsAt = t
|
||||
}
|
||||
}
|
||||
}
|
||||
// A pre-v0.148 snapshot carries no stamp: its dump was whatever the 02:30 local run left behind,
|
||||
// so the pair's two halves may be hours or days apart. Surfaced, never blocked — the confirm
|
||||
// dialog says so and the safety dump makes it reversible.
|
||||
res.Skewed = res.OffsiteRunID == ""
|
||||
res.LooksEmpty = m.sniffScratchDump(scratchDumpDir, stack)
|
||||
|
||||
// --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------
|
||||
// Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir,
|
||||
// so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first
|
||||
// mutation so the refusal below costs nothing.
|
||||
var dbServices []string
|
||||
if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
|
||||
svcs, dsErr := DBServiceNames(composePath)
|
||||
if dsErr != nil {
|
||||
// "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse.
|
||||
m.logger.Printf("[WARN] [offbox] %s: could not read the live compose services: %v", stack, dsErr)
|
||||
}
|
||||
dbServices = svcs
|
||||
}
|
||||
|
||||
// --- THE UNDO, BEFORE THE ACT ---------------------------------------------------------------
|
||||
// Taken while the stack is still UP (a stopped database cannot be dumped) and before a single
|
||||
// byte is overwritten, so a failure here aborts with the live app completely untouched.
|
||||
safety, err := m.writeSafetyDump(ctx, stack, liveNs)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.SafetyDump = safety
|
||||
hasDB := safety != ""
|
||||
if hasDB {
|
||||
if _, sErr := os.Stat(safety); sErr != nil {
|
||||
// Fail-closed: never replay when the undo is not verifiably on disk.
|
||||
return res, fmt.Errorf("a biztonsági mentés nem található a lemezen — a visszaállítás biztonsági okból nem indult el")
|
||||
}
|
||||
// Fail-closed (R-47): the app HAS a database but no compose service can be identified to
|
||||
// start alone for the replay. The only alternative would be to start everything and replay
|
||||
// into the race that produced H4 — refusing with the live app untouched is the better outcome.
|
||||
if len(dbServices) == 0 {
|
||||
return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stack)
|
||||
}
|
||||
}
|
||||
|
||||
// --- FILES ----------------------------------------------------------------------------------
|
||||
// R-166: mark the stop→restore→start window BEFORE stopping. A controller killed anywhere inside
|
||||
// it used to leave the app down with nothing on disk recording that it was owed a restart — and a
|
||||
// full offsite restore is a LONG window, so this is the shape most likely to be interrupted.
|
||||
if err := m.appStop.Begin("offbox-reconstitute:"+stack, ReasonOffboxReconstitute, []string{stack}); err != nil {
|
||||
return res, fmt.Errorf("a(z) %s leállítása előtti jelölő nem menthető: %w", stack, err)
|
||||
}
|
||||
// restartStack starts the app and clears the marker ONLY when the start actually succeeded — a
|
||||
// failed start leaves the marker so the next startup retries. Every bring-up below goes through
|
||||
// it; a bare StartStack here would clear nothing and strand the marker on the success path.
|
||||
restartStack := func() error {
|
||||
err := m.stackProvider.StartStack(stack)
|
||||
if err == nil {
|
||||
m.appStop.End()
|
||||
}
|
||||
return err
|
||||
}
|
||||
if err := m.stackProvider.StopStack(stack); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err)
|
||||
}
|
||||
copier := m.fullPlaceCopier()
|
||||
for _, pl := range placements {
|
||||
if pl.isUnit {
|
||||
// The live recovery unit is still never overwritten — it is the LOCAL restore path's
|
||||
// source and clobbering it would trade one recovery route for another. The snapshot's
|
||||
// dump is replayed from the scratch unit instead, so nothing is lost by skipping it.
|
||||
continue
|
||||
}
|
||||
n, cErr := copier(pl.src, pl.dst)
|
||||
if cErr != nil {
|
||||
// Best-effort bring-up: leaving the app stopped after a partial copy would turn a failed
|
||||
// restore into an outage.
|
||||
if sErr := restartStack(); sErr != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: restart after failed placement also failed: %v", stack, sErr)
|
||||
}
|
||||
return res, fmt.Errorf("a(z) %s fájljainak visszaállítása sikertelen: %w", stack, cErr)
|
||||
}
|
||||
res.FilesPlaced += n
|
||||
}
|
||||
|
||||
// --- DATABASE -------------------------------------------------------------------------------
|
||||
// The DB container must be UP for the replay (ImportDump talks to it with its own discovered
|
||||
// credentials), but NOTHING ELSE may be — R-47. Until v0.153.0 this was a full StartStack, which
|
||||
// gave the application a window to rebuild the very schema objects the dump was about to create:
|
||||
// measured at 2 s on 2026-07-19, and the replay aborted `relation "clip_index" already exists`
|
||||
// under ON_ERROR_STOP=1 (H4). Starting only the database service closes that window entirely.
|
||||
if hasDB {
|
||||
if err := m.stackProvider.StartStackServices(stack, dbServices); err != nil {
|
||||
// Best-effort bring-up: a failed restore must not also be an outage.
|
||||
if sErr := restartStack(); sErr != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: full start after failed DB-only start also failed: %v", stack, sErr)
|
||||
}
|
||||
return res, fmt.Errorf("a(z) %s adatbázis-szolgáltatásának indítása sikertelen: %w", stack, err)
|
||||
}
|
||||
n, iErr := m.reimportDBDumpsFrom(ctx, stack, scratchDumpDir)
|
||||
res.DBsReplayed = n
|
||||
if iErr != nil {
|
||||
if sErr := restartStack(); sErr != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: full start after failed replay also failed: %v", stack, sErr)
|
||||
}
|
||||
return res, fmt.Errorf("az adatbázis visszaállítása sikertelen: %w — a korábbi állapot mentése megvan: %s", iErr, filepath.Base(safety))
|
||||
}
|
||||
}
|
||||
if err := restartStack(); err != nil {
|
||||
return res, fmt.Errorf("a(z) %s újraindítása sikertelen a fájlok visszaállítása után: %w", stack, err)
|
||||
}
|
||||
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [offbox] reconstituted %s from snapshot %s: %d file(s) placed, %d DB dump(s) replayed, safety dump=%s, skewed=%v",
|
||||
stack, id, res.FilesPlaced, res.DBsReplayed, filepath.Base(safety), res.Skewed)
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// OffsitePairInfo describes the {DB, files} pair sitting in a prepared full-restore scratch, so the
|
||||
// confirm dialog can tell the customer what they are about to restore BEFORE they commit to it.
|
||||
// Everything here is honesty-surface: none of it blocks the operation.
|
||||
type OffsitePairInfo struct {
|
||||
Ready bool
|
||||
DumpsAt time.Time // when the DB half was taken (zero = legacy unit, age unknown)
|
||||
Skewed bool // no coherence stamp → the two halves may be from different times
|
||||
LooksEmpty bool // R-44 sniff: the dump has an accounts table with no rows
|
||||
HasDump bool
|
||||
}
|
||||
|
||||
// OffsiteScratchPair reads the prepared scratch's unit manifest and reports what the pair looks
|
||||
// like. Cheap and read-only — safe to call from a page render.
|
||||
func (m *Manager) OffsiteScratchPair(stack string) OffsitePairInfo {
|
||||
var info OffsitePairInfo
|
||||
if !isSafeStackName(stack) {
|
||||
return info
|
||||
}
|
||||
scratch, _, err := m.offboxRestoreScratchDir(stack)
|
||||
if err != nil {
|
||||
return info
|
||||
}
|
||||
// The unit sits at <scratch>/<oldNs>/backups/primary/<stack>; the old namespace is unknown here,
|
||||
// so find it rather than reconstructing it.
|
||||
unit := findScratchUnitDir(scratch, stack)
|
||||
if unit == "" {
|
||||
return info
|
||||
}
|
||||
info.Ready = true
|
||||
dumpDir := filepath.Join(unit, "db-dumps")
|
||||
if entries, rErr := os.ReadDir(dumpDir); rErr == nil {
|
||||
for _, e := range entries {
|
||||
if !e.IsDir() && filepath.Ext(e.Name()) == ".sql" && !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
|
||||
info.HasDump = true
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if man := readManifest(filepath.Join(unit, "manifest.json")); man != nil {
|
||||
if man.DumpsAt != "" {
|
||||
if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil {
|
||||
info.DumpsAt = t
|
||||
}
|
||||
}
|
||||
info.Skewed = man.OffsiteRunID == ""
|
||||
} else {
|
||||
info.Skewed = true
|
||||
}
|
||||
if info.HasDump {
|
||||
info.LooksEmpty = m.sniffScratchDump(dumpDir, stack)
|
||||
}
|
||||
return info
|
||||
}
|
||||
|
||||
// findScratchUnitDir locates `backups/primary/<stack>` anywhere under a restored scratch. restic
|
||||
// rebuilds absolute source paths under the target, and the snapshot may have come from a drive that
|
||||
// no longer exists on this box, so the prefix cannot be assumed.
|
||||
func findScratchUnitDir(scratch, stack string) string {
|
||||
found := ""
|
||||
suffix := filepath.Join("backups", "primary", stack)
|
||||
_ = filepath.Walk(scratch, func(path string, fi os.FileInfo, err error) error {
|
||||
if err != nil || found != "" {
|
||||
return nil //nolint:nilerr // a walk error on one branch must not abort the search
|
||||
}
|
||||
if fi.IsDir() && strings.HasSuffix(path, suffix) {
|
||||
found = path
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return found
|
||||
}
|
||||
|
||||
// sniffScratchDump runs the R-44 content sniff over the dump about to be replayed. Best-effort and
|
||||
// warn-level: any failure to read simply reports "no warning", because a sniff that blocks a
|
||||
// restore is worse than the skew it describes.
|
||||
func (m *Manager) sniffScratchDump(dumpDir, stack string) bool {
|
||||
entries, err := os.ReadDir(dumpDir)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
for _, e := range entries {
|
||||
name := e.Name()
|
||||
if e.IsDir() || filepath.Ext(name) != ".sql" || strings.HasPrefix(name, preRestoreDumpPrefix) {
|
||||
continue
|
||||
}
|
||||
dbType := DBTypePostgres
|
||||
if strings.Contains(name, string(DBTypeMariaDB)) {
|
||||
dbType = DBTypeMariaDB
|
||||
}
|
||||
if v := ValidateDump(filepath.Join(dumpDir, name), dbType); v.LooksEmpty {
|
||||
m.logger.Printf("[WARN] [offbox] %s: the snapshot dump %s has no account rows — it may predate the customer's data", stack, name)
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,455 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-43/R-44 (v0.148.0) — the coherent-pair + true-restore tests.
|
||||
//
|
||||
// These exist because the product shipped a restore button for months that could not restore.
|
||||
// DIAG-immich-restore-2026-07-19: 11 photos, files intact, timeline empty, two "successful"
|
||||
// restores that merged 0 files and never touched postgres. Every test below asserts a behaviour
|
||||
// whose absence produced that outcome, so each one is a regression guard for a real incident
|
||||
// rather than a description of the current implementation.
|
||||
|
||||
// recordingProvider records stop/start call ORDER so the reconstitution sequence can be asserted.
|
||||
//
|
||||
// R-47 widened it: it now also records the DB-ONLY bring-up and, critically, whether a FULL start
|
||||
// has happened yet — the state the replay must observe as `false`. That single flag is what
|
||||
// separates the fixed sequence from the one that produced H4, in which the whole stack was already
|
||||
// up (and rebuilding its own schema) when the dump replay began.
|
||||
type recordingProvider struct {
|
||||
offbox3aProvider
|
||||
calls []string
|
||||
composePath string // the LIVE compose the DB-service resolver reads
|
||||
gotServices []string // services passed to StartStackServices
|
||||
fullStarted bool // a FULL StartStack has happened
|
||||
startSvcErr error // injected StartStackServices failure
|
||||
}
|
||||
|
||||
func (p *recordingProvider) StopStack(string) error { p.calls = append(p.calls, "stop"); return nil }
|
||||
func (p *recordingProvider) StartStack(string) error {
|
||||
p.fullStarted = true
|
||||
p.calls = append(p.calls, "start")
|
||||
return nil
|
||||
}
|
||||
func (p *recordingProvider) StartStackServices(_ string, services []string) error {
|
||||
p.gotServices = append([]string(nil), services...)
|
||||
p.calls = append(p.calls, "startsvc:"+strings.Join(services, ","))
|
||||
return p.startSvcErr
|
||||
}
|
||||
func (p *recordingProvider) GetStackComposePath(string) (string, bool) {
|
||||
return p.composePath, p.composePath != ""
|
||||
}
|
||||
|
||||
// The app really is up again after StartStack, so the post-restore health wait returns at once.
|
||||
// Leaving it false would make each test sit through the full 90s deadline.
|
||||
func (p *recordingProvider) RefreshAndIsRunning(string) bool { return true }
|
||||
|
||||
// recoveryProvider adds the recovery info CaptureRecoveryUnit needs (the shared 3a provider has none).
|
||||
type recoveryProvider struct {
|
||||
offbox3aProvider
|
||||
stackDir string
|
||||
}
|
||||
|
||||
func (p *recoveryProvider) GetStackRecoveryInfo(name string) (RecoveryInfo, bool) {
|
||||
return RecoveryInfo{DisplayName: "Immich", StackDir: p.stackDir}, name == "immich"
|
||||
}
|
||||
|
||||
// pgDump builds a structurally valid postgres dump big enough to clear ValidateDump's 100-byte
|
||||
// floor, with the accounts-table COPY block carrying `rows` rows. The R-44 sniff runs only on a
|
||||
// dump that already passes structural validation, so a toy fixture would silently skip it.
|
||||
func pgDump(rows int) string {
|
||||
const head = `-- PostgreSQL database dump
|
||||
-- Dumped from database version 16.10
|
||||
SET statement_timeout = 0;
|
||||
SET lock_timeout = 0;
|
||||
SET client_encoding = 'UTF8';
|
||||
CREATE TABLE public.asset (id uuid NOT NULL);
|
||||
CREATE TABLE public."user" (id uuid NOT NULL, email text);
|
||||
COPY public."user" (id, email) FROM stdin;
|
||||
`
|
||||
var b strings.Builder
|
||||
b.WriteString(head)
|
||||
for i := 0; i < rows; i++ {
|
||||
b.WriteString("id-x\tuser@example.invalid\n")
|
||||
}
|
||||
b.WriteString("\\.\n") // the COPY-block terminator
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// reconFixture builds a manager with a COMPLETED full scratch for `immich`, a snapshot whose unit
|
||||
// carries the given coherence stamp, and injectable copy/dump/import seams.
|
||||
func reconFixture(t *testing.T, runID, dumpsAt string, dumpBody string) (*Manager, *recordingProvider, *[]string) {
|
||||
t.Helper()
|
||||
drive := t.TempDir()
|
||||
m, sett := newOffboxManager(t)
|
||||
prov := &recordingProvider{offbox3aProvider: offbox3aProvider{
|
||||
hdd: map[string]string{"immich": drive}, binds: map[string][]ClassifiedBind{}, has: map[string]bool{},
|
||||
}}
|
||||
m.SetStackProvider(prov)
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: drive, Label: "USB", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// The LIVE compose the reconstitution reads to learn WHICH service holds the database (R-47).
|
||||
// Immich-shaped on purpose: an app service, a redis service that must never be mistaken for a
|
||||
// database, and a top-level `volumes:` key whose entry looks exactly like a service to a line scan.
|
||||
liveStackDir := t.TempDir()
|
||||
prov.composePath = filepath.Join(liveStackDir, "docker-compose.yml")
|
||||
if err := os.WriteFile(prov.composePath, []byte(immichLikeCompose), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
scratch, liveNs, err := m.offboxRestoreScratchDir("immich")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
oldNs := "/felhomdata/ns"
|
||||
unitP := oldNs + "/backups/primary/immich"
|
||||
dataP := oldNs + "/appdata/immich"
|
||||
placements, err := mapOffsiteRestorePaths([]string{unitP, dataP}, "immich", scratch, liveNs)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, pl := range placements {
|
||||
if err := os.MkdirAll(pl.src, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if pl.isUnit {
|
||||
dd := filepath.Join(pl.src, "db-dumps")
|
||||
if err := os.MkdirAll(dd, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if dumpBody != "" {
|
||||
if err := os.WriteFile(filepath.Join(dd, "immich-postgres.sql"), []byte(dumpBody), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
man := &RecoveryManifest{SchemaVersion: 1, AppName: "immich", OffsiteRunID: runID, DumpsAt: dumpsAt}
|
||||
if err := writeManifest(filepath.Join(pl.src, "manifest.json"), man); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
m.SetOffboxFreeFn(func(string) int64 { return 100 << 30 })
|
||||
m.SetOffboxSizer(func(string) int64 { return 1 << 20 })
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[{"short_id":"snap1","time":"2026-07-19T06:00:00Z","paths":["` + unitP + `","` + dataP + `"]}]`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
|
||||
// Seams: one DB, a safety dump that really writes a file, and a recording importer.
|
||||
db := DiscoveredDB{StackName: "immich", ContainerName: "immich-postgres", DBType: DBTypePostgres}
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{db}, nil }
|
||||
m.SetSafetyDumpFn(func(_ context.Context, d DiscoveredDB, dir string) DumpResult {
|
||||
p := filepath.Join(dir, "immich-postgres.sql")
|
||||
_ = os.MkdirAll(dir, 0o755)
|
||||
_ = os.WriteFile(p, []byte(pgDump(1)), 0o644)
|
||||
return DumpResult{DB: d, FilePath: p, Size: 42}
|
||||
})
|
||||
var imported []string
|
||||
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||
imported = append(imported, p)
|
||||
return nil
|
||||
}
|
||||
m.SetOffboxFullPlaceCopier(func(_, _ string) (int, error) { return 3, nil })
|
||||
return m, prov, &imported
|
||||
}
|
||||
|
||||
// TestReconstituteReplaysDBAndOrdersOperations is Scenario C: the whole point of R-43. A restore of
|
||||
// a DB-indexed app must stop the app, place files, restart it and REPLAY the snapshot's dump — and
|
||||
// the safety dump must exist before any of it. Before v0.148.0 the replay simply did not happen,
|
||||
// which is why the photos never came back.
|
||||
func TestReconstituteReplaysDBAndOrdersOperations(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "20260719T060000Z", "2026-07-19T06:00:00Z", pgDump(1))
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err != nil {
|
||||
t.Fatalf("reconstitute: %v", err)
|
||||
}
|
||||
if res.DBsReplayed != 1 {
|
||||
t.Fatalf("expected the snapshot dump to be replayed exactly once, got %d — this is the R-43 defect", res.DBsReplayed)
|
||||
}
|
||||
if len(*imported) != 1 || !strings.Contains((*imported)[0], "immich-postgres.sql") {
|
||||
t.Fatalf("expected an import of the snapshot dump, got %v", *imported)
|
||||
}
|
||||
// The dump replayed must come from the SCRATCH unit, never the live one: the live unit is
|
||||
// deliberately not overwritten, so replaying from it would replay the CURRENT database back over
|
||||
// itself and restore nothing.
|
||||
if !strings.Contains((*imported)[0], "offsite-restore") {
|
||||
t.Fatalf("replay source must be the restored scratch unit, got %s", (*imported)[0])
|
||||
}
|
||||
if res.FilesPlaced != 3 {
|
||||
t.Fatalf("expected the userdata placement to be counted, got %d", res.FilesPlaced)
|
||||
}
|
||||
// stop BEFORE the file copy; then ONLY the database service up for the replay (R-47 — a full
|
||||
// start here is the H4 race); the full start comes last.
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,startsvc:immich-postgres,start" {
|
||||
t.Fatalf("expected stop → db-only start → full start around the restore, got %q", got)
|
||||
}
|
||||
if res.SafetyDump == "" {
|
||||
t.Fatal("no safety dump recorded — the undo must exist")
|
||||
}
|
||||
if _, err := os.Stat(res.SafetyDump); err != nil {
|
||||
t.Fatalf("safety dump not on disk: %v", err)
|
||||
}
|
||||
if !strings.HasPrefix(filepath.Base(res.SafetyDump), preRestoreDumpPrefix) {
|
||||
t.Fatalf("safety dump must carry the pre-restore prefix so it is never replayed as a source, got %s", filepath.Base(res.SafetyDump))
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconstituteRefusesWhenSafetyDumpFails is the RED-PROOF for the undo invariant: a replay whose
|
||||
// previous state was not captured is an overwrite with no way back, so it must not happen at all —
|
||||
// and it must abort with the live app untouched (no stop, no copy).
|
||||
func TestReconstituteRefusesWhenSafetyDumpFails(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
|
||||
m.SetSafetyDumpFn(func(_ context.Context, d DiscoveredDB, _ string) DumpResult {
|
||||
return DumpResult{DB: d, Error: context.DeadlineExceeded}
|
||||
})
|
||||
var copied bool
|
||||
m.SetOffboxFullPlaceCopier(func(_, _ string) (int, error) { copied = true; return 1, nil })
|
||||
|
||||
_, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err == nil {
|
||||
t.Fatal("expected a refusal when the safety dump cannot be taken")
|
||||
}
|
||||
if len(*imported) != 0 {
|
||||
t.Fatalf("REPLAYED WITHOUT AN UNDO — the exact thing the invariant forbids: %v", *imported)
|
||||
}
|
||||
if copied {
|
||||
t.Fatal("files were overwritten despite the refusal — the abort must leave live data untouched")
|
||||
}
|
||||
if len(prov.calls) != 0 {
|
||||
t.Fatalf("the app was stopped despite the refusal, got %v", prov.calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconstituteNoDBAppMakesNoDumpOrImportCalls is Scenario E: an app without a database must flow
|
||||
// exactly as before — no safety dump, no replay — so the new leg cannot regress the simple case.
|
||||
func TestReconstituteNoDBAppMakesNoDumpOrImportCalls(t *testing.T) {
|
||||
m, _, imported := reconFixture(t, "run1", "2026-07-19T06:00:00Z", "")
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return nil, nil }
|
||||
dumped := 0
|
||||
m.SetSafetyDumpFn(func(_ context.Context, d DiscoveredDB, _ string) DumpResult {
|
||||
dumped++
|
||||
return DumpResult{DB: d}
|
||||
})
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err != nil {
|
||||
t.Fatalf("reconstitute: %v", err)
|
||||
}
|
||||
if dumped != 0 {
|
||||
t.Fatalf("a no-DB app must not produce a safety dump, got %d call(s)", dumped)
|
||||
}
|
||||
if len(*imported) != 0 {
|
||||
t.Fatalf("a no-DB app must not import anything, got %v", *imported)
|
||||
}
|
||||
if res.SafetyDump != "" || res.DBsReplayed != 0 {
|
||||
t.Fatalf("unexpected DB activity: safety=%q replayed=%d", res.SafetyDump, res.DBsReplayed)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconstituteSurfacesLegacySkewedPair is Scenario D: a pre-v0.148 snapshot carries no coherence
|
||||
// stamp, so its two halves may be from different times. That must be SURFACED (and reversible), never
|
||||
// blocked — the customer's own judgement is the gate, and refusing would deny a legitimate restore.
|
||||
func TestReconstituteSurfacesLegacySkewedPair(t *testing.T) {
|
||||
m, _, imported := reconFixture(t, "", "", pgDump(1))
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err != nil {
|
||||
t.Fatalf("a legacy pair must still be restorable, got refusal: %v", err)
|
||||
}
|
||||
if !res.Skewed {
|
||||
t.Fatal("an unstamped (pre-v0.148) snapshot must report Skewed so the confirm can say so")
|
||||
}
|
||||
if len(*imported) != 1 {
|
||||
t.Fatalf("the legacy restore must still replay, got %v", *imported)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconstituteFlagsCustomerEmptyDump is the R-44 sniff at the restore end: the immich dump that
|
||||
// started all of this was structurally valid and contained zero users. Restoring it is allowed, but
|
||||
// the customer must be told before they commit.
|
||||
func TestReconstituteFlagsCustomerEmptyDump(t *testing.T) {
|
||||
// A valid postgres dump whose accounts table has NO rows — the 2026-07-19 shape exactly.
|
||||
m, _, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(0))
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err != nil {
|
||||
t.Fatalf("the sniff must never block a restore: %v", err)
|
||||
}
|
||||
if !res.LooksEmpty {
|
||||
t.Fatal("a dump with an empty accounts table must raise the warn-level signal")
|
||||
}
|
||||
}
|
||||
|
||||
// TestOffsiteScratchPairReportsWhatTheConfirmNeeds covers the page-render surface: the confirm can
|
||||
// only be honest if this reports the pair's age and warnings before anything is started.
|
||||
func TestOffsiteScratchPairReportsWhatTheConfirmNeeds(t *testing.T) {
|
||||
m, _, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z",
|
||||
"-- PostgreSQL database dump\nCREATE TABLE a();\nCOPY public.\"user\" (id) FROM stdin;\n7\n\\.\n")
|
||||
|
||||
info := m.OffsiteScratchPair("immich")
|
||||
if !info.Ready || !info.HasDump {
|
||||
t.Fatalf("expected a ready pair with a dump, got %+v", info)
|
||||
}
|
||||
if info.Skewed {
|
||||
t.Fatal("a stamped snapshot must not be reported as skewed")
|
||||
}
|
||||
if info.LooksEmpty {
|
||||
t.Fatal("a dump with account rows must not be flagged empty")
|
||||
}
|
||||
want, _ := time.Parse(time.RFC3339, "2026-07-19T06:00:00Z")
|
||||
if !info.DumpsAt.Equal(want) {
|
||||
t.Fatalf("DumpsAt = %v, want %v", info.DumpsAt, want)
|
||||
}
|
||||
}
|
||||
|
||||
// --- R-44: the coherence pre-phase -----------------------------------------------------------
|
||||
|
||||
// TestOffsiteRunDumpsBeforeCapture is Scenarios A + B. The ORDER is the entire mechanism: dumps
|
||||
// must be refreshed BEFORE restic captures, so the snapshot pairs this run's database with this
|
||||
// run's files. Reversed, the snapshot would hold rows pointing at files that were never captured.
|
||||
//
|
||||
// It also asserts the ordering on the NIGHTLY entry point (RunOffboxBackup, no progress sink), not
|
||||
// just the manual one — before v0.148.0 the nightly ordering was an accident of two independent
|
||||
// scheduler entries at 02:30 and 04:15, which a schedule edit could silently invert.
|
||||
func TestOffsiteRunDumpsBeforeCapture(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
mkUnit(t, drive, "immich")
|
||||
if err := os.MkdirAll(filepath.Join(drive, "appdata", "immich"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["immich"] = drive
|
||||
prov.has["immich"] = true
|
||||
prov.binds["immich"] = []ClassifiedBind{mandatoryHDD("appdata/immich")}
|
||||
_ = sett.SetAppOffbox("immich", true)
|
||||
|
||||
var order []string
|
||||
m.SetOffsitePreDumpFn(func(context.Context) error {
|
||||
order = append(order, "dump")
|
||||
return nil
|
||||
})
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
switch {
|
||||
case contains(args, "cat") && contains(args, "config"):
|
||||
return []byte(`{"version":2}`), nil
|
||||
case contains(args, "backup"):
|
||||
order = append(order, "capture")
|
||||
return nil, nil
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[]`), nil
|
||||
case contains(args, "stats"):
|
||||
return []byte(`{"total_size":123}`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
if len(order) < 2 {
|
||||
t.Fatalf("expected both a dump and a capture, got %v", order)
|
||||
}
|
||||
if order[0] != "dump" {
|
||||
t.Fatalf("the dump leg MUST precede the capture (R-44); got %v", order)
|
||||
}
|
||||
if order[1] != "capture" {
|
||||
t.Fatalf("expected the capture immediately after the dump, got %v", order)
|
||||
}
|
||||
}
|
||||
|
||||
// TestOffsiteRunContinuesWhenDumpLegFails is the data-first rule: a dump failure degrades the
|
||||
// snapshot's DB half but must NOT abort the push. Refusing to ship the files would turn a partial
|
||||
// backup into no backup at all — strictly worse for the customer.
|
||||
func TestOffsiteRunContinuesWhenDumpLegFails(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
mkUnit(t, drive, "immich")
|
||||
if err := os.MkdirAll(filepath.Join(drive, "appdata", "immich"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["immich"] = drive
|
||||
prov.has["immich"] = true
|
||||
prov.binds["immich"] = []ClassifiedBind{mandatoryHDD("appdata/immich")}
|
||||
_ = sett.SetAppOffbox("immich", true)
|
||||
|
||||
m.SetOffsitePreDumpFn(func(context.Context) error { return context.DeadlineExceeded })
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("a dump failure must not fail the whole run: %v", err)
|
||||
}
|
||||
if cap.backups != 1 {
|
||||
t.Fatalf("the files must still be pushed after a dump failure, got %d capture(s)", cap.backups)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCaptureRecoveryUnitStampsAndCarriesRunID covers the stamp that makes a pair verifiable at
|
||||
// restore time, and the trap beside it: the PERIODIC refresh must neither invent a coherence claim
|
||||
// nor erase one a real run established.
|
||||
func TestCaptureRecoveryUnitStampsAndCarriesRunID(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, base := classifiedOffboxManager(t, drive)
|
||||
base.hdd["immich"] = drive
|
||||
// CaptureRecoveryUnit needs real recovery info + a compose dir to read; the shared fixture
|
||||
// provider returns none, so wrap it rather than widening a struct four other test files use.
|
||||
stackDir := filepath.Join(t.TempDir(), "immich")
|
||||
if err := os.MkdirAll(stackDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(stackDir, "docker-compose.yml"), []byte("services: {}\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetStackProvider(&recoveryProvider{offbox3aProvider: *base, stackDir: stackDir})
|
||||
|
||||
// 1) A run in flight stamps the manifest.
|
||||
end := m.beginOffsiteRunStamp("run-A")
|
||||
if err := m.CaptureRecoveryUnit("immich"); err != nil {
|
||||
t.Fatalf("capture: %v", err)
|
||||
}
|
||||
end()
|
||||
man := readManifest(RecoveryUnitManifestPath(drive, "immich"))
|
||||
if man == nil || man.OffsiteRunID != "run-A" {
|
||||
t.Fatalf("expected the in-flight run id to be stamped, got %+v", man)
|
||||
}
|
||||
if man.DumpsAt == "" {
|
||||
t.Fatal("a stamped unit must record when its dumps were taken")
|
||||
}
|
||||
|
||||
// 2) A periodic refresh (no run in flight) must CARRY the stamp forward, not blank it — a unit
|
||||
// that silently lost its stamp would be re-reported as a skewed legacy pair at restore time.
|
||||
if err := m.CaptureRecoveryUnit("immich"); err != nil {
|
||||
t.Fatalf("refresh: %v", err)
|
||||
}
|
||||
man2 := readManifest(RecoveryUnitManifestPath(drive, "immich"))
|
||||
if man2 == nil || man2.OffsiteRunID != "run-A" {
|
||||
t.Fatalf("the periodic refresh erased the coherence stamp: %+v", man2)
|
||||
}
|
||||
|
||||
// 3) A NEW run re-stamps even though nothing else about the unit changed — the idempotent-skip
|
||||
// must not swallow the one field the restore path reads.
|
||||
end2 := m.beginOffsiteRunStamp("run-B")
|
||||
if err := m.CaptureRecoveryUnit("immich"); err != nil {
|
||||
t.Fatalf("capture 2: %v", err)
|
||||
}
|
||||
end2()
|
||||
man3 := readManifest(RecoveryUnitManifestPath(drive, "immich"))
|
||||
if man3 == nil || man3.OffsiteRunID != "run-B" {
|
||||
t.Fatalf("a new run must re-stamp the unit, got %+v", man3)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// R-200 (controller v0.195.0) — THE DIAGNOSTIC HALF, and only that half.
|
||||
//
|
||||
// The question this answers, once, decisively: **is the offsite repository password actually
|
||||
// recoverable from the hub's sealed bundle?** Everything else in the recovery chain is downstream of
|
||||
// that, and until 2026-08-04 nobody had ever asked it — the round-trip proof on record (2026-06-10)
|
||||
// predates the field by a month, and the extraction step did not exist at all.
|
||||
//
|
||||
// IT COMPARES; IT DOES NOT INSTALL. The recovered password is NOT written to offboxPwPath. Comparing
|
||||
// proves recoverability; installing changes a live box's state on a path nobody has walked, and
|
||||
// "the existing repository opens under a recovered key" is a separate link with a drill around it.
|
||||
// Keep this function free of any write — if a future change makes it install, it stops being a
|
||||
// diagnostic and needs the drill's supervision.
|
||||
//
|
||||
// IT HANDLES ONLY HASHES OUTSIDE THE AGENT CALL. The agent returns the password and its sha256; this
|
||||
// reads the hash. The value is dropped on the floor here deliberately, so no controller-side code
|
||||
// path can grow a habit of holding it.
|
||||
|
||||
// OffsiteKeyRecoverer is the agent-side seam (agent >= v0.125.0,
|
||||
// POST /escrow/recover-offsite-password): it fetches this host's sealed bundle from the hub, unseals
|
||||
// it with R, and returns ONLY the offsite repository password plus its sha256.
|
||||
type OffsiteKeyRecoverer interface {
|
||||
RecoverOffsiteRepoPassword(ctx context.Context, recoveryCode string) (password, sha256hex string, err error)
|
||||
}
|
||||
|
||||
// RecoveryCheckResult is the verdict. It carries HASHES ONLY — there is no field here that could
|
||||
// leak a password into a log, a report or a terminal.
|
||||
type RecoveryCheckResult struct {
|
||||
// LocalSHA256 is the hash of the repo password currently on disk ("" when there is none).
|
||||
LocalSHA256 string
|
||||
// RecoveredSHA256 is the hash of what came out of the sealed bundle.
|
||||
RecoveredSHA256 string
|
||||
// Match is the whole point: byte-identical keys produce identical hashes.
|
||||
Match bool
|
||||
// LocalPresent distinguishes "they differ" from "there was nothing to compare against" — a
|
||||
// rebuilt box with no repo password yet is a legitimate state and must not read as a mismatch.
|
||||
LocalPresent bool
|
||||
}
|
||||
|
||||
// CheckOffsiteKeyRecoverable recovers the repository password through the agent and compares it, by
|
||||
// hash, against the one on this box's disk. It writes nothing anywhere.
|
||||
//
|
||||
// R is passed straight through to the agent and is not retained here. The CALLER owns clearing its
|
||||
// own copy; this function keeps none.
|
||||
func (m *Manager) CheckOffsiteKeyRecoverable(ctx context.Context, rec OffsiteKeyRecoverer, recoveryCode string) (RecoveryCheckResult, error) {
|
||||
var out RecoveryCheckResult
|
||||
if rec == nil {
|
||||
return out, fmt.Errorf("offbox: no agent recovery seam configured")
|
||||
}
|
||||
if strings.TrimSpace(recoveryCode) == "" {
|
||||
return out, fmt.Errorf("offbox: the recovery code is required")
|
||||
}
|
||||
// Read the local side FIRST, so a missing local password is reported as such rather than
|
||||
// surfacing as a mismatch after a successful recovery.
|
||||
localHash, ok := m.OffboxRepoPasswordHash()
|
||||
out.LocalSHA256, out.LocalPresent = localHash, ok
|
||||
|
||||
pw, recoveredHash, err := rec.RecoverOffsiteRepoPassword(ctx, recoveryCode)
|
||||
if err != nil {
|
||||
return out, err // the agent's message already names the step and contains no secret
|
||||
}
|
||||
pw = "" // the VALUE is not this function's business — §8.5, compare, do not install
|
||||
_ = pw
|
||||
out.RecoveredSHA256 = recoveredHash
|
||||
out.Match = ok && recoveredHash != "" && recoveredHash == localHash
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,349 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-200 — the diagnostic half. What is asserted here is the VERDICT and the NON-WRITE, because those
|
||||
// are the two things that make this a proof rather than a change to a live box.
|
||||
|
||||
type fakeRecoverer struct {
|
||||
pw, sha string
|
||||
err error
|
||||
gotCode string
|
||||
callable bool
|
||||
}
|
||||
|
||||
func (f *fakeRecoverer) RecoverOffsiteRepoPassword(_ context.Context, code string) (string, string, error) {
|
||||
f.callable = true
|
||||
f.gotCode = code
|
||||
return f.pw, f.sha, f.err
|
||||
}
|
||||
|
||||
// Scenario A at this layer — the recovered key's hash is compared against the on-disk one and the
|
||||
// verdict is the equality, not "no error".
|
||||
func TestCheckOffsiteKeyRecoverable_MatchAndMismatch(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
localHash, ok := m.OffboxRepoPasswordHash()
|
||||
if !ok {
|
||||
t.Fatal("precondition: no local repo password")
|
||||
}
|
||||
|
||||
// The key came back identical.
|
||||
res, err := m.CheckOffsiteKeyRecoverable(context.Background(), &fakeRecoverer{pw: "irrelevant", sha: localHash}, "R")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !res.Match || !res.LocalPresent || res.RecoveredSHA256 != localHash || res.LocalSHA256 != localHash {
|
||||
t.Fatalf("identical keys must report Match: %+v", res)
|
||||
}
|
||||
|
||||
// A DIFFERENT key must report a mismatch, not an error — "it worked and disagreed" is a finding
|
||||
// about the system and must be distinguishable from "a step failed".
|
||||
res, err = m.CheckOffsiteKeyRecoverable(context.Background(), &fakeRecoverer{pw: "x", sha: "0000000000000000000000000000000000000000000000000000000000000000"}, "R")
|
||||
if err != nil {
|
||||
t.Fatalf("a mismatch is a verdict, not an error: %v", err)
|
||||
}
|
||||
if res.Match {
|
||||
t.Fatal("a different recovered key must NOT report Match")
|
||||
}
|
||||
}
|
||||
|
||||
// A box with no local password reports that distinctly — it is the rebuilt-box shape, where the next
|
||||
// step is to install rather than to compare, and reading it as a mismatch would be wrong.
|
||||
func TestCheckOffsiteKeyRecoverable_NoLocalPassword(t *testing.T) {
|
||||
m := newBareManager(t)
|
||||
res, err := m.CheckOffsiteKeyRecoverable(context.Background(), &fakeRecoverer{pw: "x", sha: "abc"}, "R")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if res.LocalPresent || res.Match {
|
||||
t.Fatalf("no local key must report LocalPresent=false and Match=false: %+v", res)
|
||||
}
|
||||
if res.RecoveredSHA256 != "abc" {
|
||||
t.Fatalf("the recovery itself succeeded and must be reported: %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// §8.5 — THE CHECK MUST NOT INSTALL. This is the assertion that keeps a diagnostic a diagnostic.
|
||||
// RED-PROOF: add `m.InjectOffboxPassword(pw, true)` to CheckOffsiteKeyRecoverable → the on-disk
|
||||
// password changes → this FAILS.
|
||||
func TestCheckOffsiteKeyRecoverable_WritesNothing(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
before, err := os.ReadFile(m.offboxPwPath())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
dir := m.offboxDir()
|
||||
beforeEntries, _ := os.ReadDir(dir)
|
||||
|
||||
recovered := "ffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffff"
|
||||
if _, err := m.CheckOffsiteKeyRecoverable(context.Background(), &fakeRecoverer{pw: recovered, sha: HashResticPassword(recovered)}, "R"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
after, err := os.ReadFile(m.offboxPwPath())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(before, after) {
|
||||
t.Fatal("the check INSTALLED the recovered password — it must compare and never write (§8.5); " +
|
||||
"installing changes a live box on a path nobody has walked")
|
||||
}
|
||||
afterEntries, _ := os.ReadDir(dir)
|
||||
if len(afterEntries) != len(beforeEntries) {
|
||||
var names []string
|
||||
for _, e := range afterEntries {
|
||||
names = append(names, e.Name())
|
||||
}
|
||||
t.Fatalf("the check created files in the offbox dir: %v", names)
|
||||
}
|
||||
// And nothing leaked into the data dir either.
|
||||
_ = filepath.Walk(m.cfg.Paths.DataDir, func(p string, info os.FileInfo, werr error) error {
|
||||
if werr != nil || info == nil || info.IsDir() {
|
||||
return nil
|
||||
}
|
||||
body, rerr := os.ReadFile(p)
|
||||
if rerr == nil && strings.Contains(string(body), recovered) {
|
||||
t.Errorf("the recovered password was written to %s", p)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
}
|
||||
|
||||
// R goes to the agent verbatim and is not mangled or retained by this layer.
|
||||
func TestCheckOffsiteKeyRecoverable_PassesRThrough(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
const code = "correct horse battery staple sedative anaconda wobbly kingdom placard yodel"
|
||||
f := &fakeRecoverer{pw: "x", sha: "abc"}
|
||||
if _, err := m.CheckOffsiteKeyRecoverable(context.Background(), f, code); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if f.gotCode != code {
|
||||
t.Fatalf("the recovery code reached the agent as %q — a 10-word code must not be re-split or trimmed internally", f.gotCode)
|
||||
}
|
||||
}
|
||||
|
||||
// An agent-side failure surfaces as an error, and the verdict is NOT reported as a mismatch.
|
||||
func TestCheckOffsiteKeyRecoverable_AgentFailure(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
_, err := m.CheckOffsiteKeyRecoverable(context.Background(), &fakeRecoverer{err: errors.New("the recovery code did not open the sealed bundle")}, "R")
|
||||
if err == nil {
|
||||
t.Fatal("an agent failure must be an error, never a silent mismatch")
|
||||
}
|
||||
}
|
||||
|
||||
// The CLI's exit codes are load-bearing: 0 match, 2 clean mismatch, 1 a step failed. "It failed" and
|
||||
// "it worked and disagreed" must never share a status, because only one of them is a finding.
|
||||
func TestRunRecoveryCheck_ExitCodes(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
localHash, _ := m.OffboxRepoPasswordHash()
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
rec OffsiteKeyRecoverer
|
||||
in string
|
||||
want int
|
||||
}{
|
||||
{"match", &fakeRecoverer{pw: "x", sha: localHash}, "some recovery code\n", 0},
|
||||
{"mismatch", &fakeRecoverer{pw: "x", sha: "0000000000000000000000000000000000000000000000000000000000000000"}, "some recovery code\n", 2},
|
||||
{"agent failure", &fakeRecoverer{err: errors.New("wrong code")}, "some recovery code\n", 1},
|
||||
{"no code on stdin", &fakeRecoverer{pw: "x", sha: localHash}, "", 1},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
var out, errb bytes.Buffer
|
||||
got := RunRecoveryCheck(RecoveryCheckDeps{
|
||||
Manager: m, Recoverer: tc.rec, In: strings.NewReader(tc.in), Out: &out, Err: &errb,
|
||||
})
|
||||
if got != tc.want {
|
||||
t.Fatalf("exit = %d, want %d (out=%q err=%q)", got, tc.want, out.String(), errb.String())
|
||||
}
|
||||
// No printed stream may ever carry a password or a recovery code.
|
||||
combined := out.String() + errb.String()
|
||||
for _, secret := range []string{"PRIVATE-KEY-MATERIAL", "some recovery code"} {
|
||||
if strings.Contains(combined, secret) {
|
||||
t.Errorf("the diagnostic printed a secret (%s): %s", secret, combined)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// newBareManager is an offbox manager with a data dir and NO repo password — the freshly rebuilt-box
|
||||
// shape, which the no-local-password case needs and newOffboxManager deliberately does not produce.
|
||||
func newBareManager(t *testing.T) *Manager {
|
||||
t.Helper()
|
||||
logger := log.New(os.Stderr, "", 0)
|
||||
dataDir := t.TempDir()
|
||||
sett, err := settings.Load(filepath.Join(dataDir, "settings.json"), logger)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.DataDir = dataDir
|
||||
cfg.Paths.SystemDataPath = filepath.Join(dataDir, "sys")
|
||||
return NewManager(cfg, sett, logger)
|
||||
}
|
||||
|
||||
// R-200 Part 0 — the INSTALL sibling. What is asserted is the three outcomes, the confirmation gate,
|
||||
// and that R does not survive either path.
|
||||
|
||||
// An install on a box with NO local password writes it — the rebuilt-box shape, which is the only
|
||||
// situation this command exists for.
|
||||
// RED-PROOF: drop the `confirm` check so an unconfirmed run installs → the dry-run case below FAILS.
|
||||
func TestRecoverAndInstall_InstallsOnABareBox(t *testing.T) {
|
||||
m := newBareManager(t)
|
||||
pw := "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
rec := &fakeRecoverer{pw: pw, sha: HashResticPassword(pw)}
|
||||
|
||||
// 1) DRY RUN — prints the hashes, writes nothing.
|
||||
var out, errb bytes.Buffer
|
||||
if got := RecoverAndInstall(RecoveryCheckDeps{Manager: m, Recoverer: rec, In: strings.NewReader("code\n"), Out: &out, Err: &errb}, false); got != 0 {
|
||||
t.Fatalf("dry run exit = %d, want 0 (%s / %s)", got, out.String(), errb.String())
|
||||
}
|
||||
if _, present := m.OffboxRepoPasswordHash(); present {
|
||||
t.Fatal("the DRY RUN wrote the password — the confirmation gate does not hold, which is the " +
|
||||
"whole reason the operator gets to see the hashes before anything exists to undo")
|
||||
}
|
||||
if !strings.Contains(out.String(), "DRY RUN") {
|
||||
t.Errorf("the dry run must say so, got %q", out.String())
|
||||
}
|
||||
|
||||
// 2) CONFIRMED — writes it, and it reads back identical.
|
||||
out.Reset()
|
||||
errb.Reset()
|
||||
if got := RecoverAndInstall(RecoveryCheckDeps{Manager: m, Recoverer: rec, In: strings.NewReader("code\n"), Out: &out, Err: &errb}, true); got != 0 {
|
||||
t.Fatalf("confirmed exit = %d, want 0 (%s / %s)", got, out.String(), errb.String())
|
||||
}
|
||||
got, present := m.OffboxRepoPasswordHash()
|
||||
if !present || got != HashResticPassword(pw) {
|
||||
t.Fatalf("the recovered password was not placed (present=%v hash=%q)", present, got)
|
||||
}
|
||||
if !strings.Contains(out.String(), "INSTALLED") {
|
||||
t.Errorf("a successful install must say so, got %q", out.String())
|
||||
}
|
||||
// The VALUE must not have been printed on either stream.
|
||||
if strings.Contains(out.String()+errb.String(), pw) {
|
||||
t.Fatal("the repository password was printed")
|
||||
}
|
||||
}
|
||||
|
||||
// An identical key already present is "unchanged", not "installed" and not an error — and nothing is
|
||||
// written, so a re-run is harmless.
|
||||
func TestRecoverAndInstall_UnchangedWhenIdentical(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
localHash, _ := m.OffboxRepoPasswordHash()
|
||||
before, err := os.ReadFile(m.offboxPwPath())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var out, errb bytes.Buffer
|
||||
got := RecoverAndInstall(RecoveryCheckDeps{
|
||||
Manager: m, Recoverer: &fakeRecoverer{pw: "x", sha: localHash},
|
||||
In: strings.NewReader("code\n"), Out: &out, Err: &errb,
|
||||
}, true)
|
||||
if got != 0 {
|
||||
t.Fatalf("exit = %d, want 0", got)
|
||||
}
|
||||
if !strings.Contains(out.String(), "UNCHANGED") {
|
||||
t.Errorf("an identical key must report UNCHANGED, got %q", out.String())
|
||||
}
|
||||
after, _ := os.ReadFile(m.offboxPwPath())
|
||||
if !bytes.Equal(before, after) {
|
||||
t.Fatal("an UNCHANGED outcome rewrote the file")
|
||||
}
|
||||
}
|
||||
|
||||
// A DIFFERENT key already present is REFUSED — installing would clobber the key the box's current
|
||||
// repository is encrypted under, and which history to keep is not this command's decision.
|
||||
func TestRecoverAndInstall_RefusesToClobberADifferentKey(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
before, err := os.ReadFile(m.offboxPwPath())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var out, errb bytes.Buffer
|
||||
got := RecoverAndInstall(RecoveryCheckDeps{
|
||||
Manager: m,
|
||||
Recoverer: &fakeRecoverer{pw: "y", sha: "0000000000000000000000000000000000000000000000000000000000000000"},
|
||||
In: strings.NewReader("code\n"), Out: &out, Err: &errb,
|
||||
}, true)
|
||||
if got != 2 {
|
||||
t.Fatalf("exit = %d, want 2 (a refusal is its own outcome, not a generic failure)", got)
|
||||
}
|
||||
if !strings.Contains(errb.String(), "REFUSED") {
|
||||
t.Errorf("the refusal must say so, got %q", errb.String())
|
||||
}
|
||||
after, _ := os.ReadFile(m.offboxPwPath())
|
||||
if !bytes.Equal(before, after) {
|
||||
t.Fatal("a REFUSED install clobbered the existing key — the exact outcome the refusal exists to prevent")
|
||||
}
|
||||
}
|
||||
|
||||
// R must not survive either path, and the recovery code must never be printed.
|
||||
func TestRecoverAndInstall_RLeavesNoTrace(t *testing.T) {
|
||||
const code = "correct horse battery staple sedative anaconda wobbly kingdom placard yodel"
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
confirm bool
|
||||
}{{"dry run", false}, {"confirmed", true}} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
m := newBareManager(t)
|
||||
pw := "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
var out, errb bytes.Buffer
|
||||
RecoverAndInstall(RecoveryCheckDeps{
|
||||
Manager: m, Recoverer: &fakeRecoverer{pw: pw, sha: HashResticPassword(pw)},
|
||||
In: strings.NewReader(code + "\n"), Out: &out, Err: &errb,
|
||||
}, tc.confirm)
|
||||
combined := out.String() + errb.String()
|
||||
if strings.Contains(combined, code) {
|
||||
t.Errorf("the recovery code was printed: %s", combined)
|
||||
}
|
||||
if strings.Contains(combined, pw) {
|
||||
t.Errorf("the repository password was printed: %s", combined)
|
||||
}
|
||||
// POSITIVE CONTROL for the sweep below: plant R in the data dir, prove the walk finds it,
|
||||
// remove it. An absence check is worth only what its sensitivity is.
|
||||
ctrl := filepath.Join(m.cfg.Paths.DataDir, ".planted-control")
|
||||
if err := os.WriteFile(ctrl, []byte(code), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if n := countFilesContaining(t, m.cfg.Paths.DataDir, code); n != 1 {
|
||||
t.Fatalf("positive control: the sweep found %d planted copies, want 1 — the sweep is not sensitive", n)
|
||||
}
|
||||
if err := os.Remove(ctrl); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if n := countFilesContaining(t, m.cfg.Paths.DataDir, code); n != 0 {
|
||||
t.Fatalf("the recovery code survived in %d file(s) under the data dir", n)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func countFilesContaining(t *testing.T, root, needle string) int {
|
||||
t.Helper()
|
||||
n := 0
|
||||
_ = filepath.Walk(root, func(p string, info os.FileInfo, err error) error {
|
||||
if err != nil || info == nil || info.IsDir() {
|
||||
return nil
|
||||
}
|
||||
body, rerr := os.ReadFile(p)
|
||||
if rerr == nil && strings.Contains(string(body), needle) {
|
||||
n++
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return n
|
||||
}
|
||||
@@ -0,0 +1,296 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-200 — the operator-facing entry point for the recovery check, and the ONLY one this session
|
||||
// ships. Deliberately a `docker exec` escape hatch in the shape of `--print-reset-code`, not a page,
|
||||
// a card or an API a browser can reach: the customer-facing flow is designed on top of a chain that
|
||||
// has been walked, and this is the walk.
|
||||
//
|
||||
// WHY R COMES FROM STDIN AND NOT A FLAG. A flag value is visible in `ps`, in the shell history, in a
|
||||
// container's command line and in any transcript of the session that ran it. R is the one secret in
|
||||
// this system that cannot be rotated, re-issued or recovered. It is read from stdin, held in one
|
||||
// string, and cleared before the function returns — on the success path and on every failure path.
|
||||
//
|
||||
// docker exec -i felhom-controller /app/felhom-controller --recover-offsite-check < /root/r.txt
|
||||
//
|
||||
// WHAT IT PRINTS: two sha256 hashes and a verdict. Never a password, never R, never a blob. The
|
||||
// hashes are of 256-bit random secrets and are non-reversible — the same value the hub already stores
|
||||
// and serves in report ACKs.
|
||||
|
||||
// RecoveryCheckDeps is what the CLI needs; injected so the entry point is testable without a live
|
||||
// agent, a live hub or real crypto.
|
||||
type RecoveryCheckDeps struct {
|
||||
// Manager owns the on-disk repo password hash.
|
||||
Manager *Manager
|
||||
// Recoverer is the agent seam (agentapi.Client satisfies it).
|
||||
Recoverer OffsiteKeyRecoverer
|
||||
// In is where R is read from (os.Stdin in production).
|
||||
In io.Reader
|
||||
// Out / Err are the report streams (os.Stdout / os.Stderr in production).
|
||||
Out, Err io.Writer
|
||||
// Timeout bounds the whole check. 0 → 90s (an unseal shells out to age and a fetch crosses the WAN).
|
||||
Timeout time.Duration
|
||||
}
|
||||
|
||||
// RunRecoveryCheck reads R from stdin, recovers the offsite repository password through the agent,
|
||||
// and reports whether it matches the one on disk — BY HASH. Returns a process exit code:
|
||||
//
|
||||
// 0 = the hashes matched (the key is recoverable)
|
||||
// 1 = a step failed (fetch, unseal, or no local password to compare against)
|
||||
// 2 = the check ran cleanly and the hashes DIFFER — the loud case, and the one that would mean the
|
||||
// sealed bundle does not carry what four weeks of documents say it carries
|
||||
//
|
||||
// A distinct code for the mismatch on purpose: "it failed" and "it worked and disagreed" must never
|
||||
// share an exit status, because only one of them is a finding about the system rather than about the
|
||||
// run.
|
||||
func RunRecoveryCheck(d RecoveryCheckDeps) int {
|
||||
out, errw := d.Out, d.Err
|
||||
if out == nil {
|
||||
out = os.Stdout
|
||||
}
|
||||
if errw == nil {
|
||||
errw = os.Stderr
|
||||
}
|
||||
if d.Manager == nil || d.Recoverer == nil {
|
||||
fmt.Fprintln(errw, "recover-offsite-check: not configured (no backup manager or no agent channel)")
|
||||
return 1
|
||||
}
|
||||
in := d.In
|
||||
if in == nil {
|
||||
in = os.Stdin
|
||||
}
|
||||
// Read R: the first line of stdin, trimmed. A 10-word EFF code contains spaces, so only the
|
||||
// line ending is stripped — never internal whitespace.
|
||||
br := bufio.NewReader(io.LimitReader(in, 4096))
|
||||
line, rerr := br.ReadString('\n')
|
||||
R := strings.TrimRight(line, "\r\n")
|
||||
if R == "" {
|
||||
fmt.Fprintln(errw, "recover-offsite-check: no recovery code on stdin. Pipe it in:")
|
||||
fmt.Fprintln(errw, " docker exec -i felhom-controller /app/felhom-controller --recover-offsite-check < /path/to/code")
|
||||
if rerr != nil && rerr != io.EOF {
|
||||
fmt.Fprintf(errw, " (read error: %v)\n", rerr)
|
||||
}
|
||||
return 1
|
||||
}
|
||||
|
||||
timeout := d.Timeout
|
||||
if timeout == 0 {
|
||||
timeout = 90 * time.Second
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
|
||||
fmt.Fprintln(out, "=== offsite key recovery check (R-200) — compares, never installs ===")
|
||||
res, err := d.Manager.CheckOffsiteKeyRecoverable(ctx, d.Recoverer, R)
|
||||
R = "" // cleared before anything else, on every path below
|
||||
if err != nil {
|
||||
fmt.Fprintf(errw, " [FAIL] %v\n", err) // the agent's message names the step; it carries no secret
|
||||
fmt.Fprintln(errw, " nothing was written.")
|
||||
return 1
|
||||
}
|
||||
if !res.LocalPresent {
|
||||
fmt.Fprintln(errw, " [FAIL] there is no repository password on this box to compare against")
|
||||
fmt.Fprintf(out, " recovered sha256: %s\n", res.RecoveredSHA256)
|
||||
fmt.Fprintln(errw, " (the recovery itself SUCCEEDED — this box simply has no local key. That is the")
|
||||
fmt.Fprintln(errw, " rebuilt-box shape, where the next step is to INSTALL rather than compare.)")
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(out, " on-disk sha256: %s\n", res.LocalSHA256)
|
||||
fmt.Fprintf(out, " recovered sha256: %s\n", res.RecoveredSHA256)
|
||||
if !res.Match {
|
||||
fmt.Fprintln(errw, " [MISMATCH] the recovered key is NOT the key this box uses.")
|
||||
fmt.Fprintln(errw, " This is a finding about the system, not about the run: the sealed bundle does not")
|
||||
fmt.Fprintln(errw, " carry the repository password this box's off-site history is encrypted under.")
|
||||
return 2
|
||||
}
|
||||
fmt.Fprintln(out, " [MATCH] the offsite repository password IS recoverable from the sealed escrow.")
|
||||
fmt.Fprintln(out, " Nothing was written: this check compares and never installs.")
|
||||
return 0
|
||||
}
|
||||
|
||||
// RecoverAndInstall is the sibling of RunRecoveryCheck that PLACES the recovered repository password,
|
||||
// so a rebuilt box can reopen the off-site history it inherited (R-200's remaining plumbing half).
|
||||
//
|
||||
// WHY THIS IS CODE AND NOT A MANUAL STEP. The alternative — recover the password, read it off a
|
||||
// terminal, and paste it into the injection endpoint by hand — puts the offsite DATA key through a
|
||||
// human's screen, clipboard and shell history. Doing it in-process is both simpler and strictly
|
||||
// safer: the value goes agent → this process → the 0600 file and is never rendered anywhere.
|
||||
//
|
||||
// THE CONFIRMATION IS A SEPARATE INVOCATION, ON PURPOSE. Without `confirm` this prints the two hashes
|
||||
// and writes nothing — the operator sees the comparison BEFORE any write exists as a possibility.
|
||||
// With `confirm` it prints the same hashes and then installs. A single interactive prompt would have
|
||||
// had to share stdin with R, which is where R must not be competing for attention.
|
||||
//
|
||||
// THREE OUTCOMES, NAMED DISTINCTLY, because "it did nothing" and "it refused" are different facts:
|
||||
//
|
||||
// installed — this box had NO repository password (the rebuilt-box shape). The recovered one is placed.
|
||||
// unchanged — a password is present and is byte-identical to the recovered one. Nothing is written.
|
||||
// refused — a password is present and DIFFERS. Installing would clobber the key this box's CURRENT
|
||||
// repository is encrypted under, so it is refused. No force option is offered here: that
|
||||
// decision needs a human who knows which history they intend to keep.
|
||||
// RecoverInstallOutcome names the terminal states of a recovery+install. Distinct values because
|
||||
// "it did nothing", "it refused" and "it installed" are different facts and a caller — CLI or web —
|
||||
// must be able to say which happened without parsing prose.
|
||||
type RecoverInstallOutcome string
|
||||
|
||||
const (
|
||||
// RecoverInstalled — the box had NO repository password; the recovered one is now in place.
|
||||
RecoverInstalled RecoverInstallOutcome = "installed"
|
||||
// RecoverUnchanged — a password was present and is byte-identical to the recovered one.
|
||||
RecoverUnchanged RecoverInstallOutcome = "unchanged"
|
||||
// RecoverRefused — a DIFFERENT password is present; installing would clobber the key the box's
|
||||
// current repository is encrypted under.
|
||||
RecoverRefused RecoverInstallOutcome = "refused"
|
||||
// RecoverDryRun — nothing was written because confirm was false.
|
||||
RecoverDryRun RecoverInstallOutcome = "dry_run"
|
||||
)
|
||||
|
||||
// RecoverInstallResult is the non-secret outcome of a recovery. It carries HASHES ONLY — never the
|
||||
// password, never R. The hashes are of 256-bit random secrets, non-reversible, and are the same
|
||||
// values the hub already stores and serves in report ACKs.
|
||||
type RecoverInstallResult struct {
|
||||
Outcome RecoverInstallOutcome
|
||||
LocalPresent bool
|
||||
LocalSHA256 string
|
||||
RecoveredSHA256 string
|
||||
}
|
||||
|
||||
// RecoverInstallCore is THE recovery+install path in this codebase — fetch the sealed bundle through
|
||||
// the agent, unseal it with R, compare against what is on disk, and place it when that is the right
|
||||
// thing to do.
|
||||
//
|
||||
// ONE FUNCTION, TWO CALLERS (R-193). The CLI (`--recover-offsite-install`) and the customer's recovery
|
||||
// page both call this. They must not each carry a copy: two implementations of the one operation that
|
||||
// can permanently lose a customer's data would drift, and only one of them would ever be tested.
|
||||
// `RecoverAndInstall` below is a thin wrapper that maps this result onto the CLI's exit codes and
|
||||
// printed lines; the web handler maps it onto Hungarian copy. Neither contains recovery logic.
|
||||
//
|
||||
// R IS THE CALLER'S TO CLEAR. This function does not retain it: it is passed to the agent seam and
|
||||
// never stored, logged or returned. The password recovered from the bundle IS cleared here, on every
|
||||
// path, before returning — it never leaves this function in any form.
|
||||
//
|
||||
// The three outcomes and their reasoning are unchanged from the CLI's original implementation; see
|
||||
// RecoverAndInstall's header, which remains the authority on WHY a differing local password is
|
||||
// refused rather than forced.
|
||||
func RecoverInstallCore(ctx context.Context, m *Manager, rec OffsiteKeyRecoverer, R string, confirm bool) (RecoverInstallResult, error) {
|
||||
var res RecoverInstallResult
|
||||
if m == nil || rec == nil {
|
||||
return res, fmt.Errorf("recovery not configured (no backup manager or no agent channel)")
|
||||
}
|
||||
pw, recoveredHash, err := rec.RecoverOffsiteRepoPassword(ctx, R)
|
||||
if err != nil {
|
||||
return res, err // the agent's message names the step; it carries no secret
|
||||
}
|
||||
res.RecoveredSHA256 = recoveredHash
|
||||
res.LocalSHA256, res.LocalPresent = m.OffboxRepoPasswordHash()
|
||||
|
||||
switch {
|
||||
case res.LocalPresent && res.LocalSHA256 == recoveredHash:
|
||||
pw = ""
|
||||
res.Outcome = RecoverUnchanged
|
||||
return res, nil
|
||||
case res.LocalPresent:
|
||||
pw = ""
|
||||
res.Outcome = RecoverRefused
|
||||
return res, nil
|
||||
}
|
||||
if !confirm {
|
||||
pw = ""
|
||||
res.Outcome = RecoverDryRun
|
||||
return res, nil
|
||||
}
|
||||
if err := m.InjectOffboxPassword(pw, false); err != nil {
|
||||
pw = ""
|
||||
return res, fmt.Errorf("placing the recovered password: %w", err)
|
||||
}
|
||||
pw = ""
|
||||
// Re-read from disk rather than trusting what we just wrote — the observable is the file's state.
|
||||
afterHash, ok := m.OffboxRepoPasswordHash()
|
||||
if !ok || afterHash != recoveredHash {
|
||||
return res, fmt.Errorf("the password was written but does not read back as expected (on-disk %q)", afterHash)
|
||||
}
|
||||
res.Outcome = RecoverInstalled
|
||||
return res, nil
|
||||
}
|
||||
|
||||
func RecoverAndInstall(d RecoveryCheckDeps, confirm bool) int {
|
||||
out, errw := d.Out, d.Err
|
||||
if out == nil {
|
||||
out = os.Stdout
|
||||
}
|
||||
if errw == nil {
|
||||
errw = os.Stderr
|
||||
}
|
||||
if d.Manager == nil || d.Recoverer == nil {
|
||||
fmt.Fprintln(errw, "recover-offsite-install: not configured (no backup manager or no agent channel)")
|
||||
return 1
|
||||
}
|
||||
in := d.In
|
||||
if in == nil {
|
||||
in = os.Stdin
|
||||
}
|
||||
br := bufio.NewReader(io.LimitReader(in, 4096))
|
||||
line, rerr := br.ReadString('\n')
|
||||
R := strings.TrimRight(line, "\r\n")
|
||||
if R == "" {
|
||||
fmt.Fprintln(errw, "recover-offsite-install: no recovery code on stdin. Pipe it in:")
|
||||
fmt.Fprintln(errw, " docker exec -i felhom-controller /usr/local/bin/felhom-controller --recover-offsite-install [--confirm-install] < /path/to/code")
|
||||
if rerr != nil && rerr != io.EOF {
|
||||
fmt.Fprintf(errw, " (read error: %v)\n", rerr)
|
||||
}
|
||||
return 1
|
||||
}
|
||||
timeout := d.Timeout
|
||||
if timeout == 0 {
|
||||
timeout = 90 * time.Second
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
|
||||
fmt.Fprintln(out, "=== offsite key recovery INSTALL (R-200) ===")
|
||||
// THE RECOVERY ITSELF IS RecoverInstallCore — the same function the customer's recovery page
|
||||
// drives (R-193). This wrapper adds the CLI's stdin handling, its printed lines and its exit
|
||||
// codes, and NOTHING else; there is exactly one fetch→unseal→compare→install path in this
|
||||
// codebase and no chance of the two callers drifting. Pinned by
|
||||
// TestRecoverAndInstall_DrivesTheSharedCore and by the AST wiring test.
|
||||
res, err := RecoverInstallCore(ctx, d.Manager, d.Recoverer, R, confirm)
|
||||
R = "" // cleared immediately, on every path below
|
||||
if err != nil {
|
||||
fmt.Fprintf(errw, " [FAIL] %v\n", err)
|
||||
fmt.Fprintln(errw, " nothing was written.")
|
||||
return 1
|
||||
}
|
||||
if res.LocalPresent {
|
||||
fmt.Fprintf(out, " on-disk sha256: %s\n", res.LocalSHA256)
|
||||
} else {
|
||||
fmt.Fprintln(out, " on-disk sha256: (none — this box has no repository password)")
|
||||
}
|
||||
fmt.Fprintf(out, " recovered sha256: %s\n", res.RecoveredSHA256)
|
||||
|
||||
switch res.Outcome {
|
||||
case RecoverUnchanged:
|
||||
fmt.Fprintln(out, " [UNCHANGED] the box already holds exactly this key. Nothing written.")
|
||||
return 0
|
||||
case RecoverRefused:
|
||||
fmt.Fprintln(errw, " [REFUSED] a DIFFERENT repository password is already present.")
|
||||
fmt.Fprintln(errw, " Installing would clobber the key this box's current repository is encrypted under,")
|
||||
fmt.Fprintln(errw, " and which history to keep is not a decision this command may take. Nothing written.")
|
||||
return 2
|
||||
case RecoverDryRun:
|
||||
fmt.Fprintln(out, " [DRY RUN] nothing written. The recovered key is ready to install.")
|
||||
fmt.Fprintln(out, " Re-run with --confirm-install to place it.")
|
||||
return 0
|
||||
}
|
||||
fmt.Fprintln(out, " [INSTALLED] the recovered repository password is in place and reads back identical.")
|
||||
fmt.Fprintln(out, " Re-apply the offsite target and run a backup: the existing repository should open.")
|
||||
return 0
|
||||
}
|
||||
@@ -30,6 +30,16 @@ const (
|
||||
// SetOffboxFreeFn overrides the restore free-space probe (tests; the Windows go-test host has no df).
|
||||
func (m *Manager) SetOffboxFreeFn(fn func(path string) int64) { m.offboxFreeFn = fn }
|
||||
|
||||
// SetOffboxFullPlaceCopier overrides the FULL-restore overwrite copier (tests; no rsync needed).
|
||||
func (m *Manager) SetOffboxFullPlaceCopier(fn func(src, dst string) (int, error)) {
|
||||
m.offboxFullPlaceCopier = fn
|
||||
}
|
||||
|
||||
// SetSafetyDumpFn overrides the pre-restore safety dump (tests; no Docker needed).
|
||||
func (m *Manager) SetSafetyDumpFn(fn func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult) {
|
||||
m.safetyDumpFn = fn
|
||||
}
|
||||
|
||||
// offboxFree returns the free-space probe (nil seam → the real diskFreeBytes).
|
||||
func (m *Manager) offboxFree() func(string) int64 {
|
||||
if m.offboxFreeFn != nil {
|
||||
@@ -128,9 +138,10 @@ func (m *Manager) offboxSnapshotSize(ctx context.Context, id string) (int64, err
|
||||
// probe). NEVER cfg.Paths.DataDir (the rootfs — the F-A1 filler). App's HDD drive first; else the first
|
||||
// schedulable storage path; else a Hungarian refusal.
|
||||
func (m *Manager) offboxRestoreScratchDir(stack string) (scratch, nsRoot string, err error) {
|
||||
// offsiteRestoreRootFor is THE place `backups/offsite-restore` is spelled (offbox_verify_copies.go)
|
||||
// — the listing/delete surface must resolve byte-identical paths to the ones written here.
|
||||
scratchFor := func(root string) (string, string) {
|
||||
nr := m.namespaceRoot(root)
|
||||
return filepath.Join(nr, "backups", "offsite-restore", stack), nr
|
||||
return filepath.Join(m.offsiteRestoreRootFor(root), stack), m.namespaceRoot(root)
|
||||
}
|
||||
isNet := func(path string) bool { return m.settings != nil && m.settings.IsNetworkStoragePath(path) }
|
||||
// (1) the app's own drive — preferred, but ONLY if it is not NETWORK storage (F-3afix-1). restic
|
||||
@@ -160,7 +171,36 @@ func (m *Manager) offboxRestoreScratchDir(stack string) (scratch, nsRoot string,
|
||||
}
|
||||
}
|
||||
}
|
||||
return "", "", fmt.Errorf("nincs elérhető adatmeghajtó a visszaállításhoz")
|
||||
// R-252: name the reason AND the way to act on it. This refusal is what a rebuilt box hits — the
|
||||
// drives are physically fine and still mounted, it is their REGISTRATION that the destroyed guest
|
||||
// took with it — and until v0.207.0 it said only that a drive was missing, which reads like data
|
||||
// loss and offers nothing to do.
|
||||
return "", "", fmt.Errorf("nincs regisztrált adatmeghajtó, ezért nincs hová visszaállítani — " +
|
||||
"a meghajtók megvannak, csak újra kell csatolni őket a Tárhely → Meghajtók oldalon, utána " +
|
||||
"ez a visszaállítás működni fog")
|
||||
}
|
||||
|
||||
// HasRestoreDestination reports whether an offsite restore has anywhere on this box to write.
|
||||
//
|
||||
// R-252: the restore PAGE asks this question through the same helper the resolver answers it with,
|
||||
// so the notice cannot appear on a box that would restore fine (Scenario E) nor stay hidden on one
|
||||
// that would refuse. A second copy of the predicate is exactly how a page ends up promising what the
|
||||
// handler then refuses — which is the neighbouring defect, R-253.
|
||||
//
|
||||
// It mirrors the resolver's BOX-level branches (2) and (3) — the schedulable storage paths. Branch
|
||||
// (1), the app's own HDD path, is deliberately not consulted: an installed app's HDD path IS a
|
||||
// registered storage path, so the two cannot disagree in practice, and where they could, erring
|
||||
// toward showing the notice is erring toward telling the customer something true.
|
||||
func (m *Manager) HasRestoreDestination() bool {
|
||||
if m.settings == nil {
|
||||
return false
|
||||
}
|
||||
for _, sp := range m.settings.GetSchedulableStoragePaths() {
|
||||
if strings.TrimSpace(sp.Path) != "" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// RestoreOffboxScratch restores an app's latest offsite snapshot to an on-data-drive scratch dir
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// Offsite shares leg — R-7b Part 3, the remote leg of Model B′.
|
||||
//
|
||||
// It is a SIBLING of the per-app loop in runOffboxInternal, not a modification of it. The B′
|
||||
// invariant — every per-app restic invocation stays byte-identical — is the headline guarantee here
|
||||
// and is enforced by TestOffboxSharesLegLeavesAppCallsByteIdentical.
|
||||
//
|
||||
// Shape: ONE additional `restic backup` call tagged [felhom-offbox, _shares], whose paths are the
|
||||
// payload staging dir plus every MANDATORY (Felhőmentés-on) share folder. It reuses resticStep (so
|
||||
// it inherits the C2 crash-lock self-heal), the caller's already-ensured repo, the caller's
|
||||
// already-taken single-flight, and the SAME enlargement-gate arithmetic the per-app path uses. It
|
||||
// runs BEFORE retention, so `forget --group-by host,tags` covers the `_shares` group for free with
|
||||
// no flag change.
|
||||
//
|
||||
// Degradation contract: when the quota gate trips, the push degrades to the MANIFEST ONLY — never to
|
||||
// nothing. Definitions protection must not regress just because the files no longer fit; a customer
|
||||
// who is over quota should still get their „Megosztás" page back from a DR restore.
|
||||
|
||||
// sharesLegResult carries the outcome of the offsite shares leg back to the run.
|
||||
type sharesLegResult struct {
|
||||
ran bool // the leg produced a restic call
|
||||
count int // share folders included (0 = definitions-only push)
|
||||
blocked bool // the quota gate degraded this push to manifest-only
|
||||
estBytes int64 // the estimate the gate weighed (for the blocked notification)
|
||||
warns []string
|
||||
status string // persisted SharesLastStatus
|
||||
}
|
||||
|
||||
// runOffboxSharesLeg pushes the shares source. Caller holds the running flag and has already ensured
|
||||
// the repo. Returns the leg result plus a hard error only when the restic call itself failed.
|
||||
func (m *Manager) runOffboxSharesLeg(ctx context.Context, base, env []string, t *settings.OffboxTarget) (sharesLegResult, error) {
|
||||
var res sharesLegResult
|
||||
if !m.sharesEnabled() {
|
||||
// Sharing off / no shares registered: a clean no-op. NO `_shares` restic group is created —
|
||||
// an empty group would age through retention forever and imply a protection that isn't there.
|
||||
return res, nil
|
||||
}
|
||||
|
||||
shares := m.classifiedShares()
|
||||
var mandatory []classifiedShare
|
||||
for _, sh := range shares {
|
||||
if sh.mandatory {
|
||||
mandatory = append(mandatory, sh)
|
||||
}
|
||||
}
|
||||
if len(shares) > 0 && len(mandatory) == 0 {
|
||||
// Every share is tier-2-only. The FILES correctly stay off-site-excluded (Scenario B), but the
|
||||
// definitions still ride offsite: they are ~1 KB and they are what makes a DR restore give the
|
||||
// customer their share configuration back rather than an empty page.
|
||||
m.logger.Printf("[INFO] [shares] offsite: no share is marked for the cloud — pushing share definitions only")
|
||||
}
|
||||
|
||||
payloadDir, passdbOK, perr := m.buildSharesPayload()
|
||||
if perr != nil {
|
||||
// Without a payload there is nothing to anchor a restore on; push the files anyway rather than
|
||||
// skipping protection, but say so loudly.
|
||||
m.logger.Printf("[ERROR] [shares] offsite: payload staging failed — pushing share files without the definition manifest: %v", perr)
|
||||
res.warns = append(res.warns, "A megosztás-beállítások távoli mentése nem sikerült — a fájlok mentése megtörtént.")
|
||||
payloadDir = ""
|
||||
}
|
||||
if !passdbOK {
|
||||
res.warns = append(res.warns, "A megosztás jelszava nem került a mentésbe (a megosztás szolgáltatás nem futott) — visszaállítás után újra meg kell adni.")
|
||||
}
|
||||
|
||||
paths := make([]string, 0, len(mandatory)+1)
|
||||
if payloadDir != "" {
|
||||
paths = append(paths, payloadDir)
|
||||
}
|
||||
sharePaths := make([]string, 0, len(mandatory))
|
||||
for _, sh := range mandatory {
|
||||
sharePaths = append(sharePaths, sh.Path)
|
||||
}
|
||||
|
||||
// Pre-push enlargement gate — the SAME arithmetic as the per-app path (offbox.go): last-known repo
|
||||
// raw-data bytes + this push's estimate crossing the soft quota degrades the push instead of
|
||||
// failing it. Here the degradation floor is the manifest rather than a recovery unit.
|
||||
if len(sharePaths) > 0 && t != nil && t.QuotaGB > 0 {
|
||||
var est int64
|
||||
for _, p := range sharePaths {
|
||||
est += m.offboxSize()(p)
|
||||
}
|
||||
if t.RepoSizeBytes+est >= int64(t.QuotaGB)*offboxGiB {
|
||||
m.logger.Printf("[INFO] [shares] offsite: enlargement blocked by quota (est %s + repo %s ≥ %d GB) — definitions-only push continues",
|
||||
humanizeBytes(est), humanizeBytes(t.RepoSizeBytes), t.QuotaGB)
|
||||
res.blocked = true
|
||||
res.estBytes = est
|
||||
sharePaths = nil
|
||||
}
|
||||
}
|
||||
paths = append(paths, sharePaths...)
|
||||
if len(paths) == 0 {
|
||||
m.logger.Printf("[WARN] [shares] offsite: nothing to push (no payload, no eligible share) — skipped")
|
||||
res.status = "skipped"
|
||||
return res, nil
|
||||
}
|
||||
|
||||
args := append([]string{"backup", "--tag", "felhom-offbox", "--tag", SharesPseudoStack}, paths...)
|
||||
bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||||
out, berr := m.resticStep(bctx, env, base, "backup:"+SharesPseudoStack, args...)
|
||||
cancel()
|
||||
if berr != nil {
|
||||
m.logger.Printf("[ERROR] [shares] offsite push failed: %v: %s", berr, truncate(out))
|
||||
res.status = "error"
|
||||
return res, fmt.Errorf("offbox backup %s: %w", SharesDisplayName, berr)
|
||||
}
|
||||
res.ran = true
|
||||
res.count = len(sharePaths)
|
||||
res.status = "ok"
|
||||
if res.blocked {
|
||||
res.status = "blocked"
|
||||
}
|
||||
m.logger.Printf("[INFO] [shares] offsite push OK: %d share folder(s) + definitions", res.count)
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// recordSharesOffsiteStatus persists the per-tier status the „Megosztás" page renders. Kept separate
|
||||
// from the app-wide offsite status so a page can state SHARES truth without inferring it.
|
||||
func (m *Manager) recordSharesOffsiteStatus(res sharesLegResult) {
|
||||
if m.settings == nil || res.status == "" {
|
||||
return
|
||||
}
|
||||
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.SharesLastRun = time.Now().UTC().Format(time.RFC3339)
|
||||
o.SharesLastStatus = res.status
|
||||
o.SharesLastCount = res.count
|
||||
}); err != nil {
|
||||
m.logger.Printf("[WARN] [shares] offsite status persist failed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// SharesOffsiteStatus returns the last shares-leg outcome for the „Megosztás" page: the RFC3339 run
|
||||
// stamp, the status label and how many share folders the push covered. ok=false when no offsite
|
||||
// target is configured or the leg has never run.
|
||||
func (m *Manager) SharesOffsiteStatus() (lastRun, status string, count int, ok bool) {
|
||||
if m.settings == nil {
|
||||
return "", "", 0, false
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || t.SharesLastStatus == "" {
|
||||
return "", "", 0, false
|
||||
}
|
||||
return t.SharesLastRun, t.SharesLastStatus, t.SharesLastCount, true
|
||||
}
|
||||
|
||||
// sharesBlockedWarning renders the customer-facing note for a quota-degraded shares push. It goes
|
||||
// through DisplayStackName's vocabulary deliberately: the reserved `_shares` key must never appear
|
||||
// in Hungarian prose.
|
||||
func sharesBlockedWarning() string {
|
||||
return fmt.Sprintf("Figyelmeztetés: a tárhelykeret miatt a(z) %s tartalma nem került a távoli mentésbe — csak a megosztás-beállítások.",
|
||||
strings.ToLower(SharesDisplayName))
|
||||
}
|
||||
@@ -0,0 +1,259 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-7b offsite shares leg. The headline test in this file is the B′ ISOLATION PROOF: adding the
|
||||
// shares source must leave every per-app restic invocation BYTE-IDENTICAL. That is the whole premise
|
||||
// of Model B′ — if it does not hold, the design has silently become engine-loop surgery.
|
||||
|
||||
// sharesOffboxEnv wires an offbox manager with one app (unit on `drive`) and the shares feature on,
|
||||
// so a run can be taken with and without shares against the SAME paths.
|
||||
type sharesOffboxEnv struct {
|
||||
m *Manager
|
||||
sett *settings.Settings
|
||||
drive string
|
||||
unit string
|
||||
}
|
||||
|
||||
func newSharesOffboxEnv(t *testing.T, app string) *sharesOffboxEnv {
|
||||
t.Helper()
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
unit := mkUnit(t, drive, app)
|
||||
prov.hdd[app] = drive
|
||||
if err := sett.SetAppOffbox(app, true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.SetSMBEnabled(true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.SetSharesPassdbCapturer(func() ([]byte, error) { return []byte("FAKE-PASSDB"), nil })
|
||||
return &sharesOffboxEnv{m: m, sett: sett, drive: drive, unit: unit}
|
||||
}
|
||||
|
||||
// addOffsiteShare registers an available share on the env's drive.
|
||||
func (e *sharesOffboxEnv) addOffsiteShare(t *testing.T, name string, offsite bool) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(e.drive, name)
|
||||
if err := os.MkdirAll(p, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := e.sett.AddSMBShare(settings.SMBShare{Name: name, Path: p, Offsite: offsite, CreatedAt: "2026-07-18T00:00:00Z"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// run takes one offsite run and returns the capture.
|
||||
func (e *sharesOffboxEnv) run(t *testing.T) *backupCapture {
|
||||
t.Helper()
|
||||
cap := &backupCapture{}
|
||||
e.m.SetOffboxRunner(cap.runner())
|
||||
if err := e.m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
return cap
|
||||
}
|
||||
|
||||
// THE B′ ISOLATION PROOF. One app, then the same app plus one mandatory share: the app's restic argv
|
||||
// must be byte-identical across both runs, and the shares source must appear as exactly ONE
|
||||
// additional call. Red-proof: make the shares leg append its paths into the app's argv instead of
|
||||
// issuing its own call — this test fails.
|
||||
func TestOffboxSharesLegLeavesAppCallsByteIdentical(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
|
||||
baseline := env.run(t)
|
||||
baseArgs := baseline.byStack["immich"]
|
||||
if len(baseArgs) == 0 {
|
||||
t.Fatal("precondition: the baseline run produced no app backup call")
|
||||
}
|
||||
if baseline.backups != 1 {
|
||||
t.Fatalf("precondition: baseline should be exactly 1 backup call, got %d", baseline.backups)
|
||||
}
|
||||
|
||||
env.addOffsiteShare(t, "dokumentumok", true)
|
||||
withShares := env.run(t)
|
||||
|
||||
gotArgs := withShares.byStack["immich"]
|
||||
if strings.Join(gotArgs, "\x00") != strings.Join(baseArgs, "\x00") {
|
||||
t.Errorf("B′ INVARIANT VIOLATED — the app's restic argv changed when shares were added:\n baseline: %v\n with shares: %v", baseArgs, gotArgs)
|
||||
}
|
||||
if withShares.backups != 2 {
|
||||
t.Errorf("expected exactly ONE additional restic call for the shares source, got %d total", withShares.backups)
|
||||
}
|
||||
if _, ok := withShares.byStack[SharesPseudoStack]; !ok {
|
||||
t.Fatalf("no restic call tagged %q was issued: %v", SharesPseudoStack, withShares.byStack)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario A: a mandatory share reaches offsite — correct tags, the manifest staging dir, and the
|
||||
// share folder. Red-proof: flip the mandatory→offsite mapping (push only non-mandatory shares) and
|
||||
// this fails.
|
||||
func TestOffboxSharesLegPushesMandatoryShare(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
sharePath := env.addOffsiteShare(t, "dokumentumok", true)
|
||||
|
||||
cap := env.run(t)
|
||||
args := cap.byStack[SharesPseudoStack]
|
||||
if len(args) == 0 {
|
||||
t.Fatal("no shares call issued")
|
||||
}
|
||||
if !contains(args, "felhom-offbox") || !contains(args, SharesPseudoStack) {
|
||||
t.Errorf("shares call must carry BOTH tags [felhom-offbox, %s]: %v", SharesPseudoStack, args)
|
||||
}
|
||||
if !contains(args, sharePath) {
|
||||
t.Errorf("shares call missing the mandatory share path %q: %v", sharePath, args)
|
||||
}
|
||||
if !contains(args, env.m.SharesPayloadDir()) {
|
||||
t.Errorf("shares call missing the manifest staging dir %q: %v", env.m.SharesPayloadDir(), args)
|
||||
}
|
||||
// The manifest on disk must be the registry.
|
||||
blob, err := os.ReadFile(filepath.Join(env.m.SharesPayloadDir(), sharesManifestName))
|
||||
if err != nil {
|
||||
t.Fatalf("manifest not staged: %v", err)
|
||||
}
|
||||
if !strings.Contains(string(blob), "dokumentumok") {
|
||||
t.Errorf("manifest does not describe the share: %s", blob)
|
||||
}
|
||||
// Per-tier status must be recorded for the „Megosztás" page.
|
||||
_, status, count, ok := env.m.SharesOffsiteStatus()
|
||||
if !ok || status != "ok" || count != 1 {
|
||||
t.Errorf("SharesOffsiteStatus = (%q, %d, %v), want (ok, 1, true)", status, count, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario B: an OPTIONAL share is tier-2-only — its path must appear in NO restic argument.
|
||||
func TestOffboxSharesLegExcludesOptionalShare(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
mandatoryPath := env.addOffsiteShare(t, "dokumentumok", true)
|
||||
optionalPath := env.addOffsiteShare(t, "filmek", false)
|
||||
|
||||
cap := env.run(t)
|
||||
for tag, args := range cap.byStack {
|
||||
if contains(args, optionalPath) {
|
||||
t.Errorf("OPTIONAL share path leaked into the %q restic call: %v", tag, args)
|
||||
}
|
||||
}
|
||||
if !contains(cap.byStack[SharesPseudoStack], mandatoryPath) {
|
||||
t.Error("the mandatory share should still be pushed")
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C: the quota gate degrades the push to the MANIFEST ONLY — definitions protection never
|
||||
// regresses — the blocked set gains the reserved key, and the notification is edge-triggered so a
|
||||
// second identical run does NOT re-notify. Red-proof: drop the manifest-only degradation (skip the
|
||||
// whole leg when blocked) and the "manifest still pushed" assertion fails.
|
||||
func TestOffboxSharesLegQuotaDegradesToManifestOnly(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
sharePath := env.addOffsiteShare(t, "dokumentumok", true)
|
||||
|
||||
// A 1 GB quota with a 2 GB share estimate: the gate must trip.
|
||||
if err := env.sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.QuotaGB = 1 }); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
env.m.SetOffboxSizer(func(string) int64 { return 2 * offboxGiB })
|
||||
|
||||
var notified []string
|
||||
env.m.SetOffboxEnlargeBlockedNotifier(func(stack string, _ int64, _, _ int) {
|
||||
notified = append(notified, stack)
|
||||
})
|
||||
|
||||
cap := env.run(t)
|
||||
args := cap.byStack[SharesPseudoStack]
|
||||
if len(args) == 0 {
|
||||
t.Fatal("the blocked run must still push the definitions, not skip the leg entirely")
|
||||
}
|
||||
if contains(args, sharePath) {
|
||||
t.Errorf("a quota-blocked push must NOT carry the share folder: %v", args)
|
||||
}
|
||||
if !contains(args, env.m.SharesPayloadDir()) {
|
||||
t.Errorf("a quota-blocked push MUST still carry the manifest (definitions protection never regresses): %v", args)
|
||||
}
|
||||
// The persisted blocked set keeps the RAW key (templates index by it)…
|
||||
tgt := env.sett.GetOffboxTarget()
|
||||
if !containsStr(tgt.EnlargedBlocked, SharesPseudoStack) {
|
||||
t.Errorf("EnlargedBlocked should contain the raw %q key, got %v", SharesPseudoStack, tgt.EnlargedBlocked)
|
||||
}
|
||||
// …while the NOTIFICATION boundary renders the Hungarian display name.
|
||||
if len(notified) != 1 || notified[0] != SharesDisplayName {
|
||||
t.Errorf("notification should fire once as %q, got %v", SharesDisplayName, notified)
|
||||
}
|
||||
// The customer-facing warning must not leak the reserved key either.
|
||||
if strings.Contains(tgt.LastWarning, SharesPseudoStack) {
|
||||
t.Errorf("the reserved key leaked into Hungarian prose: %q", tgt.LastWarning)
|
||||
}
|
||||
|
||||
// Edge-trigger: an identical second run must NOT re-notify.
|
||||
notified = nil
|
||||
env.run(t)
|
||||
if len(notified) != 0 {
|
||||
t.Errorf("a persistently-blocked shares source must not re-notify nightly, got %v", notified)
|
||||
}
|
||||
}
|
||||
|
||||
// Sharing disabled / no shares: no `_shares` restic group is created at all.
|
||||
func TestOffboxSharesLegNoOpWhenSharingOff(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
if err := env.sett.SetSMBEnabled(false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cap := env.run(t)
|
||||
if _, ok := cap.byStack[SharesPseudoStack]; ok {
|
||||
t.Error("a disabled sharing feature must create no _shares snapshot group")
|
||||
}
|
||||
if cap.backups != 1 {
|
||||
t.Errorf("expected only the app's call, got %d", cap.backups)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario F, offsite side: a share on an unavailable drive reaches NO restic argument, and the run
|
||||
// still covers the healthy shares.
|
||||
func TestOffboxSharesLegSkipsDeadMount(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
live := env.addOffsiteShare(t, "elo", true)
|
||||
dead := filepath.Join(env.drive, "halott")
|
||||
if err := env.sett.AddSMBShare(settings.SMBShare{Name: "halott", Path: dead, Offsite: true, CreatedAt: "2026-07-18T00:00:00Z"}); err != nil {
|
||||
t.Fatal(err)
|
||||
} // folder deliberately never created → unavailable
|
||||
|
||||
cap := env.run(t)
|
||||
args := cap.byStack[SharesPseudoStack]
|
||||
if contains(args, dead) {
|
||||
t.Errorf("an unavailable share path reached the restic argv: %v", args)
|
||||
}
|
||||
if !contains(args, live) {
|
||||
t.Errorf("the healthy share must still be pushed: %v", args)
|
||||
}
|
||||
}
|
||||
|
||||
// A run whose ONLY cloud content is shares must not be told "nothing is selected".
|
||||
func TestOffboxSharesLegSuppressesZeroToggleNotice(t *testing.T) {
|
||||
env := newSharesOffboxEnv(t, "immich")
|
||||
if err := env.sett.SetAppOffbox("immich", false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
env.addOffsiteShare(t, "dokumentumok", true)
|
||||
|
||||
env.run(t)
|
||||
if w := env.sett.GetOffboxTarget().LastWarning; strings.Contains(w, "nincs mentésre jelölt alkalmazás") {
|
||||
t.Errorf("a box whose cloud content is its shares is covered — misleading warning: %q", w)
|
||||
}
|
||||
}
|
||||
|
||||
// containsStr is a small slice helper (the package's `contains` takes the restic argv shape).
|
||||
func containsStr(hay []string, needle string) bool {
|
||||
for _, h := range hay {
|
||||
if h == needle {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// R-203 Part 2 — "ok" must mean the mandatory data is in the snapshot.
|
||||
//
|
||||
// The defect these pin is NOT that the gap went undetected. It WAS detected, and warned about, in
|
||||
// Hungarian, naming the app and the folders — that warning is what stopped the drill. The defect is
|
||||
// that the run reported `ok` beside it, and a warning standing beside a success is read as a success.
|
||||
|
||||
func mandatoryUserdata(rel string) ClassifiedBind {
|
||||
return ClassifiedBind{ComposeBind: appbackup.ComposeBind{Root: appbackup.RootUserdata, RelPath: rel}, Class: appbackup.ClassMandatory}
|
||||
}
|
||||
|
||||
// Scenario C — a MANDATORY declared path absent on disk is a STRUCTURAL gap, not just prose.
|
||||
//
|
||||
// RED-PROOF: stop recording capGaps into res.mandatoryGaps (or drop the third return) and the verdict
|
||||
// has nothing to act on — the run reports `ok` over a mandatory gap, which is production behaviour up
|
||||
// to v0.196.0.
|
||||
func TestOffboxCaptureSet_MandatoryGapIsStructural(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["calibre-web"] = drive
|
||||
prov.binds["calibre-web"] = []ClassifiedBind{mandatoryUserdata("media/books")}
|
||||
prov.has["calibre-web"] = true
|
||||
|
||||
// The declared directory does not exist on disk — exactly the shape the drill hit.
|
||||
extra, warns, gaps := m.offboxCaptureSet("calibre-web")
|
||||
if len(gaps) != 1 || gaps[0] != "media/books" {
|
||||
t.Fatalf("a missing MANDATORY path must be reported as a structural gap, got %v", gaps)
|
||||
}
|
||||
if len(warns) == 0 {
|
||||
t.Error("the customer-facing Hungarian warning must SURVIVE this change — it is what caught the defect")
|
||||
}
|
||||
if len(extra) != 0 {
|
||||
t.Errorf("a missing path must not be handed to restic, got %v", extra)
|
||||
}
|
||||
|
||||
// Create it: no gap, no warning, and the path IS captured.
|
||||
nsRoot := appbackup.NamespaceRootFor(drive, m.systemDataPath)
|
||||
if err := os.MkdirAll(filepath.Join(appbackup.UserdataDir(nsRoot), "media", "books"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extra2, warns2, gaps2 := m.offboxCaptureSet("calibre-web")
|
||||
if len(gaps2) != 0 || len(warns2) != 0 {
|
||||
t.Fatalf("a PRESENT mandatory path must be silent, got gaps %v warns %v", gaps2, warns2)
|
||||
}
|
||||
if len(extra2) != 1 {
|
||||
t.Fatalf("a present mandatory path must be handed to restic, got %v", extra2)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario D — an OPTIONAL declared path absent on disk changes nothing.
|
||||
//
|
||||
// RED-PROOF: remove the `p.Class == ClassMandatory` check in the stat-filter → an optional gap starts
|
||||
// being reported, and together with the verdict would flip every app with an unused optional folder
|
||||
// to not-ok, which is how a status stops being read.
|
||||
//
|
||||
// STATED BECAUSE IT CHANGES WHAT THIS PROVES: TierOffsite's tierKeeps() already admits ClassMandatory
|
||||
// only, so an optional path cannot reach the stat-filter today. The class check is therefore a NO-OP
|
||||
// and NO customer-visible warning disappears with it. It is written for parity with Tier 2 and so the
|
||||
// verdict can never be flipped by an optional folder if that tier filter ever widens.
|
||||
func TestOffboxCaptureSet_OptionalGapIsSilent(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["komga"] = drive
|
||||
prov.binds["komga"] = []ClassifiedBind{optionalUserdata("media/comics")}
|
||||
prov.has["komga"] = true
|
||||
|
||||
extra, warns, gaps := m.offboxCaptureSet("komga")
|
||||
if len(gaps) != 0 {
|
||||
t.Fatalf("an absent OPTIONAL path must be silent, got gaps %v", gaps)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("an absent OPTIONAL path must raise no customer warning, got %v", warns)
|
||||
}
|
||||
if len(extra) != 0 {
|
||||
t.Fatalf("an absent path must not be captured, got %v", extra)
|
||||
}
|
||||
}
|
||||
|
||||
// A mandatory path that IS present alongside an absent optional one: still silent, still captured.
|
||||
func TestOffboxCaptureSet_MixedClassesOnlyMandatoryCounts(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, _, prov := classifiedOffboxManager(t, drive)
|
||||
prov.hdd["mixed"] = drive
|
||||
prov.binds["mixed"] = []ClassifiedBind{mandatoryUserdata("docs"), optionalUserdata("cache")}
|
||||
prov.has["mixed"] = true
|
||||
|
||||
nsRoot := appbackup.NamespaceRootFor(drive, m.systemDataPath)
|
||||
if err := os.MkdirAll(filepath.Join(appbackup.UserdataDir(nsRoot), "docs"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
extra, warns, gaps := m.offboxCaptureSet("mixed")
|
||||
if len(gaps) != 0 || len(warns) != 0 {
|
||||
t.Fatalf("a present mandatory + absent optional must be silent, got gaps %v warns %v", gaps, warns)
|
||||
}
|
||||
if len(extra) != 1 {
|
||||
t.Fatalf("the mandatory path must be captured, got %v", extra)
|
||||
}
|
||||
}
|
||||
|
||||
// The verdict rule itself, over its inputs. The surrounding run needs a live restic, so the decision
|
||||
// is asserted where it is made rather than through a fake repository.
|
||||
func TestMandatoryGapsDecideTheVerdict(t *testing.T) {
|
||||
verdict := func(gaps map[string][]string) string {
|
||||
if len(gaps) > 0 {
|
||||
return "incomplete"
|
||||
}
|
||||
return "ok"
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
gaps map[string][]string
|
||||
want string
|
||||
}{
|
||||
{"no gaps", nil, "ok"},
|
||||
{"empty map", map[string][]string{}, "ok"},
|
||||
{"one app one folder", map[string][]string{"calibre-web": {"media/books"}}, "incomplete"},
|
||||
{"two apps", map[string][]string{"a": {"x"}, "b": {"y"}}, "incomplete"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := verdict(tc.gaps)
|
||||
if got != tc.want {
|
||||
t.Fatalf("gaps %v → %q, want %q", tc.gaps, got, tc.want)
|
||||
}
|
||||
// "incomplete" must be distinct from every value that already existed, so a checker or a
|
||||
// template matching on those cannot silently treat a coverage gap as one of them.
|
||||
if got == "ok" && tc.want == "incomplete" {
|
||||
t.Fatal("a coverage gap must never read as ok")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C, THROUGH THE RUN — the verdict itself, not just the capture set.
|
||||
//
|
||||
// The first version of this file tested offboxCaptureSet alone, and its "red-proof" PASSED: the
|
||||
// mutation (dropping the gap recording) lives in runOffboxInternal, which that test never reaches.
|
||||
// A mutation that the test cannot observe is not a red-proof, and the fix is the test, not the code.
|
||||
//
|
||||
// RED-PROOF (now real): make the gap recording unreachable (`if false && len(capGaps) > 0`) or
|
||||
// restore `o.LastStatus = "ok"` unconditionally → this FAILS with the run reporting ok over a
|
||||
// mandatory gap, which is production behaviour up to v0.196.0.
|
||||
func TestOffboxRun_MandatoryGapMakesTheRunIncomplete(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
mkUnit(t, drive, "calibre-web")
|
||||
prov.hdd["calibre-web"] = drive
|
||||
prov.has["calibre-web"] = true
|
||||
// Declared MANDATORY and deliberately ABSENT on disk — the drill's shape.
|
||||
prov.binds["calibre-web"] = []ClassifiedBind{mandatoryUserdata("media/books")}
|
||||
_ = sett.SetAppOffbox("calibre-web", true)
|
||||
|
||||
var gapNotified map[string][]string
|
||||
m.SetOffboxGapNotify(func(g map[string][]string) { gapNotified = g })
|
||||
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("the run itself must SUCCEED — a coverage gap is not a failed run: %v", err)
|
||||
}
|
||||
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.LastStatus != "incomplete" {
|
||||
t.Fatalf("LastStatus = %q, want \"incomplete\" — a run that dropped a MANDATORY directory is "+
|
||||
"not a successful run, and reporting ok beside a warning is how this defect hid", got.LastStatus)
|
||||
}
|
||||
// What WAS captured is still recorded — half a backup is not no backup.
|
||||
if got.LastSuccess == "" {
|
||||
t.Error("LastSuccess must still record what was captured (§8.5) — suppressing it would be its own lie")
|
||||
}
|
||||
if cap.backups != 1 {
|
||||
t.Errorf("the unit must still be pushed, got %d backup calls", cap.backups)
|
||||
}
|
||||
// And the OPERATOR is told, not only the log.
|
||||
if len(gapNotified) != 1 || len(gapNotified["calibre-web"]) != 1 || gapNotified["calibre-web"][0] != "media/books" {
|
||||
t.Fatalf("the operator gap signal did not fire with the app and folder, got %v", gapNotified)
|
||||
}
|
||||
}
|
||||
|
||||
// The companion: no gap → ok, and no operator signal. Without this, "incomplete" everywhere would
|
||||
// also pass the test above.
|
||||
func TestOffboxRun_NoGapStaysOk(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
mkUnit(t, drive, "calibre-web")
|
||||
nsRoot := appbackup.NamespaceRootFor(drive, m.systemDataPath)
|
||||
if err := os.MkdirAll(filepath.Join(appbackup.UserdataDir(nsRoot), "media", "books"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["calibre-web"] = drive
|
||||
prov.has["calibre-web"] = true
|
||||
prov.binds["calibre-web"] = []ClassifiedBind{mandatoryUserdata("media/books")}
|
||||
_ = sett.SetAppOffbox("calibre-web", true)
|
||||
|
||||
fired := false
|
||||
m.SetOffboxGapNotify(func(map[string][]string) { fired = true })
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
if got := sett.GetOffboxTarget(); got.LastStatus != "ok" {
|
||||
t.Fatalf("LastStatus = %q, want ok — a complete run must not be downgraded", got.LastStatus)
|
||||
}
|
||||
if fired {
|
||||
t.Error("the operator gap signal must NOT fire when nothing was missed")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-234 — a run that SKIPPED an app the customer selected is not a successful run.
|
||||
//
|
||||
// The same paragraph the R-203 verdict block already carries — "a warning beside a success is read
|
||||
// as a success" — was applied to one of the two shapes it describes. An app missing a declared
|
||||
// mandatory FOLDER made the run `incomplete`; an app skipped ENTIRELY, with nothing of it in the
|
||||
// snapshot at all, still reported `ok`. The smaller gap moved the verdict and the bigger one did not.
|
||||
//
|
||||
// Run-level on purpose: the classification and the verdict are both inside the run, and the sibling
|
||||
// test file records what happened when its first version asserted the capture helper alone — its
|
||||
// red-proof passed while the defect was untouched.
|
||||
|
||||
// Scenario A — a selected, DEPLOYED app with no recovery unit makes the run incomplete, names itself,
|
||||
// and does not suppress what was captured.
|
||||
//
|
||||
// RED-PROOF: drop `unprotected` from the verdict condition (leave only mandatoryGaps) → this FAILS
|
||||
// with the run reporting ok over a skipped app, which is production behaviour up to v0.204.0.
|
||||
func TestOffboxRun_SkippedSelectedAppIsIncomplete(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
|
||||
// `kept` has a unit and is pushed; `dropped` is selected and deployed but has NO unit, so the
|
||||
// per-app loop skips it. backedUp>0 is what made the existing no-silent-success guard stay quiet.
|
||||
mkUnit(t, drive, "kept")
|
||||
prov.hdd["kept"] = drive
|
||||
prov.has["kept"] = true
|
||||
prov.hdd["dropped"] = drive
|
||||
prov.has["dropped"] = true
|
||||
prov.deployed = map[string]bool{"kept": true, "dropped": true}
|
||||
_ = sett.SetAppOffbox("kept", true)
|
||||
_ = sett.SetAppOffbox("dropped", true)
|
||||
|
||||
var gapNotified map[string][]string
|
||||
m.SetOffboxGapNotify(func(g map[string][]string) { gapNotified = g })
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("the run itself must SUCCEED — a skipped app is a coverage gap, not a failed run: %v", err)
|
||||
}
|
||||
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.LastStatus != "incomplete" {
|
||||
t.Fatalf("LastStatus = %q, want \"incomplete\" — the customer selected an app and the run did not "+
|
||||
"carry it; on 2026-08-06 this reported „✓ Rendben” and the restore refused minutes later", got.LastStatus)
|
||||
}
|
||||
// Scenario A: the counters and the anchor still record what WAS captured.
|
||||
if got.LastSuccess == "" {
|
||||
t.Error("LastSuccess must still record what was captured — half a backup is not no backup")
|
||||
}
|
||||
if cap.backups != 1 {
|
||||
t.Errorf("the app that HAD a unit must still be pushed, got %d backup calls", cap.backups)
|
||||
}
|
||||
// Scenario E: which app, and why.
|
||||
if !strings.Contains(got.LastWarning, "dropped") {
|
||||
t.Errorf("the warning must NAME the skipped app, got %q", got.LastWarning)
|
||||
}
|
||||
if !strings.Contains(got.LastWarning, "nincs helyi ment") {
|
||||
t.Errorf("the warning must say WHY it was skipped, got %q", got.LastWarning)
|
||||
}
|
||||
if !strings.Contains(got.LastWarning, "következő ment") {
|
||||
t.Errorf("the warning must say WHEN it will be protected, got %q", got.LastWarning)
|
||||
}
|
||||
// Scenario B: the operator hears about it, in the same vocabulary as a folder gap.
|
||||
if len(gapNotified["dropped"]) == 0 {
|
||||
t.Fatalf("the operator signal must carry the skipped app, got %v", gapNotified)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario C — a healthy run is untouched. Without this, "always incomplete" would also pass above,
|
||||
// and a status that is never green is a status that stops being read.
|
||||
//
|
||||
// RED-PROOF: count EVERY skip (drop the classification switch and use len(res.missing)) → a healthy
|
||||
// run goes amber and this FAILS.
|
||||
func TestOffboxRun_HealthyRunStaysOk(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
mkUnit(t, drive, "kept")
|
||||
prov.hdd["kept"] = drive
|
||||
prov.has["kept"] = true
|
||||
_ = sett.SetAppOffbox("kept", true)
|
||||
|
||||
fired := false
|
||||
m.SetOffboxGapNotify(func(map[string][]string) { fired = true })
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.LastStatus != "ok" {
|
||||
t.Fatalf("LastStatus = %q, want ok — every selected app was carried", got.LastStatus)
|
||||
}
|
||||
if fired {
|
||||
t.Error("the operator signal must NOT fire when nothing was missed")
|
||||
}
|
||||
if strings.Contains(got.LastWarning, "NEM kerültek be") {
|
||||
t.Errorf("a healthy run must carry no skip warning, got %q", got.LastWarning)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario D — a box with NOTHING selected keeps today's behaviour: ok, with the existing
|
||||
// zero-selection notice. An unconfigured box reporting incomplete forever is its own defect.
|
||||
//
|
||||
// RED-PROOF: count the empty selection as a gap → this box goes permanently amber and this FAILS.
|
||||
func TestOffboxRun_NothingSelectedIsNotAGap(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, _ := classifiedOffboxManager(t, drive)
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.LastStatus != "ok" {
|
||||
t.Fatalf("LastStatus = %q, want ok — nothing was selected, so nothing was skipped", got.LastStatus)
|
||||
}
|
||||
if !strings.Contains(got.LastWarning, "nincs mentésre jelölt alkalmazás") {
|
||||
t.Errorf("the existing zero-selection notice must survive, got %q", got.LastWarning)
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario F — a selected app that is NOT deployed. Decided deliberately: it is NAMED with what to do
|
||||
// about it, and it does NOT move the verdict, because a box left amber forever by an app somebody
|
||||
// removed is a status nobody reads.
|
||||
func TestOffboxRun_SelectedButUndeployedIsNamedNotCounted(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
mkUnit(t, drive, "kept")
|
||||
prov.hdd["kept"] = drive
|
||||
prov.has["kept"] = true
|
||||
prov.deployed = map[string]bool{"kept": true} // "removed-app" deliberately absent
|
||||
_ = sett.SetAppOffbox("kept", true)
|
||||
// selected, no unit, and NOT in the deployed set
|
||||
_ = sett.SetAppOffbox("removed-app", true)
|
||||
|
||||
fired := false
|
||||
m.SetOffboxGapNotify(func(map[string][]string) { fired = true })
|
||||
cap := &backupCapture{}
|
||||
m.SetOffboxRunner(cap.runner())
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
got := sett.GetOffboxTarget()
|
||||
if got.LastStatus != "ok" {
|
||||
t.Fatalf("LastStatus = %q, want ok — an app that is not installed cannot be protected, and must "+
|
||||
"not hold the box amber forever", got.LastStatus)
|
||||
}
|
||||
if !strings.Contains(got.LastWarning, "removed-app") {
|
||||
t.Errorf("the undeployed selection must still be NAMED, got %q", got.LastWarning)
|
||||
}
|
||||
if !strings.Contains(got.LastWarning, "vedd ki a kijelöl") {
|
||||
t.Errorf("it must say what to do about it, got %q", got.LastWarning)
|
||||
}
|
||||
if fired {
|
||||
t.Error("an undeployed app must not raise the operator gap signal")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,145 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Verification copies — the listing/delete surface for `<nsRoot>/backups/offsite-restore/<app>`.
|
||||
//
|
||||
// WHY THIS EXISTS (v0.147.0, feedback slice 4a): an offsite verification restore wrote its result to
|
||||
// a path the customer was never told, and nothing anywhere listed what had accumulated. Pressing
|
||||
// „Ellenőrző visszaállítás" produced a flash saying it had been restored "to a verification folder
|
||||
// on the drive" — which folder, on which drive, and how much space it was now using were all
|
||||
// invisible. So copies piled up and the only way to find them was SSH.
|
||||
//
|
||||
// The path segments were already open-coded in three places; offsiteRestoreRootFor() is now the one
|
||||
// place `backups/offsite-restore` is spelled, and offboxRestoreScratchDir() builds on it.
|
||||
|
||||
// OffsiteRestoreCopy is one verification copy on disk.
|
||||
type OffsiteRestoreCopy struct {
|
||||
Stack string `json:"stack"` // app slug, or SharesPseudoStack for the shares copy
|
||||
Path string `json:"path"` // absolute path — the thing the customer could not see
|
||||
Size int64 `json:"size"` // bytes
|
||||
SizeHuman string `json:"size_human"` // pre-humanized for the template
|
||||
Created time.Time `json:"created"` // dir mtime; restic writes the tree once, so this is the restore time
|
||||
}
|
||||
|
||||
// offsiteRestoreRootFor returns `<nsRoot>/backups/offsite-restore` for a drive path. THE single place
|
||||
// these segments are written.
|
||||
func (m *Manager) offsiteRestoreRootFor(drivePath string) string {
|
||||
return filepath.Join(m.namespaceRoot(drivePath), "backups", "offsite-restore")
|
||||
}
|
||||
|
||||
// offsiteRestoreDriveRoots returns every drive path a verification copy could live under, in the same
|
||||
// preference order offboxRestoreScratchDir uses to CHOOSE one — so listing can never miss a copy the
|
||||
// restore path was capable of creating. Deduplicated, order preserved.
|
||||
func (m *Manager) offsiteRestoreDriveRoots() []string {
|
||||
seen := map[string]bool{}
|
||||
var roots []string
|
||||
add := func(p string) {
|
||||
p = strings.TrimSpace(p)
|
||||
if p == "" || seen[p] {
|
||||
return
|
||||
}
|
||||
seen[p] = true
|
||||
roots = append(roots, p)
|
||||
}
|
||||
// App HDDs first (offboxRestoreScratchDir's rule 1), then every schedulable path (rules 2 and 3).
|
||||
if m.stackProvider != nil {
|
||||
for _, s := range m.stackProvider.ListDeployedStacks() {
|
||||
add(m.stackProvider.GetStackHDDPath(s.Name))
|
||||
}
|
||||
}
|
||||
if m.settings != nil {
|
||||
for _, sp := range m.settings.GetSchedulableStoragePaths() {
|
||||
add(sp.Path)
|
||||
}
|
||||
}
|
||||
return roots
|
||||
}
|
||||
|
||||
// ListOffsiteRestoreCopies enumerates every verification copy across every candidate drive, newest
|
||||
// first. Missing directories are not an error — "none yet" is the normal state.
|
||||
func (m *Manager) ListOffsiteRestoreCopies() []OffsiteRestoreCopy {
|
||||
sizer := m.offboxSize()
|
||||
var out []OffsiteRestoreCopy
|
||||
seen := map[string]bool{}
|
||||
for _, drive := range m.offsiteRestoreDriveRoots() {
|
||||
root := m.offsiteRestoreRootFor(drive)
|
||||
entries, err := os.ReadDir(root)
|
||||
if err != nil {
|
||||
continue // no copies on this drive (or the drive is not mounted) — not an error
|
||||
}
|
||||
for _, e := range entries {
|
||||
if !e.IsDir() {
|
||||
continue
|
||||
}
|
||||
p := filepath.Join(root, e.Name())
|
||||
if seen[p] {
|
||||
continue // two stacks can resolve to the same drive; list each path once
|
||||
}
|
||||
seen[p] = true
|
||||
c := OffsiteRestoreCopy{Stack: e.Name(), Path: p}
|
||||
if fi, err := e.Info(); err == nil {
|
||||
c.Created = fi.ModTime()
|
||||
}
|
||||
c.Size = sizer(p)
|
||||
c.SizeHuman = humanizeBytes(c.Size)
|
||||
out = append(out, c)
|
||||
}
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].Created.After(out[j].Created) })
|
||||
return out
|
||||
}
|
||||
|
||||
// DeleteOffsiteRestoreCopy removes ONE verification copy.
|
||||
//
|
||||
// This is the only delete path v0.147.0 adds, so it is guarded twice over. The stack name must pass
|
||||
// isSafeStackName (no separators, no traversal), and the resolved path must sit STRICTLY INSIDE a
|
||||
// `backups/offsite-restore` root that this Manager itself computed — a path that merely looks right
|
||||
// is refused. Both checks are on the RESOLVED path, not the input, so a symlinked scratch cannot
|
||||
// walk the delete out of the sandbox.
|
||||
func (m *Manager) DeleteOffsiteRestoreCopy(stack string) error {
|
||||
if !isSafeStackName(stack) {
|
||||
return fmt.Errorf("érvénytelen alkalmazásnév")
|
||||
}
|
||||
for _, drive := range m.offsiteRestoreDriveRoots() {
|
||||
root := m.offsiteRestoreRootFor(drive)
|
||||
target := filepath.Join(root, stack)
|
||||
|
||||
fi, err := os.Stat(target)
|
||||
if err != nil || !fi.IsDir() {
|
||||
continue
|
||||
}
|
||||
// Prefix safety: only ever remove strictly inside `backups/offsite-restore/`. Same shape as
|
||||
// the F5 stale-primary prune (backup.go) — refuse loudly rather than best-effort skip, since
|
||||
// reaching here with an out-of-sandbox path means a helper above is wrong.
|
||||
cleanTarget := filepath.Clean(target)
|
||||
cleanRoot := filepath.Clean(root) + string(filepath.Separator)
|
||||
if !strings.HasPrefix(cleanTarget+string(filepath.Separator), cleanRoot) {
|
||||
m.logger.Printf("[WARN] [offbox] refusing to delete verification copy outside %s: %s", root, cleanTarget)
|
||||
return fmt.Errorf("a törlés útvonala kívül esik az ellenőrző mappán")
|
||||
}
|
||||
if err := os.RemoveAll(cleanTarget); err != nil {
|
||||
return fmt.Errorf("a másolat törlése nem sikerült: %w", err)
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] deleted verification copy: %s", cleanTarget)
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("nincs ilyen ellenőrző másolat")
|
||||
}
|
||||
|
||||
// OffsiteRestoreScratchPath exposes WHERE a verification restore for stack would land, so the UI can
|
||||
// name the full path in the completion message instead of saying "a verification folder somewhere".
|
||||
func (m *Manager) OffsiteRestoreScratchPath(stack string) string {
|
||||
scratch, _, err := m.offboxRestoreScratchDir(stack)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return scratch
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// v0.147.0 slice 4a — the verification-copy listing/delete surface.
|
||||
//
|
||||
// DeleteOffsiteRestoreCopy is the ONLY delete this slice adds, so these tests are about what it must
|
||||
// REFUSE as much as what it must do. Every refusal is asserted as a NON-EFFECT: the neighbouring copy
|
||||
// and the customer's live data must still be on disk afterwards. A guard that returns an error but
|
||||
// deletes anyway passes a naive test and loses data.
|
||||
//
|
||||
// RED-PROOF (run manually, confirmed): neutralise the isSafeStackName check in
|
||||
// DeleteOffsiteRestoreCopy and TestDeleteVerifyCopyRefusesUnsafeNames fails hard — `stack: ""`
|
||||
// resolves to the offsite-restore ROOT and os.RemoveAll takes every verification copy with it. That
|
||||
// is the failure this guard exists to prevent, and it is data loss, not a bad error message.
|
||||
//
|
||||
// The HasPrefix containment check inside DeleteOffsiteRestoreCopy could NOT be red-proofed
|
||||
// independently: with isSafeStackName in front of it, no input this API accepts can reach it with an
|
||||
// escaping path, so removing it leaves every test green (and the variable unused). It is deliberate
|
||||
// defence-in-depth against a future caller or a refactor that loosens the name check — kept, but
|
||||
// honestly labelled here as unproven-by-test rather than pretending to a red-proof it does not have.
|
||||
|
||||
// verifyCopyEnv wires a manager with one drive and materialised verification copies.
|
||||
type verifyCopyEnv struct {
|
||||
m *Manager
|
||||
drive string
|
||||
root string // <nsRoot>/backups/offsite-restore
|
||||
}
|
||||
|
||||
func newVerifyCopyEnv(t *testing.T, copies ...string) *verifyCopyEnv {
|
||||
t.Helper()
|
||||
drive := t.TempDir()
|
||||
m, _, _ := classifiedOffboxManager(t, drive)
|
||||
root := m.offsiteRestoreRootFor(drive)
|
||||
for _, c := range copies {
|
||||
p := filepath.Join(root, c)
|
||||
if err := os.MkdirAll(p, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(p, "payload.bin"), []byte("restored bytes"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return &verifyCopyEnv{m: m, drive: drive, root: root}
|
||||
}
|
||||
|
||||
func TestListOffsiteRestoreCopiesReportsPathAndSize(t *testing.T) {
|
||||
e := newVerifyCopyEnv(t, "immich", "nextcloud")
|
||||
|
||||
got := e.m.ListOffsiteRestoreCopies()
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("listed %d copies, want 2: %+v", len(got), got)
|
||||
}
|
||||
byStack := map[string]OffsiteRestoreCopy{}
|
||||
for _, c := range got {
|
||||
byStack[c.Stack] = c
|
||||
}
|
||||
for _, name := range []string{"immich", "nextcloud"} {
|
||||
c, ok := byStack[name]
|
||||
if !ok {
|
||||
t.Fatalf("%s missing from the listing", name)
|
||||
}
|
||||
// THE POINT of the listing: the customer could not previously see WHERE the copy was.
|
||||
want := filepath.Join(e.root, name)
|
||||
if c.Path != want {
|
||||
t.Errorf("%s path = %q, want %q", name, c.Path, want)
|
||||
}
|
||||
if c.SizeHuman == "" {
|
||||
t.Errorf("%s has no humanized size", name)
|
||||
}
|
||||
if c.Created.IsZero() {
|
||||
t.Errorf("%s has no creation time", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestListOffsiteRestoreCopiesEmptyIsNotAnError(t *testing.T) {
|
||||
// No offsite-restore directory at all — the normal state on a box that never ran a verification
|
||||
// restore. Must be an empty list, not a crash and not a phantom entry.
|
||||
e := newVerifyCopyEnv(t)
|
||||
if got := e.m.ListOffsiteRestoreCopies(); len(got) != 0 {
|
||||
t.Errorf("listed %d copies on a clean box, want 0: %+v", len(got), got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeleteVerifyCopyRemovesOnlyTheNamedOne(t *testing.T) {
|
||||
e := newVerifyCopyEnv(t, "immich", "nextcloud")
|
||||
// Live customer data next to the sandbox — must be untouched by any delete.
|
||||
live := filepath.Join(e.drive, "immich-live")
|
||||
if err := os.MkdirAll(live, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(live, "photo.jpg"), []byte("irreplaceable"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if err := e.m.DeleteOffsiteRestoreCopy("immich"); err != nil {
|
||||
t.Fatalf("delete: %v", err)
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(e.root, "immich")); !os.IsNotExist(err) {
|
||||
t.Error("the named copy survived the delete")
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(e.root, "nextcloud", "payload.bin")); err != nil {
|
||||
t.Errorf("a NEIGHBOURING copy was destroyed: %v", err)
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(live, "photo.jpg")); err != nil {
|
||||
t.Errorf("LIVE CUSTOMER DATA was destroyed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeleteVerifyCopyRefusesUnsafeNames(t *testing.T) {
|
||||
e := newVerifyCopyEnv(t, "immich")
|
||||
// Something outside the sandbox that a traversal would reach.
|
||||
outside := filepath.Join(e.drive, "backups", "primary")
|
||||
if err := os.MkdirAll(outside, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(outside, "unit.tar"), []byte("recovery unit"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for _, bad := range []string{
|
||||
"../primary",
|
||||
"../../..",
|
||||
"..",
|
||||
"immich/../../primary",
|
||||
"/etc",
|
||||
"",
|
||||
} {
|
||||
if err := e.m.DeleteOffsiteRestoreCopy(bad); err == nil {
|
||||
t.Errorf("delete(%q) was ACCEPTED — it must be refused", bad)
|
||||
}
|
||||
// The refusal must also be a NON-EFFECT.
|
||||
if _, err := os.Stat(filepath.Join(outside, "unit.tar")); err != nil {
|
||||
t.Fatalf("delete(%q) destroyed data outside the sandbox: %v", bad, err)
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(e.root, "immich", "payload.bin")); err != nil {
|
||||
t.Fatalf("delete(%q) destroyed an unrelated copy: %v", bad, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeleteVerifyCopyStaysInsideTheSandbox(t *testing.T) {
|
||||
e := newVerifyCopyEnv(t, "immich")
|
||||
// A well-formed name that simply does not exist must be an error, never a silent success that
|
||||
// could mask a path-resolution bug.
|
||||
if err := e.m.DeleteOffsiteRestoreCopy("no-such-app"); err == nil {
|
||||
t.Error("deleting a non-existent copy reported success")
|
||||
}
|
||||
// Everything under the sandbox root must resolve strictly inside it.
|
||||
for _, c := range e.m.ListOffsiteRestoreCopies() {
|
||||
rel, err := filepath.Rel(e.root, c.Path)
|
||||
if err != nil || rel == ".." || filepath.IsAbs(rel) || len(rel) > 2 && rel[:2] == ".." {
|
||||
t.Errorf("listed copy %q resolves outside %q", c.Path, e.root)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestOffsiteRestoreScratchPathMatchesTheListing pins the two halves together: the path the UI names
|
||||
// in the completion flash must be the same path the listing (and therefore the delete button) uses.
|
||||
// If these ever diverge, the customer is told about a directory the page cannot show or remove.
|
||||
func TestOffsiteRestoreScratchPathMatchesTheListing(t *testing.T) {
|
||||
e := newVerifyCopyEnv(t, "immich")
|
||||
named := e.m.OffsiteRestoreScratchPath("immich")
|
||||
if named == "" {
|
||||
t.Fatal("OffsiteRestoreScratchPath returned empty — the flash would fall back to the vague wording")
|
||||
}
|
||||
var listed string
|
||||
for _, c := range e.m.ListOffsiteRestoreCopies() {
|
||||
if c.Stack == "immich" {
|
||||
listed = c.Path
|
||||
}
|
||||
}
|
||||
if named != listed {
|
||||
t.Errorf("flash names %q but the listing shows %q", named, listed)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// the real demo-hp target shape — the values the sanitiser must remove literally
|
||||
func diagTarget() *settings.OffboxTarget {
|
||||
return &settings.OffboxTarget{
|
||||
Host: "u629488-sub3.your-storagebox.de", User: "u629488-sub3",
|
||||
RepoPath: "/home/felhom-repo", Port: 23,
|
||||
}
|
||||
}
|
||||
|
||||
// F-DIAG — four causes collapsed into one string, and that string was a RAW error passthrough.
|
||||
//
|
||||
// Two separate defects in one line of code:
|
||||
// - an operator could not tell a full quota from a dead network without reading logs;
|
||||
// - `err.Error()` from restic/ssh carries the repo reference `sftp:<user>@<host>:<path>`, so the
|
||||
// notification carried a customer-identifying location (and potentially a credential) off the box,
|
||||
// breaking the keys-not-values rule at the one place the text leaves the machine.
|
||||
|
||||
func TestClassifyOffsiteFailure_EachCauseIsDistinct(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
err error
|
||||
want OffsiteFailureClass
|
||||
}{
|
||||
{"quota gate", fmt.Errorf("A távoli mentés túllépte a tárhelykeretet (51/50 GB) — törölj régi mentéseket vagy kérj nagyobb keretet."), OffsiteFailQuota},
|
||||
{"orphaned repo", fmt.Errorf("probe: %w", ErrOffboxOrphaned), OffsiteFailOrphaned},
|
||||
{"no repo", fmt.Errorf("restic: unable to open config file: Stat: file does not exist\nIs there a repository at the following location?"), OffsiteFailNoRepo},
|
||||
{"no units", fmt.Errorf("off-box backup produced no snapshots: 3 app(s) toggled but no recovery unit was found on any connected drive (missing: a, b, c)"), OffsiteFailNoUnits},
|
||||
{"transport refused", fmt.Errorf("dial tcp 1.2.3.4:23: connect: connection refused"), OffsiteFailTransport},
|
||||
{"transport timeout", fmt.Errorf("ssh: handshake failed: i/o timeout"), OffsiteFailTransport},
|
||||
{"transport auth", fmt.Errorf("ssh: permission denied (publickey)"), OffsiteFailTransport},
|
||||
{"unclassified", fmt.Errorf("restic: some future error nobody has seen"), OffsiteFailUnknown},
|
||||
}
|
||||
seen := map[OffsiteFailureClass]bool{}
|
||||
for _, c := range cases {
|
||||
got := ClassifyOffsiteFailure(c.err)
|
||||
if got != c.want {
|
||||
t.Errorf("%s: class = %q, want %q", c.name, got, c.want)
|
||||
}
|
||||
seen[got] = true
|
||||
}
|
||||
// The whole point of F-DIAG: the causes must not collapse.
|
||||
if len(seen) < 5 {
|
||||
t.Errorf("only %d distinct classes across %d causes — the causes are still collapsing", len(seen), len(cases))
|
||||
}
|
||||
}
|
||||
|
||||
// An unclassifiable error must say so rather than being folded into a neighbour. Inventing a precision
|
||||
// the code does not have is how a confident-but-wrong diagnosis ships.
|
||||
func TestClassifyOffsiteFailure_UnknownIsHonest(t *testing.T) {
|
||||
if got := ClassifyOffsiteFailure(fmt.Errorf("something entirely new")); got != OffsiteFailUnknown {
|
||||
t.Errorf("an unclassifiable error was folded into %q instead of being reported as unknown", got)
|
||||
}
|
||||
msg := offsiteFailureMessage(diagTarget(), fmt.Errorf("something entirely new"), time.Minute)
|
||||
if !strings.Contains(msg, "ismeretlen okból") {
|
||||
t.Errorf("the unknown case does not admit it is unknown: %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// THE SECRETS TEST. The repo reference must never survive into a message.
|
||||
//
|
||||
// RED-PROOF: make sanitiseOffsiteError return err.Error() unchanged → this fails with
|
||||
// "the repo reference reached the message".
|
||||
func TestOffsiteFailureMessage_NeverCarriesTheRepoReference(t *testing.T) {
|
||||
leaky := []error{
|
||||
fmt.Errorf(`Fatal: unable to open repository at sftp:u629488-sub3@u629488-sub3.your-storagebox.de:/home/felhom-repo: connection refused`),
|
||||
fmt.Errorf(`ssh: connect to host u629488-sub3.your-storagebox.de port 23: Connection refused`),
|
||||
fmt.Errorf(`restic: repo "sftp:u629488-sub3@u629488-sub3.your-storagebox.de:/home/felhom-repo" locked`),
|
||||
}
|
||||
for _, e := range leaky {
|
||||
msg := offsiteFailureMessage(diagTarget(), e, 42*time.Second)
|
||||
for _, forbidden := range []string{
|
||||
"sftp:",
|
||||
"your-storagebox.de",
|
||||
"u629488-sub3",
|
||||
"/home/felhom-repo",
|
||||
} {
|
||||
if strings.Contains(msg, forbidden) {
|
||||
t.Errorf("the repo reference reached the message (%q leaked):\n %s", forbidden, msg)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(msg, "<repo>") {
|
||||
t.Errorf("the redaction placeholder is absent — the detail may have been dropped silently instead of sanitised:\n %s", msg)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The message must still be ACTIONABLE. Sanitising must not reduce it to a shrug — an operator needs
|
||||
// the cause line plus enough residual detail to act.
|
||||
func TestOffsiteFailureMessage_StaysActionable(t *testing.T) {
|
||||
msg := offsiteFailureMessage(diagTarget(), fmt.Errorf("dial tcp: connect: connection refused"), 90*time.Second)
|
||||
if !strings.Contains(msg, "nem érhető el") {
|
||||
t.Errorf("the transport cause is not named: %q", msg)
|
||||
}
|
||||
if !strings.Contains(msg, "connection refused") {
|
||||
t.Errorf("all actionable detail was stripped along with the secret: %q", msg)
|
||||
}
|
||||
if !strings.Contains(msg, "1m30s") {
|
||||
t.Errorf("the duration was lost: %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// A very long error must be bounded — an unbounded restic dump in an email is its own problem.
|
||||
func TestSanitiseOffsiteError_IsBounded(t *testing.T) {
|
||||
long := fmt.Errorf("%s", strings.Repeat("x", 5000))
|
||||
if got := sanitiseOffsiteErrorFor(diagTarget(), long); len(got) > 320 {
|
||||
t.Errorf("sanitised error is %d chars — unbounded", len(got))
|
||||
}
|
||||
if sanitiseOffsiteErrorFor(diagTarget(), nil) != "" {
|
||||
t.Error("a nil error produced text")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,374 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-47 (v0.153.0) — the DB replay must not race the app, on BOTH restore paths.
|
||||
//
|
||||
// Every test here is a regression guard for a measured incident, not a description of the code.
|
||||
// On 2026-07-19 (DIAG-immich-restore-round2-2026-07-19, H4) the offsite reconstitution started the
|
||||
// WHOLE stack before replaying the dump. immich-server used the window to rebuild `clip_index` two
|
||||
// seconds before the dump's own CREATE INDEX; the replay aborted `relation "clip_index" already
|
||||
// exists` under ON_ERROR_STOP=1, and immich then reported schema drift. The data survived only by
|
||||
// accident of pg_dump's ordering (COPY before CREATE INDEX) — a collision earlier in the script
|
||||
// would have left a genuinely half-restored database, reported identically.
|
||||
//
|
||||
// The property under test is therefore an ORDERING plus a STATE-AT-REPLAY-TIME: at the moment the
|
||||
// import fires, the database service must be up and the full stack must NOT be. Asserting only
|
||||
// "no error" would pass on the pre-fix shape, which is exactly how this shipped.
|
||||
|
||||
// immichLikeCompose is the catalog's immich template reduced to what the resolver reads: the app
|
||||
// services, the DB service, a redis that must never be mistaken for a database, and the top-level
|
||||
// `volumes:`/`networks:` keys (including `immich_postgres_data`) that a line scan would misread.
|
||||
const immichLikeCompose = `services:
|
||||
immich-server:
|
||||
image: ghcr.io/immich-app/immich-server:v3.0.3
|
||||
immich-machine-learning:
|
||||
image: ghcr.io/immich-app/immich-machine-learning:v3.0.3
|
||||
immich-postgres:
|
||||
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0
|
||||
immich-redis:
|
||||
image: redis:7-alpine
|
||||
volumes:
|
||||
immich_ml_cache:
|
||||
immich_postgres_data:
|
||||
networks:
|
||||
traefik-public:
|
||||
external: true
|
||||
`
|
||||
|
||||
// noDBCompose is a DB-free app: nothing here may ever trigger the DB-only phase.
|
||||
const noDBCompose = `services:
|
||||
app:
|
||||
image: ghcr.io/x/app:1
|
||||
cache:
|
||||
image: redis:7-alpine
|
||||
`
|
||||
|
||||
// --- Group A: offsite reconstitute, DB-bearing app (the H4 killer) --------------------------------
|
||||
|
||||
// TestReconstituteReplaysWithOnlyTheDBServiceUp is the core R-47 assertion for the offsite path.
|
||||
// It does not merely check the call ORDER — it captures the provider's state AT THE MOMENT the
|
||||
// import fires, because that is what H4 was: the sequence looked right, and the app was up.
|
||||
//
|
||||
// COMPANION RED-PROOF: replacing the DB-only bring-up in ReconstituteFromOffsite with the pre-fix
|
||||
// full StartStack makes this fail on `full stack was ALREADY UP when the replay fired`.
|
||||
func TestReconstituteReplaysWithOnlyTheDBServiceUp(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "20260719T060000Z", "2026-07-19T06:00:00Z", pgDump(1))
|
||||
|
||||
var dbUpAtReplay, fullUpAtReplay bool
|
||||
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||
dbUpAtReplay = len(prov.gotServices) > 0
|
||||
fullUpAtReplay = prov.fullStarted
|
||||
*imported = append(*imported, p)
|
||||
return nil
|
||||
}
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err != nil {
|
||||
t.Fatalf("reconstitute: %v", err)
|
||||
}
|
||||
if res.DBsReplayed != 1 {
|
||||
t.Fatalf("expected exactly one replay, got %d", res.DBsReplayed)
|
||||
}
|
||||
if !dbUpAtReplay {
|
||||
t.Fatal("the database service was NOT started before the replay — ImportDump has no container to talk to")
|
||||
}
|
||||
if fullUpAtReplay {
|
||||
t.Fatal("the FULL stack was already up when the replay fired — this is H4 exactly: the app races the dump's schema")
|
||||
}
|
||||
if got := strings.Join(prov.gotServices, ","); got != "immich-postgres" {
|
||||
t.Fatalf("DB-only phase started %q, want only the database service immich-postgres", got)
|
||||
}
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,startsvc:immich-postgres,start" {
|
||||
t.Fatalf("sequence = %q, want stop → db-only start → replay → full start", got)
|
||||
}
|
||||
// The undo must have existed before any of it.
|
||||
if res.SafetyDump == "" {
|
||||
t.Fatal("no safety dump recorded")
|
||||
}
|
||||
if _, sErr := os.Stat(res.SafetyDump); sErr != nil {
|
||||
t.Fatalf("safety dump not on disk before the mutation: %v", sErr)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group B: offsite reconstitute, no-DB app (flow unchanged) ------------------------------------
|
||||
|
||||
// TestReconstituteNoDBAppNeverStartsServicesOnly asserts the NEGATIVE: an app with no database must
|
||||
// take exactly one full start and must never enter the DB-only phase. Without this, a bug that
|
||||
// armed the phase for every app would show up first as a customer's stack half-started.
|
||||
func TestReconstituteNoDBAppNeverStartsServicesOnly(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "run1", "2026-07-19T06:00:00Z", "")
|
||||
prov.composePath = writeLiveCompose(t, noDBCompose)
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return nil, nil }
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err != nil {
|
||||
t.Fatalf("a no-DB app must restore unchanged, got: %v", err)
|
||||
}
|
||||
if len(prov.gotServices) != 0 {
|
||||
t.Fatalf("the DB-only phase ran for an app with no database: %v", prov.gotServices)
|
||||
}
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,start" {
|
||||
t.Fatalf("sequence = %q, want the unchanged stop → full start", got)
|
||||
}
|
||||
if len(*imported) != 0 || res.DBsReplayed != 0 {
|
||||
t.Fatalf("a no-DB app must not replay anything: imported=%v replayed=%d", *imported, res.DBsReplayed)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group C: fail-closed, both paths -------------------------------------------------------------
|
||||
|
||||
// TestReconstituteRefusesWhenNoDBServiceIdentifiable is the security-adjacent gate. A dump exists and
|
||||
// a live database was discovered, but the live compose names no startable database service. The only
|
||||
// alternative to refusing would be to start everything and replay into the H4 race, so this must
|
||||
// refuse — and it must refuse with ZERO mutations, which is what the effect assertions below prove.
|
||||
// Asserting `err != nil` alone would pass even if the app had already been stopped and overwritten.
|
||||
//
|
||||
// COMPANION RED-PROOF: deleting the `len(dbServices) == 0` gate makes this fail on
|
||||
// `the app was stopped despite the refusal`.
|
||||
func TestReconstituteRefusesWhenNoDBServiceIdentifiable(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
|
||||
// A live compose whose services are all app/cache images — nothing to start alone.
|
||||
prov.composePath = writeLiveCompose(t, noDBCompose)
|
||||
var copied bool
|
||||
m.SetOffboxFullPlaceCopier(func(_, _ string) (int, error) { copied = true; return 1, nil })
|
||||
|
||||
_, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err == nil {
|
||||
t.Fatal("expected a refusal: a dump exists but no database service can be started for it")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "nem azonosítható") {
|
||||
t.Fatalf("refusal must say the database service could not be identified, got: %v", err)
|
||||
}
|
||||
if len(prov.calls) != 0 {
|
||||
t.Fatalf("ZERO mutations required, but the provider was called: %v", prov.calls)
|
||||
}
|
||||
if copied {
|
||||
t.Fatal("files were overwritten despite the refusal")
|
||||
}
|
||||
if len(*imported) != 0 {
|
||||
t.Fatalf("a replay happened despite the refusal: %v", *imported)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestoreFromUnitRefusesWhenNoDBServiceIdentifiable is the local path's sibling gate, with the
|
||||
// same zero-mutation requirement: no stop, no volume restore, no definition recreate.
|
||||
func TestRestoreFromUnitRefusesWhenNoDBServiceIdentifiable(t *testing.T) {
|
||||
m, prov, _ := r47UnitFixture(t, noDBCompose, true)
|
||||
|
||||
err := m.RestoreFromRecoveryUnit("app")
|
||||
if err == nil {
|
||||
t.Fatal("expected a refusal: the unit carries a dump but names no startable database service")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "nem azonosítható") {
|
||||
t.Fatalf("refusal must say the database service could not be identified, got: %v", err)
|
||||
}
|
||||
if len(prov.calls) != 0 {
|
||||
t.Fatalf("ZERO mutations required, but the provider was called: %v", prov.calls)
|
||||
}
|
||||
if prov.stopped {
|
||||
t.Fatal("the app was stopped despite the refusal")
|
||||
}
|
||||
if prov.gotEnv != nil {
|
||||
t.Fatal("the definition was recreated despite the refusal")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group D: local restore-from-unit ordering ----------------------------------------------------
|
||||
|
||||
// TestRestoreFromUnitReplaysWithOnlyTheDBServiceUp is Group A's twin on the local path — the SAME
|
||||
// class defect lived here, in the shape `RecreateStackFromUnit` (which ended in a full `up -d`)
|
||||
// followed by the replay. Splitting the persist from the start is what makes this orderable at all.
|
||||
//
|
||||
// COMPANION RED-PROOF: restoring the pre-fix shape (RecreateStackDefinitionFromUnit performing a
|
||||
// full start, replay after) makes this fail on `full stack was ALREADY UP when the replay fired`.
|
||||
func TestRestoreFromUnitReplaysWithOnlyTheDBServiceUp(t *testing.T) {
|
||||
m, prov, imported := r47UnitFixture(t, immichLikeCompose, true)
|
||||
|
||||
var dbUpAtReplay, fullUpAtReplay, definitionPersisted bool
|
||||
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||
dbUpAtReplay = len(prov.gotServices) > 0
|
||||
fullUpAtReplay = prov.fullStarted
|
||||
definitionPersisted = prov.gotEnv != nil
|
||||
*imported = append(*imported, p)
|
||||
return nil
|
||||
}
|
||||
|
||||
if err := m.RestoreFromRecoveryUnit("app"); err != nil {
|
||||
t.Fatalf("restore-from-unit: %v", err)
|
||||
}
|
||||
if len(*imported) != 1 {
|
||||
t.Fatalf("expected exactly one replay, got %v", *imported)
|
||||
}
|
||||
if !definitionPersisted {
|
||||
t.Fatal("the app definition was not persisted before the replay — the DB service could not have been started from it")
|
||||
}
|
||||
if !dbUpAtReplay {
|
||||
t.Fatal("the database service was NOT started before the replay")
|
||||
}
|
||||
if fullUpAtReplay {
|
||||
t.Fatal("the FULL stack was already up when the replay fired — the H4 race, on the local path")
|
||||
}
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,recreate,startsvc:immich-postgres,start" {
|
||||
t.Fatalf("sequence = %q, want stop → recreate(definition only) → db-only start → replay → full start", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestoreFromUnitNoDumpsTakesOneFullStart is the local no-DB negative: without a replayable dump
|
||||
// there is no DB-only window at all, just the definition and one full start.
|
||||
func TestRestoreFromUnitNoDumpsTakesOneFullStart(t *testing.T) {
|
||||
m, prov, imported := r47UnitFixture(t, noDBCompose, false)
|
||||
|
||||
if err := m.RestoreFromRecoveryUnit("app"); err != nil {
|
||||
t.Fatalf("restore-from-unit: %v", err)
|
||||
}
|
||||
if len(prov.gotServices) != 0 {
|
||||
t.Fatalf("the DB-only phase ran with nothing to replay: %v", prov.gotServices)
|
||||
}
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,recreate,start" {
|
||||
t.Fatalf("sequence = %q, want stop → recreate → full start", got)
|
||||
}
|
||||
if len(*imported) != 0 {
|
||||
t.Fatalf("nothing should have been replayed, got %v", *imported)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestoreFromUnitIgnoresSafetyDumpsWhenDecidingToReplay guards the one file-naming trap in the
|
||||
// gate: `pre-restore-*.sql` safety dumps live in the SAME directory as the real dumps (deliberately —
|
||||
// an undo the customer cannot see is not much of one) but are never a replay source. Counting them
|
||||
// would arm the DB-only phase, and its refusal, for an app that has nothing to replay.
|
||||
func TestRestoreFromUnitIgnoresSafetyDumpsWhenDecidingToReplay(t *testing.T) {
|
||||
m, prov, _ := r47UnitFixture(t, noDBCompose, false)
|
||||
// A safety dump present for a DB-less app must not arm anything — including the refusal.
|
||||
mustWrite(t, filepath.Join(AppDBDumpPath(prov.hdd, "app"),
|
||||
preRestoreDumpPrefix+"20260720T101010Z-app-postgres.sql"), pgDump(1))
|
||||
|
||||
if err := m.RestoreFromRecoveryUnit("app"); err != nil {
|
||||
t.Fatalf("a lone safety dump must not turn into a refusal: %v", err)
|
||||
}
|
||||
if len(prov.gotServices) != 0 {
|
||||
t.Fatalf("a safety dump armed the DB-only phase: %v", prov.gotServices)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Group E: a failed replay never strands the box DB-only ---------------------------------------
|
||||
|
||||
// TestReconstituteReplayFailureStillBringsTheStackUp: the DB-only window is a deliberate half-started
|
||||
// state, so EVERY exit from it must end in a full start. Otherwise a failed restore leaves the
|
||||
// customer with a running database and no application — an outage caused by the recovery tool.
|
||||
func TestReconstituteReplayFailureStillBringsTheStackUp(t *testing.T) {
|
||||
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
|
||||
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
|
||||
return context.DeadlineExceeded
|
||||
}
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich")
|
||||
if err == nil {
|
||||
t.Fatal("a failed replay must be surfaced, not swallowed")
|
||||
}
|
||||
if !prov.fullStarted {
|
||||
t.Fatal("the stack was left DB-ONLY after a failed replay — the app is down and nothing will bring it up")
|
||||
}
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,startsvc:immich-postgres,start" {
|
||||
t.Fatalf("sequence = %q, want the best-effort full start after the failure", got)
|
||||
}
|
||||
// The existing message shape stays: the operator needs the undo's filename.
|
||||
if !strings.Contains(err.Error(), filepath.Base(res.SafetyDump)) {
|
||||
t.Fatalf("the error must name the safety dump so the operator can undo, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconstituteDBOnlyStartFailureStillBringsTheStackUp covers the other exit from the window: the
|
||||
// DB-only start itself failing. Same requirement — the app must not be left down.
|
||||
func TestReconstituteDBOnlyStartFailureStillBringsTheStackUp(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
|
||||
prov.startSvcErr = context.DeadlineExceeded
|
||||
|
||||
if _, err := m.ReconstituteFromOffsite(context.Background(), "immich"); err == nil {
|
||||
t.Fatal("a failed DB-only start must be surfaced")
|
||||
}
|
||||
if !prov.fullStarted {
|
||||
t.Fatal("the stack was left down after a failed DB-only start")
|
||||
}
|
||||
if len(*imported) != 0 {
|
||||
t.Fatalf("nothing may be replayed when the database never came up: %v", *imported)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestoreFromUnitReplayFailureStillBringsTheStackUp is the local path's version, and it also
|
||||
// pins the pre-existing semantics: a replay error becomes a dataErr and surfaces as the "completed
|
||||
// with data errors" outcome, with the app back up.
|
||||
func TestRestoreFromUnitReplayFailureStillBringsTheStackUp(t *testing.T) {
|
||||
m, prov, _ := r47UnitFixture(t, immichLikeCompose, true)
|
||||
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
|
||||
return context.DeadlineExceeded
|
||||
}
|
||||
|
||||
err := m.RestoreFromRecoveryUnit("app")
|
||||
if err == nil {
|
||||
t.Fatal("a failed replay must be surfaced, not swallowed")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "completed with data errors") {
|
||||
t.Fatalf("the pre-existing outcome semantics must be preserved, got: %v", err)
|
||||
}
|
||||
if !prov.fullStarted {
|
||||
t.Fatal("the stack was left DB-ONLY after a failed replay")
|
||||
}
|
||||
if got := strings.Join(prov.calls, ","); got != "stop,recreate,startsvc:immich-postgres,start" {
|
||||
t.Fatalf("sequence = %q, want the full start to follow the failed replay", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- fixtures -------------------------------------------------------------------------------------
|
||||
|
||||
// writeLiveCompose drops a compose file in its own temp dir and returns the path.
|
||||
func writeLiveCompose(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "docker-compose.yml")
|
||||
if err := os.WriteFile(p, []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// r47UnitFixture builds a Manager whose local recovery unit for "app" carries the given compose and,
|
||||
// optionally, a replayable `app-postgres.sql` dump. The provider records call order; the DB
|
||||
// discovery/import seams are injected so no Docker is touched.
|
||||
func r47UnitFixture(t *testing.T, compose string, withDump bool) (*Manager, *fakeRecoveryProvider, *[]string) {
|
||||
t.Helper()
|
||||
drive := filepath.Join(t.TempDir(), "drive")
|
||||
composeDir := RecoveryUnitComposePath(drive, "app")
|
||||
mustWrite(t, filepath.Join(composeDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: app\n")
|
||||
mustWrite(t, filepath.Join(composeDir, "docker-compose.yml"), compose)
|
||||
man := &RecoveryManifest{SchemaVersion: 1, AppName: "app", ControllerVer: "v"}
|
||||
if err := writeManifest(RecoveryUnitManifestPath(drive, "app"), man); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if withDump {
|
||||
mustWrite(t, filepath.Join(AppDBDumpPath(drive, "app"), "app-postgres.sql"), pgDump(1))
|
||||
}
|
||||
|
||||
prov := &fakeRecoveryProvider{hdd: drive, running: true}
|
||||
m := &Manager{
|
||||
logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: filepath.Join(drive, "..", "sys"),
|
||||
stackProvider: prov,
|
||||
}
|
||||
|
||||
db := DiscoveredDB{StackName: "app", ContainerName: "immich-postgres", DBType: DBTypePostgres}
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{db}, nil }
|
||||
var imported []string
|
||||
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||
imported = append(imported, p)
|
||||
return nil
|
||||
}
|
||||
return m, prov, &imported
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
@@ -12,22 +13,32 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// RecoveryManifest describes an app's self-contained, SECRET-FREE recovery unit (Phase 2).
|
||||
// RecoveryManifest describes an app's self-contained recovery unit.
|
||||
//
|
||||
// The unit on a drive is `<nsRoot>/backups/primary/<app>/` and contains:
|
||||
// compose/ docker-compose.yml + .felhom.yml + a SECRET-STRIPPED app.yaml
|
||||
// db-dumps/ app-consistent DB dump(s) (written by the dump flow)
|
||||
// volume-dumps/ named-volume tars (written by the dump flow)
|
||||
// manifest.json this file
|
||||
//
|
||||
// The unit holds NO secret values, NO data-encrypting keys, and NOT the Docker image — only the
|
||||
// pinned image tag(s) (re-pulled on restore) and the NAMES of the secret/data-key env vars. The
|
||||
// secret values are recovered at restore time from the guest's own app.yaml (live on the rootfs,
|
||||
// or via the PBS whole-guest snapshot) — see Restore. "Restore from the unit alone" is therefore
|
||||
// honestly "unit + the guest's app.yaml"; SecretSource records that dependency explicitly.
|
||||
// compose/ docker-compose.yml + .felhom.yml + app.yaml (0600; carries the PORTABLE secrets)
|
||||
// db-dumps/ app-consistent DB dump(s) (written by the dump flow)
|
||||
// volume-dumps/ named-volume tars (written by the dump flow)
|
||||
// manifest.json this file
|
||||
//
|
||||
// D5 (schema 2) changed what the unit holds. Before it held NO secret at all, which made
|
||||
// "restore from the drive alone" false: the fast, local, customer-doable Tier-1/2 restore secretly
|
||||
// depended on the slow, operator-driven whole-guest restore, because a data-encrypting key or a DB
|
||||
// password absent from the guest cannot be regenerated without rendering the restored data
|
||||
// unreachable. The unit now carries the PORTABLE secret class (stacks.PortableSecretEnvVars) in its
|
||||
// 0600 app.yaml, and Tier-1/2 needs the DRIVE AND NOTHING ELSE.
|
||||
//
|
||||
// It still holds NO `type: password` admin login (those are internet-reachable, so their blast radius
|
||||
// is not bounded by the drive — they stay in the guest and are regenerated on restore) and NOT the
|
||||
// Docker image, only the pinned tag(s), re-pulled on restore. SecretSource records the split.
|
||||
//
|
||||
// A schema-1 unit carries no secrets: the restore degrades to the pre-D5 guest-only behaviour rather
|
||||
// than failing, and the next capture rewrites it (the app.yaml checksum changes).
|
||||
type RecoveryManifest struct {
|
||||
SchemaVersion int `json:"schema_version"`
|
||||
AppName string `json:"app_name"`
|
||||
@@ -37,13 +48,26 @@ type RecoveryManifest struct {
|
||||
Drive string `json:"drive"` // HDD_PATH (in-guest mount)
|
||||
NamespaceRoot string `json:"namespace_root"` // resolved felhom-data namespace root
|
||||
ImagePins []string `json:"image_pins"` // image NOT stored — re-pulled on restore
|
||||
SecretEnvVars []string `json:"secret_env_vars"` // NAMES only — recovered from guest/PBS
|
||||
SecretEnvVars []string `json:"secret_env_vars"` // NAMES of every secret/password field
|
||||
DataKeyEnvVars []string `json:"data_key_env_vars"` // fail-closed gate on restore
|
||||
SecretSource string `json:"secret_source"` // human note: where secrets come from
|
||||
ConfigFiles []string `json:"config_files"` // captured into compose/
|
||||
DBDumps []string `json:"db_dumps"`
|
||||
VolumeDumps []string `json:"volume_dumps"`
|
||||
Checksums map[string]string `json:"checksums"` // sha256 of captured compose/ files
|
||||
// PortableSecretEnvVars (D5) are the NAMES of the secrets this unit's app.yaml CARRIES. Names only
|
||||
// — the manifest is 0644 and never holds a value. The restore reads it to know which app.yaml env
|
||||
// entries are secrets rather than plain config; absent (schema 1) ⇒ the unit carries none.
|
||||
PortableSecretEnvVars []string `json:"portable_secret_env_vars,omitempty"`
|
||||
// R-43/R-44 (v0.148.0): the coherence stamp. An offsite run refreshes the dumps FIRST and then
|
||||
// captures the unit, so a manifest carrying an OffsiteRunID asserts "the db-dumps/ in this unit
|
||||
// were taken by that run" — i.e. the snapshot is an internally coherent {DB@T, files@T} pair.
|
||||
// A manifest WITHOUT these fields is a pre-v0.148 unit whose dump age is unknown and may skew
|
||||
// arbitrarily from the files beside it (the DIAG-immich-restore-2026-07-19 failure); the restore
|
||||
// confirm surfaces that honestly rather than blocking. Empty on the periodic refresh, which must
|
||||
// never claim a coherence it did not establish — it carries the prior stamp forward instead.
|
||||
OffsiteRunID string `json:"offsite_run_id,omitempty"`
|
||||
DumpsAt string `json:"dumps_at,omitempty"` // RFC3339 UTC — when this run's dump leg finished
|
||||
}
|
||||
|
||||
// SetVersion records the controller version stamped into recovery-unit manifests.
|
||||
@@ -58,9 +82,11 @@ func (m *Manager) SetTier2Notifier(fn func(stackName, destLabel string, dur time
|
||||
m.tier2Notify = fn
|
||||
}
|
||||
|
||||
// CaptureRecoveryUnit writes/refreshes an app's secret-free recovery unit: it captures the
|
||||
// compose + metadata + a secret-stripped app.yaml into compose/, enumerates the DB/volume dumps
|
||||
// already present, and writes manifest.json. It NEVER writes a secret value or the Docker image.
|
||||
// CaptureRecoveryUnit writes/refreshes an app's recovery unit: it captures the compose + metadata +
|
||||
// an app.yaml carrying the PORTABLE secret class (D5) into compose/, enumerates the DB/volume dumps
|
||||
// already present, and writes manifest.json. It never writes the Docker image (only the pinned tag),
|
||||
// and never writes a WITHHELD secret — the split is decided in buildUnitAppYaml, pinned by
|
||||
// TestCaptureRecoveryUnitCarriesPortableSecretsOnly.
|
||||
//
|
||||
// Idempotent: it builds the captured content in memory first and SKIPS all writes when the unit is
|
||||
// already current (same config checksums, same dump set, same controller version) — so it can run on
|
||||
@@ -97,7 +123,7 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
checksums[fname] = sha256Hex(data)
|
||||
configFiles = append(configFiles, fname)
|
||||
}
|
||||
appYaml := buildStrippedAppYaml(info)
|
||||
appYaml := buildUnitAppYaml(info)
|
||||
files = append(files, capFile{"app.yaml", appYaml, 0600})
|
||||
checksums["app.yaml"] = sha256Hex(appYaml)
|
||||
configFiles = append(configFiles, "app.yaml")
|
||||
@@ -107,13 +133,26 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
version := m.versionLocked()
|
||||
|
||||
manifestPath := RecoveryUnitManifestPath(nsRoot, stackName)
|
||||
cur := readManifest(manifestPath)
|
||||
|
||||
// R-43/R-44: the coherence stamp of the offsite run currently in flight ("" on the periodic
|
||||
// refresh and on the local dump run). When empty we CARRY THE PRIOR STAMP FORWARD rather than
|
||||
// blanking it — a periodic refresh must neither claim a coherence it did not establish nor
|
||||
// destroy the record of one that a real run did.
|
||||
runID, dumpsAt := m.offsiteRunStamp()
|
||||
if runID == "" && cur != nil {
|
||||
runID, dumpsAt = cur.OffsiteRunID, cur.DumpsAt
|
||||
}
|
||||
|
||||
// Skip if the unit is already current — avoids needless drive writes on the periodic refresh.
|
||||
if cur := readManifest(manifestPath); cur != nil &&
|
||||
// The run-id is part of "current": an offsite run must re-stamp the manifest even when nothing
|
||||
// else changed, because the stamp is exactly the claim the restore path reads.
|
||||
if cur != nil &&
|
||||
cur.ControllerVer == version &&
|
||||
stringMapEqual(cur.Checksums, checksums) &&
|
||||
stringSliceEqual(cur.DBDumps, dbDumps) &&
|
||||
stringSliceEqual(cur.VolumeDumps, volDumps) {
|
||||
stringSliceEqual(cur.VolumeDumps, volDumps) &&
|
||||
cur.OffsiteRunID == runID {
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -128,33 +167,181 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
}
|
||||
|
||||
manifest := &RecoveryManifest{
|
||||
SchemaVersion: 1,
|
||||
AppName: stackName,
|
||||
DisplayName: info.DisplayName,
|
||||
ControllerVer: version,
|
||||
CreatedAt: time.Now().UTC().Format(time.RFC3339),
|
||||
Drive: drivePath,
|
||||
NamespaceRoot: nsRoot,
|
||||
ImagePins: info.ImagePins,
|
||||
SecretEnvVars: info.SecretEnvVars,
|
||||
DataKeyEnvVars: info.DataKeyEnvVars,
|
||||
SecretSource: "guest app.yaml (live rootfs) or PBS whole-guest snapshot — never stored in this unit",
|
||||
ConfigFiles: configFiles,
|
||||
DBDumps: dbDumps,
|
||||
VolumeDumps: volDumps,
|
||||
Checksums: checksums,
|
||||
SchemaVersion: 2, // D5: compose/app.yaml carries the portable secret class
|
||||
AppName: stackName,
|
||||
DisplayName: info.DisplayName,
|
||||
ControllerVer: version,
|
||||
CreatedAt: time.Now().UTC().Format(time.RFC3339),
|
||||
Drive: drivePath,
|
||||
NamespaceRoot: nsRoot,
|
||||
ImagePins: info.ImagePins,
|
||||
SecretEnvVars: info.SecretEnvVars,
|
||||
DataKeyEnvVars: info.DataKeyEnvVars,
|
||||
PortableSecretEnvVars: info.PortableSecretEnvVars,
|
||||
SecretSource: "portable secrets (data keys, DB passwords, internal signing secrets) are IN this unit's compose/app.yaml (0600); internet-reachable admin logins are NOT, and come from the guest's app.yaml or are regenerated on restore",
|
||||
ConfigFiles: configFiles,
|
||||
DBDumps: dbDumps,
|
||||
VolumeDumps: volDumps,
|
||||
Checksums: checksums,
|
||||
OffsiteRunID: runID,
|
||||
DumpsAt: dumpsAt,
|
||||
}
|
||||
if err := writeManifest(manifestPath, manifest); err != nil {
|
||||
return fmt.Errorf("writing manifest: %w", err)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [backup] Recovery unit captured for %s → %s (images=%d, secrets-referenced=%d, data_keys=%d)",
|
||||
stackName, RecoveryUnitPath(nsRoot, stackName), len(info.ImagePins), len(info.SecretEnvVars), len(info.DataKeyEnvVars))
|
||||
// Counts and NAMES only — never a value (D5 puts more secrets through this path than before).
|
||||
m.logger.Printf("[INFO] [backup] Recovery unit captured for %s → %s (images=%d, secrets-referenced=%d, data_keys=%d, portable-carried=%d/%d, withheld=%d)",
|
||||
stackName, RecoveryUnitPath(nsRoot, stackName), len(info.ImagePins), len(info.SecretEnvVars),
|
||||
len(info.DataKeyEnvVars), len(info.PortableSecrets), len(info.PortableSecretEnvVars),
|
||||
len(withheldSecretNames(info)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// UnitSpace is the target filesystem's occupancy at the moment a capture failed — the numbers that
|
||||
// answer "why" without an operator logging in. Nil when the filesystem could not be read at all
|
||||
// (system.GetDiskUsage returns nil on error), which is reported as unknown rather than as full.
|
||||
type UnitSpace struct {
|
||||
Path string
|
||||
UsedGB float64
|
||||
AvailGB float64
|
||||
TotalGB float64
|
||||
UsedPercent float64
|
||||
}
|
||||
|
||||
// String renders the space figures for an operator, or says plainly that they are unknown. An absent
|
||||
// reading must never render as zeros — "0 GB free" and "we could not look" are opposite diagnoses.
|
||||
func (u *UnitSpace) String() string {
|
||||
if u == nil {
|
||||
return "target filesystem usage unavailable"
|
||||
}
|
||||
return fmt.Sprintf("%s: %.1f/%.1f GB used (%.0f%%), %.1f GB free",
|
||||
u.Path, u.UsedGB, u.TotalGB, u.UsedPercent, u.AvailGB)
|
||||
}
|
||||
|
||||
// SetUnitNotify wires the per-app recovery-unit capture failure alert (R-158 / R-167). INIT-ONLY —
|
||||
// call once at startup, in main.go, alongside SetOffboxNotify. Nil-safe: an unwired seam is silently
|
||||
// the pre-v0.191.0 behaviour, which is a `[WARN]` line and nothing else.
|
||||
func (m *Manager) SetUnitNotify(fn func(stackName string, err error, usage *UnitSpace)) {
|
||||
m.unitNotify = fn
|
||||
}
|
||||
|
||||
// unitTargetSpace reads the occupancy of the filesystem a unit for `stackName` would be written to.
|
||||
// Nil on an unreadable path — never a fabricated zero (§8.4: an unreadable filesystem is not a full
|
||||
// one, and the drive gate already owns the absent-drive case).
|
||||
func (m *Manager) unitTargetSpace(stackName string) *UnitSpace {
|
||||
path := m.GetAppDrivePath(stackName)
|
||||
if path == "" {
|
||||
return nil
|
||||
}
|
||||
di := system.GetDiskUsage(path)
|
||||
if di == nil {
|
||||
return nil
|
||||
}
|
||||
return &UnitSpace{
|
||||
Path: path, UsedGB: di.UsedGB, AvailGB: di.AvailGB,
|
||||
TotalGB: di.TotalGB, UsedPercent: di.UsedPercent,
|
||||
}
|
||||
}
|
||||
|
||||
// ── The capture floor (R-165 / decision B2) ──────────────────────────────────────────────────────
|
||||
//
|
||||
// WHAT IT REPLACES. Until the `mp1`→`mp0` merge, the 20 G backup partition was a BULKHEAD as well as
|
||||
// a ceiling: an app whose unit outgrew it was refused per app, its last good unit preserved
|
||||
// byte-identical, and the overflow **could not reach `/var/lib/docker`** because that was a different
|
||||
// filesystem. After the merge it can, and a full Docker data-root is a stopped box, not a slow one.
|
||||
// This floor is that bulkhead, done deliberately instead of by accident.
|
||||
//
|
||||
// IT IS ABOUT THE FILESYSTEM'S HEADROOM, NEVER THE UNIT'S SIZE. A per-unit size cap would be R-163
|
||||
// rebuilt inside one volume — the wall moved rather than removed — so a large unit on a filesystem
|
||||
// with ample room is captured, whatever its size.
|
||||
//
|
||||
// IT REFUSES; IT NEVER DELETES. Nothing on this filesystem is generational: a unit is ONE fixed path
|
||||
// per app (`backups/primary/<app>`) refreshed in place, and a DB dump is `<stack>-<dbtype>.sql`, also
|
||||
// fixed. So "prune the oldest" could only mean deleting a DIFFERENT app's only local recovery unit to
|
||||
// make room for this one, and that is not a trade this system makes. `pruneStalePrimaryDirs` is NOT a
|
||||
// retention policy — it removes ORPHANED directories left when an app moves drives, and has no notion
|
||||
// of age — so it must never be repurposed here.
|
||||
const (
|
||||
// FloorUsedPercent / FloorFreeGiB — the reserve. Two terms, whichever binds first, the same shape
|
||||
// as `internal/fillwatch` (proven live on 2026-08-02: the critical alert fired on the free-byte
|
||||
// term at 91% used, where a percent-only rule stayed silent).
|
||||
//
|
||||
// THEY SIT DELIBERATELY BEYOND fillwatch's CRITICAL BAND (95% / 2 GiB), so the customer is ALWAYS
|
||||
// warned before a refusal can happen. A floor that fires before its own warning is a silent
|
||||
// failure wearing a threshold; `TestFloorSitsBelowTheCriticalWarningBand` pins the ordering.
|
||||
//
|
||||
// 1 GiB is the reserve, not a working budget: §7.5 measures a DB-backed app's unit at up to ~2× its
|
||||
// data, so no fixed number can guarantee a capture fits. What this guarantees is different and is
|
||||
// the bulkhead's actual job — that a capture cannot consume the last of the space the container
|
||||
// runtime needs to keep running.
|
||||
FloorUsedPercent = 97.0
|
||||
FloorFreeGiB = 1.0
|
||||
)
|
||||
|
||||
// ErrCaptureFloor marks an app's backup refused for headroom. It is a REFUSAL, not a failure of the
|
||||
// backup machinery — the distinction matters to a reader of the alert, which is why the message names
|
||||
// the reserve rather than reporting an I/O error.
|
||||
//
|
||||
// R-181 widened what it covers: it now refuses the app's DB dump, volume dump and capture together
|
||||
// (see admission.go), not the capture alone. The sentinel keeps its name because callers match it and
|
||||
// "capture" still reads correctly for "capturing this app's backup"; the MESSAGE is what changed, and
|
||||
// the message is what an operator sees.
|
||||
var ErrCaptureFloor = errors.New("refused: backing up this app would leave the filesystem below the reserve")
|
||||
|
||||
// floorVerdict is the PURE predicate: given a reading and this app's estimated write, does the floor
|
||||
// refuse, and on which term? Separated so the thresholds are unit-testable without a filesystem, a
|
||||
// stack provider or a clock.
|
||||
//
|
||||
// TWO QUESTIONS, NOT ONE (R-181). "Is the filesystem already below the reserve?" is the headroom term
|
||||
// and was all B2 asked. "Would THIS app's write take it below?" is the size term, and its absence is
|
||||
// how an app was admitted at 96% used and then allowed to write 2 GB. Both terms are evaluated
|
||||
// against BOTH thresholds — a large write can cross the percentage bound on a small volume and the
|
||||
// free-byte bound on a large one, which is the same reason the reserve has two terms at all.
|
||||
//
|
||||
// §8.4 — A NIL READING NEITHER REFUSES NOR WARNS. An unreadable filesystem is the drive gate's
|
||||
// business and has its own alert; refusing on it would block every backup on a box whose drive merely
|
||||
// blipped, and warning on it would be a false alarm with a misleading cause.
|
||||
//
|
||||
// estGiB == 0 (no previous dump to estimate from) degrades to the headroom term alone, deliberately:
|
||||
// refusing an app that has never been backed up would make the FIRST backup the one that can never
|
||||
// happen (Scenario E).
|
||||
func (m *Manager) floorVerdict(u *UnitSpace, estGiB float64) (*UnitSpace, floorReason) {
|
||||
if u == nil {
|
||||
return nil, floorAdmit
|
||||
}
|
||||
if u.UsedPercent >= FloorUsedPercent || u.AvailGB < FloorFreeGiB {
|
||||
return u, floorHeadroom
|
||||
}
|
||||
if estGiB > 0 {
|
||||
availAfter := u.AvailGB - estGiB
|
||||
usedAfter := u.UsedPercent
|
||||
if u.TotalGB > 0 {
|
||||
usedAfter = (u.UsedGB + estGiB) / u.TotalGB * 100
|
||||
}
|
||||
if availAfter < FloorFreeGiB || usedAfter >= FloorUsedPercent {
|
||||
return u, floorSize
|
||||
}
|
||||
}
|
||||
return u, floorAdmit
|
||||
}
|
||||
|
||||
// readUnitSpace goes through the seam when one is injected, so a test can state the filesystem's
|
||||
// occupancy as an input instead of manufacturing it on a real disk. Nil seam → the real statfs.
|
||||
func (m *Manager) readUnitSpace(stackName string) *UnitSpace {
|
||||
if m.unitSpaceFn != nil {
|
||||
return m.unitSpaceFn(stackName)
|
||||
}
|
||||
return m.unitTargetSpace(stackName)
|
||||
}
|
||||
|
||||
// captureAllRecoveryUnits refreshes the recovery unit for every deployed stack. Best-effort:
|
||||
// a per-app failure is logged and does not abort the others.
|
||||
// a per-app failure is logged, NOTIFIED (R-158), and does not abort the others.
|
||||
//
|
||||
// R-181: the reserve is consulted through `admitApp`, which is the SAME verdict the DB-dump and
|
||||
// volume-dump legs of this run already consulted for this app. When a run is in flight the answer
|
||||
// here is a memo lookup — an app refused before its first write is refused here too, silently,
|
||||
// because it was already alerted once. Outside a run (the periodic status refresh) it decides fresh.
|
||||
func (m *Manager) captureAllRecoveryUnits() {
|
||||
if m.stackProvider == nil {
|
||||
return
|
||||
@@ -164,8 +351,20 @@ func (m *Manager) captureAllRecoveryUnits() {
|
||||
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
|
||||
continue // drive not writable — skip, the existing unit stays as-is
|
||||
}
|
||||
m.noteAttempted(stack.Name)
|
||||
// The reserve, checked BEFORE anything is written. Per app, and the loop continues.
|
||||
if !m.admitApp(stack.Name) {
|
||||
continue
|
||||
}
|
||||
if err := m.CaptureRecoveryUnit(stack.Name); err != nil {
|
||||
m.noteFailure(stack.Name, "recovery-unit capture", err.Error())
|
||||
m.logger.Printf("[WARN] [backup] Recovery unit capture failed for %s: %v", stack.Name, err)
|
||||
// R-158: per app, and the loop CONTINUES — one app's failure must not silence the
|
||||
// others, and it must not abort their captures either. The space figures are read at
|
||||
// the moment of failure, because the point is to answer "why" (usually: no room).
|
||||
if m.unitNotify != nil {
|
||||
m.unitNotify(stack.Name, err, m.unitTargetSpace(stack.Name))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -176,29 +375,63 @@ func (m *Manager) versionLocked() string {
|
||||
return m.version
|
||||
}
|
||||
|
||||
// strippedAppYaml is the on-disk shape of the secret-free app.yaml captured into the unit.
|
||||
// strippedAppYaml is the on-disk shape of the app.yaml captured into the unit. The name is historical:
|
||||
// since D5 the `env` map carries the PORTABLE secrets alongside the plain config (see buildUnitAppYaml).
|
||||
type strippedAppYaml struct {
|
||||
Deployed bool `yaml:"deployed"`
|
||||
Env map[string]string `yaml:"env"`
|
||||
}
|
||||
|
||||
// buildStrippedAppYaml renders a secret-free app.yaml (non-secret env only) as bytes. Deterministic:
|
||||
// yaml.v3 sorts map keys and the secret-name list comes in stable metadata order, so identical input
|
||||
// yields identical bytes (needed for the checksum-skip guard).
|
||||
func buildStrippedAppYaml(info RecoveryInfo) []byte {
|
||||
body, err := yaml.Marshal(strippedAppYaml{Deployed: true, Env: info.NonSecretEnv})
|
||||
// buildUnitAppYaml renders the unit's app.yaml as bytes: the non-secret env PLUS the portable secret
|
||||
// values (D5). Deterministic: yaml.v3 sorts map keys and the name lists come in stable metadata order,
|
||||
// so identical input yields identical bytes (needed for the checksum-skip guard).
|
||||
//
|
||||
// This is the ONE place the capture side decides what does and does not reach the drive — there is no
|
||||
// second path that writes a unit app.yaml. The caller writes the result 0600.
|
||||
func buildUnitAppYaml(info RecoveryInfo) []byte {
|
||||
env := make(map[string]string, len(info.NonSecretEnv)+len(info.PortableSecrets))
|
||||
for k, v := range info.NonSecretEnv {
|
||||
env[k] = v
|
||||
}
|
||||
// Portable secrets last: NonSecretEnv is disjoint from the secret set by construction
|
||||
// (GetStackRecoveryInfo), so this cannot shadow a plain config value.
|
||||
for k, v := range info.PortableSecrets {
|
||||
env[k] = v
|
||||
}
|
||||
body, err := yaml.Marshal(strippedAppYaml{Deployed: true, Env: env})
|
||||
if err != nil {
|
||||
body = []byte("deployed: true\nenv: {}\n")
|
||||
}
|
||||
header := "# Captured by felhom-controller recovery unit — SECRET-FREE.\n" +
|
||||
"# Secret/data-key values are intentionally omitted; recover them at restore from the\n" +
|
||||
"# guest's own app.yaml (live rootfs, or the PBS whole-guest snapshot). Stripped names:\n"
|
||||
if len(info.SecretEnvVars) > 0 {
|
||||
header += "# " + strings.Join(info.SecretEnvVars, ", ") + "\n"
|
||||
header := "# Captured by felhom-controller recovery unit.\n" +
|
||||
"# This file CARRIES SECRETS (D5) so a Tier-1/2 restore needs the drive and nothing else:\n" +
|
||||
"# data-encrypting keys, database passwords and internal signing secrets. Mode 0600.\n"
|
||||
if len(info.PortableSecretEnvVars) > 0 {
|
||||
header += "# Carried: " + strings.Join(info.PortableSecretEnvVars, ", ") + "\n"
|
||||
}
|
||||
// The withheld class is named, not valued — an operator reading the unit must be able to see WHY a
|
||||
// credential is missing rather than suspecting a capture bug.
|
||||
if withheld := withheldSecretNames(info); len(withheld) > 0 {
|
||||
header += "# WITHHELD (internet-reachable logins — stay in the guest, regenerated on restore): " +
|
||||
strings.Join(withheld, ", ") + "\n"
|
||||
}
|
||||
return []byte(header + string(body))
|
||||
}
|
||||
|
||||
// withheldSecretNames returns the secret names deliberately NOT carried by the unit, in stable order.
|
||||
func withheldSecretNames(info RecoveryInfo) []string {
|
||||
portable := make(map[string]bool, len(info.PortableSecretEnvVars))
|
||||
for _, n := range info.PortableSecretEnvVars {
|
||||
portable[n] = true
|
||||
}
|
||||
var out []string
|
||||
for _, n := range info.SecretEnvVars {
|
||||
if !portable[n] {
|
||||
out = append(out, n)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// writeManifest writes the manifest JSON atomically.
|
||||
func writeManifest(dst string, manifest *RecoveryManifest) error {
|
||||
data, err := json.MarshalIndent(manifest, "", " ")
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// R-158 / R-167 (D-c, operator half) — a per-app Tier-1 recovery-unit capture failure must reach a
|
||||
// hub channel.
|
||||
//
|
||||
// THE GAP THESE CLOSE. `captureAllRecoveryUnits` logged `[WARN] Recovery unit capture failed for %s`
|
||||
// and stopped there. The manager carried three notify seams — tier2Notify, offboxNotify,
|
||||
// offboxEnlargeBlockedNotify — and none for the unit capture, so the one page a person opens to ask
|
||||
// whether ONE app is backed up (`/backups/apps`) was the one page that never said. Fifth instance in
|
||||
// this project of a mechanism built and left disconnected.
|
||||
|
||||
// unitFailProvider lists a fixed set of stacks and refuses GetStackRecoveryInfo for the named ones,
|
||||
// which is the earliest real failure inside CaptureRecoveryUnit ("stack %q not found").
|
||||
type unitFailProvider struct {
|
||||
stacks []string
|
||||
fail map[string]bool
|
||||
dir string
|
||||
}
|
||||
|
||||
func (p *unitFailProvider) GetStackComposePath(string) (string, bool) { return "", false }
|
||||
func (p *unitFailProvider) ListDeployedStacks() []StackSummary {
|
||||
out := make([]StackSummary, 0, len(p.stacks))
|
||||
for _, s := range p.stacks {
|
||||
out = append(out, StackSummary{Name: s})
|
||||
}
|
||||
return out
|
||||
}
|
||||
func (p *unitFailProvider) GetStackHDDMounts(string) []string { return nil }
|
||||
func (p *unitFailProvider) GetStackHDDPath(string) string { return "" }
|
||||
func (p *unitFailProvider) GetImportRoot() string { return "" }
|
||||
func (p *unitFailProvider) GetDockerVolumes(string) []string { return nil }
|
||||
func (p *unitFailProvider) StopStack(string) error { return nil }
|
||||
func (p *unitFailProvider) StartStack(string) error { return nil }
|
||||
func (p *unitFailProvider) RefreshAndIsRunning(string) bool { return true }
|
||||
func (p *unitFailProvider) GetStackRecoveryInfo(name string) (RecoveryInfo, bool) {
|
||||
if p.fail[name] {
|
||||
return RecoveryInfo{}, false
|
||||
}
|
||||
return RecoveryInfo{StackDir: filepath.Join(p.dir, "stacks", name)}, true
|
||||
}
|
||||
func (p *unitFailProvider) RecoverStackSecrets(string, []string) map[string]string { return nil }
|
||||
func (p *unitFailProvider) RecreateStackDefinitionFromUnit(string, string, map[string]string) error {
|
||||
return nil
|
||||
}
|
||||
func (p *unitFailProvider) StartStackServices(string, []string) error { return nil }
|
||||
func (p *unitFailProvider) GetStackClassifiedBinds(string) ([]appbackup.ClassifiedBind, bool) {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
var _ appbackup.StackDataProvider = (*unitFailProvider)(nil)
|
||||
|
||||
type unitEvent struct {
|
||||
app string
|
||||
err string
|
||||
usage *UnitSpace
|
||||
}
|
||||
|
||||
func newUnitNotifyManager(t *testing.T, stacks []string, fail map[string]bool) (*Manager, *[]unitEvent) {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
m := &Manager{
|
||||
logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: dir,
|
||||
stackProvider: &unitFailProvider{stacks: stacks, fail: fail, dir: dir},
|
||||
}
|
||||
var got []unitEvent
|
||||
m.SetUnitNotify(func(name string, err error, usage *UnitSpace) {
|
||||
got = append(got, unitEvent{app: name, err: err.Error(), usage: usage})
|
||||
})
|
||||
return m, &got
|
||||
}
|
||||
|
||||
// --- Scenario C — a local unit capture failure reaches the operator ------------------------------
|
||||
|
||||
func TestCaptureAll_FailureNotifiesOnceWithTheSpaceFigures(t *testing.T) {
|
||||
m, got := newUnitNotifyManager(t, []string{"immich"}, map[string]bool{"immich": true})
|
||||
|
||||
m.captureAllRecoveryUnits()
|
||||
|
||||
if len(*got) != 1 {
|
||||
t.Fatalf("got %d unit-failure events, want exactly 1 — a per-app Tier-1 capture failure "+
|
||||
"reached no hub channel, which is the R-158 gap un-fixed", len(*got))
|
||||
}
|
||||
e := (*got)[0]
|
||||
if e.app != "immich" {
|
||||
t.Fatalf("event names app %q, want immich — an operator cannot act on an unnamed app", e.app)
|
||||
}
|
||||
if e.err == "" {
|
||||
t.Fatal("the event carries no error — the operator is told a capture failed but not why")
|
||||
}
|
||||
// The space figures are the point: the overwhelmingly likely cause is a full filesystem, and
|
||||
// these answer "why" without an operator logging in.
|
||||
if e.usage == nil {
|
||||
t.Fatal("the event carries no space figures for a readable target filesystem — this is the " +
|
||||
"pair of numbers that makes the alert actionable, and the same pair the customer fill " +
|
||||
"warning reports (which is why the two ship together)")
|
||||
}
|
||||
if e.usage.Path == "" || e.usage.TotalGB <= 0 {
|
||||
t.Fatalf("space figures are not populated: %+v", e.usage)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Scenario D — one failing app does not silence the others ------------------------------------
|
||||
|
||||
func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
|
||||
m, got := newUnitNotifyManager(t,
|
||||
[]string{"homebox", "immich", "nextcloud"},
|
||||
map[string]bool{"immich": true})
|
||||
|
||||
m.captureAllRecoveryUnits()
|
||||
|
||||
if len(*got) != 1 {
|
||||
t.Fatalf("got %d events, want exactly 1 — either the loop ABORTED on the middle app "+
|
||||
"(and its siblings were never captured), or one failure produced several events: %+v",
|
||||
len(*got), *got)
|
||||
}
|
||||
if (*got)[0].app != "immich" {
|
||||
t.Fatalf("event names %q, want immich", (*got)[0].app)
|
||||
}
|
||||
|
||||
// The siblings must have been ATTEMPTED after the failure — a positive observable, not the
|
||||
// absence of an event. The provider records nothing, so assert via the failure set instead:
|
||||
// flip the LAST app to failing and require both events.
|
||||
m2, got2 := newUnitNotifyManager(t,
|
||||
[]string{"homebox", "immich", "nextcloud"},
|
||||
map[string]bool{"immich": true, "nextcloud": true})
|
||||
m2.captureAllRecoveryUnits()
|
||||
if len(*got2) != 2 {
|
||||
t.Fatalf("got %d events, want 2 — the app AFTER the first failure was never reached, so the "+
|
||||
"loop is aborting rather than continuing: %+v", len(*got2), *got2)
|
||||
}
|
||||
if (*got2)[0].app != "immich" || (*got2)[1].app != "nextcloud" {
|
||||
t.Fatalf("events %+v, want immich then nextcloud in loop order", *got2)
|
||||
}
|
||||
}
|
||||
|
||||
// A successful capture must be SILENT. An alert that fires on success is an alert an operator learns
|
||||
// to ignore.
|
||||
func TestCaptureAll_SuccessIsSilent(t *testing.T) {
|
||||
m, got := newUnitNotifyManager(t, []string{"homebox"}, nil)
|
||||
m.captureAllRecoveryUnits()
|
||||
if len(*got) != 0 {
|
||||
t.Fatalf("a successful capture fired %d event(s): %+v", len(*got), *got)
|
||||
}
|
||||
}
|
||||
|
||||
// The seam must be nil-safe: an unwired notify is the pre-v0.191.0 behaviour (a WARN line), never a
|
||||
// panic that takes the whole nightly backup down with it.
|
||||
func TestCaptureAll_UnwiredNotifyDoesNotPanic(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
m := &Manager{
|
||||
logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: dir,
|
||||
stackProvider: &unitFailProvider{stacks: []string{"immich"}, fail: map[string]bool{"immich": true}, dir: dir},
|
||||
}
|
||||
m.captureAllRecoveryUnits() // no SetUnitNotify — must not panic
|
||||
}
|
||||
|
||||
// §8.4 in the failure direction: an unreadable target filesystem is reported as UNKNOWN, never as
|
||||
// zeros. "0 GB free" and "we could not look" are opposite diagnoses, and rendering the second as the
|
||||
// first is the presence-is-not-success trap pointing the other way.
|
||||
func TestUnitSpace_NilRendersAsUnavailableNotZero(t *testing.T) {
|
||||
var u *UnitSpace
|
||||
s := u.String()
|
||||
if !strings.Contains(s, "unavailable") {
|
||||
t.Fatalf("nil UnitSpace renders as %q — it must say the reading is unavailable", s)
|
||||
}
|
||||
if strings.Contains(s, "0.0") {
|
||||
t.Fatalf("nil UnitSpace renders zeros (%q) — an operator would read \"the disk is full\" "+
|
||||
"from a filesystem nobody could read", s)
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user