From 0d402f711ddf9f53ff4b395166edaeba19fb70c6 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Sun, 13 Sep 2026 11:41:31 +0200 Subject: [PATCH] =?UTF-8?q?v0.237.0:=20the=20Update=20button=20takes=20a?= =?UTF-8?q?=20backup=20first,=20and=20tells=20the=20truth=20(update=20arc?= =?UTF-8?q?=20slice=204=20=E2=80=94=20R-448,=20R-443,=20R-439)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit POST /api/stacks/{name}/update is now a guarded job answering 202: cheap refusals (hold — R-439, busy, migration, deploying, memory via the deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable Tier-2 copy) → backup-first when the proven copy is older than update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin → pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s settle, update.health_timeout 5m). Not healthy → the app is stopped and HELD (RestoreHold reason update_failed, same store and gate as R-379) and the page names the backup to restore from; the pin stays. Success is only ever update_phase=done after health (R-443). UpdateStack is deleted. The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint and shared with the backups page (row pinned unchanged). The copy is aged by the last successful Tier-2 copy, not the manifest created_at — measured on demo-hp that created_at moves only on definition changes. Crash safety: update-journal.json before each phase; RecoverUpdates before the boot sweep, ResumeInterruptedUpdates after the guards are wired. Three unattended start paths ignored a hold and now honour it: the drive-return gate (restart + boot recreate) and the nightly volume dump. The nightly capture and Tier-2 run skip held apps so the restore point survives. No automatic rollback — measured per-app; route back = restore. Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail. Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- CHANGELOG.md | 65 ++ CONTEXT.md | 12 +- REUSE.md | 9 +- controller/README.md | 52 +- controller/cmd/controller/main.go | 94 +- .../cmd/controller/slice4_wiring_test.go | 104 +++ controller/internal/api/router.go | 37 +- controller/internal/api/slice4_update_test.go | 165 ++++ controller/internal/backup/backup.go | 9 + .../internal/backup/offbox_reconstitute.go | 9 + controller/internal/backup/recovery_unit.go | 8 + controller/internal/backup/restore_unit.go | 3 + .../backup/slice4_update_guard_test.go | 168 ++++ controller/internal/backup/tier2.go | 6 + controller/internal/backup/update_guard.go | 325 +++++++ controller/internal/config/config.go | 46 + controller/internal/settings/settings.go | 19 + controller/internal/stacks/deploy.go | 112 ++- controller/internal/stacks/installed_test.go | 6 +- controller/internal/stacks/manager.go | 105 +-- controller/internal/stacks/update.go | 802 ++++++++++++++++++ controller/internal/stacks/update_test.go | 569 +++++++++++++ controller/internal/web/handlers.go | 11 +- controller/internal/web/intermediary.go | 22 + controller/internal/web/slice4_update_test.go | 74 ++ 25 files changed, 2709 insertions(+), 123 deletions(-) create mode 100644 controller/cmd/controller/slice4_wiring_test.go create mode 100644 controller/internal/api/slice4_update_test.go create mode 100644 controller/internal/backup/slice4_update_guard_test.go create mode 100644 controller/internal/backup/update_guard.go create mode 100644 controller/internal/stacks/update.go create mode 100644 controller/internal/stacks/update_test.go create mode 100644 controller/internal/web/slice4_update_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index da2c8f3..654430f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,68 @@ +## v0.237.0 — the Update button takes a backup first, and tells the truth (2026-09-13, update arc slice 4 — R-448, R-443, R-439) + +**MinAgent: 0.129.0** (unchanged) + +**What it replaced.** `UpdateStack` advanced the pin, pulled, ran `up -d` and returned. No copy first, +no check of memory, disk, a running backup or a held app, and success the moment `up` returned — +measured in `SPIKE-app-update-2026-09-01` §4 as **HTTP 200 over an app that was already +crash-looping** (R-443). A held app could be updated at all (R-439). **`UpdateStack` is deleted**; its +only caller was the API. + +**The guarded update** (`internal/stacks/update.go`), a job answering **202** at once, with phases +`checking → backing-up (only if stale) → safety-dump → pinning → pulling → starting → verifying → +done | failed` on `GET /api/stacks/{name}` (`updating`, `update_phase`, `update_phase_label`, +`update_error`, `hold_reason`): + +1. **Cheap refusals first, each a 409 with a Hungarian sentence, before the intent is recorded:** held + (R-439 — `update` joins `start`/`restart` in the router's hold check), a backup/restore/app-data op + or quiesce holding it, a migration, already updating, deploying, memory (the deploy's own check, + extracted as `memoryVerdict`, releasing the app's current request), disk (**fixed 2 GB floor** on the + Docker data root — the image size is not known without a registry query). +2. **The precondition is the existing verified backup** (operator ruling 2026-09-02): an openable + Tier-2 recovery unit with a PROVEN copy date — `backup.Tier2UnitRestorePoint`, **extracted from the + backups page handler, not duplicated** (the page renders identically, pinned). No such copy → + refused. Copy older than `update.backup_max_age` (default `24h`) → `RunAppBackupNow` runs this app's + own legs (DB dump → volume dump → unit capture → Tier-2) first; it must succeed AND yield a fresh + restorable copy, or nothing moves. +3. **Safety dump before the pin moves** (`WriteUpdateSafetyDump` = R-361's `writeSafetyDump`). +4. **Pin before pull**; a **pull failure puts the pin and definition BACK** (nothing ran). +5. **Health, not the exit code:** the app's `.felhom.yml` check through the existing probe, or — with + none declared — every container running, none restarting, for 60 s; bounded by + `update.health_timeout` (default `5m`). Not healthy → **the app is stopped and HELD** + (`settings.RestoreHold`, new `reason: update_failed` + `copy_date`, same store and same gate as + R-379), the pin stays on the new version, and the page carries the hold sentence naming the backup + to restore from. A successful unit restore lifts an update hold (only that kind). + +**Measured before it was designed, and the design changed because of it.** The copy's age is the last +SUCCESSFUL Tier-2 copy, not the unit manifest's `created_at`: on demo-hp bookstack's mirror held a +database dump from 2026-09-13T00:30Z under a manifest dated 2026-09-12T02:15:29Z, because a capture +rewrites the manifest only when the app's DEFINITION changes. Aged by the manifest, a quiet app would +be "stale" forever and "back up first" would not fix it. + +**Crash safety is a journal** (`/update-journal.json`, fsynced, written before each phase). +`RecoverUpdates` runs before the boot sweep: interrupted before the pin → dropped; while pinning or +pulling → pin put back; after `up` → the app is marked Updating (the boot sweep and the dead-app alarm +leave it alone) and `ResumeInterruptedUpdates` re-runs `up` + the health wait once the backup side is +wired, ending healthy or held. + +**"A hold that only one path honours is not a hold" — three unattended paths did not honour one, and +now do:** the drive-return gate (`intermediary.go` restart + boot recreate), and the nightly volume +dump (`DumpAppVolumesSafe` ends in `StartStack`). The nightly capture and Tier-2 run also skip a held +app, so they cannot overwrite the restore point the hold text names. + +**Deliberately NOT built:** putting the old version back automatically. Measured per-app +(`SPIKE-upgrade-test-2026-09-06`) and unpredictable; the route back is the restore. The multi-major +jump is not gated here (Slice 6); it fails health and is held honestly. + +**Tests.** stacks: A–G (success only after health, stale → backup first, backup failure, no unit, +seven cheap refusals, pull failure pin-back, health failure hold, unsaved hold, three recovery +shapes), hold text on GetStacks, the exact phase labels. backup: proven copy time (incl. the measured +bookstack case), hold text, restore clears only update holds, busy, nightly legs skip held apps. api: +R-439, R-443 (202, never "completed"), no-backup 409 records no intent. web: backup row unchanged by the +extraction, drive-return skips a held app. cmd: guards wired, recovery before the boot sweep, resume +after the guards, boot gate + dead-app alarm know about updates. **Six companion red-proofs, each seen +to fail** — three first ran INERT or did not compile and were fixed before they counted (REPORT.md). + ## v0.236.0 — "delete my data too" deletes the data, or says that it could not (2026-09-13, R-442) **The defect.** A customer removes an app and ticks *also delete my data*. The box says it worked; diff --git a/CONTEXT.md b/CONTEXT.md index 68777ef..16eb32a 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -7,7 +7,17 @@ > > Ask Claude Code: "Please update CONTEXT.md with what we did today" -Last updated: 2026-09-13 (v0.236.0 — R-442: "delete my data too" deletes it or refuses) +Last updated: 2026-09-13 (v0.237.0 — update arc slice 4: the guarded update) + +> **2026-09-13 — v0.237.0 (slice 4: R-448, R-443, R-439).** Update is a guarded 202 job: +> refusals (hold, busy, memory, disk, no restorable Tier-2 copy) → backup-first if the proven copy is +> older than `update.backup_max_age` (24h) → safety dump → pin → pull (failure: pin BACK) → up → +> health (`.felhom.yml` check or 60 s settle, `update.health_timeout` 5m; failure: stop + HOLD, +> `RestoreHold.Reason=update_failed`, pin stays). `UpdateStack` deleted. **The copy is aged by the last +> successful Tier-2 copy, NOT the manifest `created_at`** — measured on demo-hp that `created_at` moves +> only on definition changes. Journal `update-journal.json`; `RecoverUpdates` before the boot sweep. +> Three unattended paths that ignored a hold now honour it (drive-return gate ×2, nightly volume dump); +> capture and Tier-2 skip held apps. **No automatic rollback** — measured per-app; route back = restore. > **2026-09-13 — v0.236.0 (R-442).** Removal resolves the drive from the app's OWN `app.yaml` > `HDD_PATH` (the `07` ~L437 rule), never the global `cfg.Paths.HDDPath` (set on no box); a data diff --git a/REUSE.md b/REUSE.md index 6182370..b184632 100644 --- a/REUSE.md +++ b/REUSE.md @@ -109,7 +109,7 @@ | `Manager.PersistUnitRedeployConfig` (R-47, v0.153.0) | controller/internal/stacks/deploy.go | `(name, env map[string]string) error` | the PERSIST half of `RedeployFromEnv` — app.yaml + locked fields + in-memory flags, **starts nothing** | **TRAP: the restore paths must use THIS, never `RedeployFromEnv`.** RedeployFromEnv ends in a full `up -d`, which before the replay IS the H4 race. RedeployFromEnv is now literally this + the unchanged up-and-report tail | | `Manager.StartStackServices` (R-47, v0.153.0) | controller/internal/stacks/manager.go | `(name string, services []string) error` | scoped `compose up -d ...` — the DB-only window a dump is replayed in | **REFUSES an empty list** (argument-less `up -d` is a FULL start — the one silent fall-through that would reintroduce the race). No `logPostStartStatus`: the app containers are absent on purpose. Never `RestartStack` here — it is a full up in disguise | | `appbackup.DBServiceNames` / `dbTypeForImage` (R-47, v0.153.0) | controller/internal/appbackup/dbservices.go | `(composePath string) ([]string, error)` | naming the compose SERVICE(s) holding a database, sorted | yaml.v3 `services:` MAP parse — **never a line scan** (immich's top-level `immich_ml_cache:` / `immich_postgres_data:` volume keys look exactly like services). `dbTypeForImage` is shared with `DiscoverDatabases`, which is what makes "a dump exists ⇒ a service can be named" hold. An error means CANNOT-TELL, never "no database" — callers refuse when a dump exists | -| `Manager.StartStack/StopStack/RestartStack/UpdateStack` | controller/internal/stacks/manager.go | `(name string) error` | Lifecycle | Protected stacks refuse stop; all funnel through composeExec. **NOT writers of desired state (R-166)** — 14 call sites, only 2 are the customer; recording intent here would make a nightly backup indistinguishable from the customer pressing Stop. Use `SetDesiredState` at the intent point instead | +| `Manager.StartStack/StopStack/RestartStack` | controller/internal/stacks/manager.go | `(name string) error` | Lifecycle | **`UpdateStack` was DELETED in v0.237.0 — use `StartGuardedUpdate` (update.go).** Protected stacks refuse stop; all funnel through composeExec. **NOT writers of desired state (R-166)** — 14 call sites, only 2 are the customer; recording intent here would make a nightly backup indistinguishable from the customer pressing Stop. Use `SetDesiredState` at the intent point instead | | `Manager.SetDesiredState` / `DesiredStateOf` / `BackfillDesiredState` (R-166, v0.189.0) | controller/internal/stacks/desiredstate.go | `(name, desired string) error` / `(Stack) string` / `() int` | THE customer-intent record — `app.yaml` `desired_state`, tri-state `""`/`running`/`stopped` | **ONE OWNER: the customer's action.** Writers are the API action switch, `DeployStack`, `UpdateOptionalConfig`'s redeploy branch, and the `.fab` restore adapter — nothing else, ever. **`""` (absent) means UNKNOWN, never "running"**: every pre-v0.189.0 app.yaml reads absent, so treating it as running would start every deliberately-stopped app on upgrade. Write intent BEFORE the act and REFUSE the act if it fails (§8.2). Backfill is **running-only** — never infer `stopped` from zero containers, that inference IS the defect | | `Manager.DriveLive` (R-171, v0.190.0) | controller/internal/stacks/deploy.go | `(hddPath string) bool` | is an app's data drive a live mountpoint RIGHT NOW | Wraps the **same** `isMountPoint` seam the userdata belt uses (`manager.go`) — never write a second liveness check, the two would drift invisibly. The system/local path is legitimately not a mountpoint and returns true | | `bootrecon.StartGate` (R-171, v0.190.0) | controller/internal/bootrecon/bootrecon.go | `MayStart(stack) (bool, reason)` | THE one question the boot sweep asks before starting anything | **Fail-safe: cannot determine ⇒ return FALSE.** One seam for all three holders (absent drive · quiesce · an in-flight app-data operation) because they differ only in the reason string. Implemented in `main.go` (`bootDriveGate`) reusing `quiesce.SuppressedStacks()`, `AppStopGuard.HeldStacks()` and `Manager.DriveLive` — never re-derive any of them. Held apps go to `Result.HeldByDrive`, **never** `StillDown` (that is the dead-app alarm's bucket) | @@ -120,7 +120,12 @@ | `Manager.DeleteStack` / `RemoveStack` | controller/internal/stacks/delete.go | `(name, removeHDDData[, backupPaths])` | THE guarded removal paths | Orphan/protected/deploying/running checks + ProtectedHDDPaths filter before any RemoveAll. **R-442 (v0.236.0): the drive is the app's OWN `app.yaml` `HDD_PATH` (`appHDDPath`), never `cfg.Paths.HDDPath`; a data removal that cannot be resolved returns a typed `*RemoveRefusedError` BEFORE `compose down` — handlers `errors.As` it to 409 + `Message`** | | `resolveContainerState` / `aggregateState` | controller/internal/stacks/manager.go | `(dockerState, dockerStatus)` / `([]ContainerInfo)` | State classification | `.State` says "running" even when unhealthy — `.Status` parse is the fix | | `Manager.recordInstalledImages` (v0.233.0) | controller/internal/stacks/installed.go | `(name, stackDir string, env []string)` | writing down what each compose SERVICE is ACTUALLY running, into `app.yaml.installed_images` | Called after a successful compose up from `StartStack`/`RestartStack`/`UpdateStack`/`runComposeDeploy`. **Reads the CONTAINER, never `docker-compose.yml`** — that file is the value the syncer has already moved (spike §3: 25 minutes of disagreement). **A failed write NEVER refuses the action** — the deliberate OPPOSITE of `SetDesiredState`: intent refused, observation logged at ERROR. **NOT from `StartStackServices`** (the R-47 DB-only window would overwrite a complete record with a partial one). Skips the write when ref+digest are unchanged, and carries `at` forward so it means "running since". Its OWN seam (`installedExecFn`) with a **context + 30 s timeout** — the two existing exec helpers have neither | -| `Manager.SetPin` / `AdoptPins` / `RenderPlanFor` / `AppliedComposePath` (v0.235.0) | controller/internal/stacks/pin.go | `SetPin(name, stackDir, pin, composeSrc) error` | THE version freeze — `app.yaml.pinned_images` + the stored `applied-compose.yml` | **`PinnedImages` is INTENT, `InstalledImages` is an OBSERVATION — never feed one from the other** (the R-166 category error, one field over). Four writers only: deploy, `UpdateStack` (via `advancePinToCatalog`, which advances the pin and re-renders BEFORE the pull, and REFUSES the update if the pin cannot be written), the restore adapter (this is what closes R-441), and `AdoptPins`. `AdoptPins` reuses `observationCoversTemplate` — do NOT write a second completeness rule — and skips loudly rather than inventing a pin. Absent pin = pre-v0.235.0 behaviour | +| `Manager.SetPin` / `AdoptPins` / `RenderPlanFor` / `AppliedComposePath` (v0.235.0) | controller/internal/stacks/pin.go | `SetPin(name, stackDir, pin, composeSrc) error` | THE version freeze — `app.yaml.pinned_images` + the stored `applied-compose.yml` | **`PinnedImages` is INTENT, `InstalledImages` is an OBSERVATION — never feed one from the other** (the R-166 category error, one field over). Four writers only: deploy, the guarded update (via `advancePinToCatalog`, which advances the pin and re-renders BEFORE the pull, and REFUSES the update if the pin cannot be written; a failed pull puts the pin BACK via `SetPin`), the restore adapter (this is what closes R-441), and `AdoptPins`. `AdoptPins` reuses `observationCoversTemplate` — do NOT write a second completeness rule — and skips loudly rather than inventing a pin. Absent pin = pre-v0.235.0 behaviour | +| `Manager.UpdatePreflight` / `StartGuardedUpdate` / `RecoverUpdates` / `ResumeInterruptedUpdates` (v0.237.0) | controller/internal/stacks/update.go | `UpdatePreflight(name) *UpdateRefusal`; `StartGuardedUpdate(name) error` | THE update — refusals, then a 202 job with phases on `Stack.Updating/UpdatePhase/UpdateError` | **Never report an update complete before health is known (R-443).** Every cheap refusal runs BEFORE the intent write. Safety dump BEFORE the pin moves; pin BEFORE pull; pull failure → pin back; health failure → stop + HOLD, pin stays. Journal-before-mutate (`update-journal.json`); `RecoverUpdates` MUST run before the boot sweep and `ResumeInterruptedUpdates` AFTER `SetUpdateGuards`. Seams: `updateComposeFn`, `updateHealthFn`, `updateMemoryFn`, `updateDiskFreeFn`, `updateNowFn` (R-457: the age check and the test read ONE clock). Unwired guards ⇒ every update refused | +| `stacks.UpdateGuards` + `updateGuardsAdapter` (v0.237.0) | controller/internal/stacks/update.go, controller/cmd/controller/main.go | `HoldFor`, `Busy`, `RestorePoint`, `BackupNow`, `SafetyDump`, `HoldAfterFailedUpdate` | the ONLY bridge from the update job to the backup side (stacks cannot import backup) | Wired by `stackMgr.SetUpdateGuards` — pinned by `TestSlice4_UpdateGuardsAreWiredAtStartup`. Add a guard HERE, never by importing backup into stacks | +| `backup.Manager.Tier2UnitRestorePoint` + `Tier2RestorePoint.ProvenCopyTime` (v0.237.0) | controller/internal/backup/update_guard.go | `(stack) (Tier2RestorePoint, error)` | "can this app be restored from Tier 2, and from when" — the predicate that gates BOTH the „Teljes visszaállítás" action and an update | **ONE predicate, two callers** (extracted from `buildAppBackupRows`, not copied). `CopyDate` is what the page NAMES (the package date, R-403); `ProvenCopyTime` is how OLD the data is — the last successful copy, because the manifest's `created_at` moves only when the DEFINITION changes (measured: a fresh dump under a 22-h-older manifest). Do not age a copy by `CopyDate` | +| `backup.Manager.HoldAfterFailedUpdate` / `RunAppBackupNow` / `WriteUpdateSafetyDump` / `UpdateBusy` (v0.237.0) | controller/internal/backup/update_guard.go | see file | the update's hold, per-app backup-now, safety dump, busy check | The hold is `settings.RestoreHold` with `Reason: update_failed` — SAME store and gate as R-379, never a second map. `RunAppBackupNow` composes the nightly legs for ONE app (admission, DB dump, volume dump, capture, Tier-2) — do not write a second backup orchestration. A successful unit restore lifts an UPDATE hold only | +| `stacks.Manager.memoryVerdict` (v0.237.0) | controller/internal/stacks/deploy.go | `(newReq, newLimit, releasedReq, releasedLimit int) (refusal, warning string)` | the deploy's memory check, shared with the update | An update RELEASES the app's current request first. Deploy passes `0, 0` and is byte-identical in wording and log line | | `Syncer.SetRenderPlanFn` + `renderSource` (v0.235.0) | controller/internal/sync/sync.go | `func(appName string) stacks.RenderPlan` | the catalog render table | **NIL-SAFE: no seam = copy verbatim = the old product.** Catalog images == pin → verbatim (fixes flow + self-healing, both deliberately kept); differ → the WHOLE stored definition, **never a ref substitution into a newer template** (`wger 2.6`). `.felhom.yml` always verbatim (R-458). The syncer must NEVER read app.yaml. Re-reads the applied file before writing it — a test caught it writing an empty compose over a live app | | `Stack.CatalogImages` vs `Stack.TemplateImages` (v0.235.0) | controller/internal/stacks/manager.go | both `map[string]string` | badge input vs "what the next `up -d` gives this app" | **THE TRAP: same type, same shape, opposite meaning after the freeze.** `TemplateImages` reads the LIVE (possibly frozen) compose file; `CatalogImages` reads the syncer's clone. `web.compareInstalledToTemplate` MUST use `CatalogImages` or it answers „Naprakész" on exactly the apps that are behind, with every test green. Red-proved | | `Manager.BackfillInstalledImages` (v0.234.0) | controller/internal/stacks/installed.go | `() int` | seeding `installed_images` for apps that have NO record — call ONCE at startup | Beside `BackfillDesiredState` in `cmd/controller/main.go`, after it and BEFORE the boot reconciler (pinned by an AST-walking test that asserts the ORDER). **READS only** — starts nothing, writes no compose file. **Never overwrites an existing record** (an app that has one is not even observed). **REFUSES a partial observation** (`observationCoversTemplate`): `web.compareInstalledToTemplate` reads a service-count mismatch as BEHIND, so seeding a degraded app from what is visible renders „Frissítés elérhető" over an app that is current. The bring-up paths may write a partial because they follow a SUCCESSFUL `up -d` where a gap is real news; a backfill meets any state and must be stricter | diff --git a/controller/README.md b/controller/README.md index e917234..4a66198 100644 --- a/controller/README.md +++ b/controller/README.md @@ -472,7 +472,7 @@ through them: `EffectiveLifecycle()`, `CanInstall()`, `IsAbandoned()`. Two additions, and **neither changes how an update behaves**. Slice 1 is a record; slice 2 is a label. **Slice 1 — `app.yaml` gains `installed_images`.** After every successful `compose up` from -`StartStack`, `RestartStack`, `UpdateStack` and the deploy path, `Manager.recordInstalledImages` +`StartStack`, `RestartStack`, the guarded update (v0.237.0; `UpdateStack` before it) and the deploy path, `Manager.recordInstalledImages` (`internal/stacks/installed.go`) reads what each container is ACTUALLY running and writes it down, **keyed by compose SERVICE name**: @@ -527,8 +527,8 @@ the current template pins and returns a `*MetaBadge` rendered by the existing `m page. **Known limitation:** for the 23 floating pins (`postgres:16-alpine`, `mariadb:11.6`, …) the reference can be identical while the image behind it has moved, so those apps can read „Naprakész" when they may not be. Digest-level comparison needs a registry query and is deferred. -- **Information only.** The badge is wired to no action; `Frissítés`/`Újraindítás`/`Leállítás` are - byte-identical to before (`TestScenarioE_TheUpdateButtonIsUntouched`). +- **Information only.** The badge is wired to no action. (What the `Frissítés` button itself does + changed in v0.237.0 — see "The guarded update" below.) Reasoning and the seven-slice plan: `felhom.eu/documentation/architecture/09-update-architecture.md`. @@ -539,7 +539,7 @@ deliberate Update moves it. Everything else in a template — health checks, mem fields — still arrives on the 15-minute cycle, and a broken definition still repairs itself. - **`app.yaml` gains `pinned_images`** (service → ref): what the app is SUPPOSED to run. **Not** - `installed_images`, which is an observation. Written only by the deploy path, `UpdateStack`, a + `installed_images`, which is an observation. Written only by the deploy path, the guarded update, a restore, and the one-time `AdoptPins`. **Absent = unpinned = pre-v0.235.0 behaviour.** - **`applied-compose.yml`** in the stack dir stores the exact definition the pin came from. The syncer copies only `docker-compose.yml` and `.felhom.yml`, so that name is safe. @@ -555,6 +555,50 @@ fields — still arrives on the 15-minute cycle, and a broken definition still r Reasoning: `felhom.eu/documentation/architecture/09-update-architecture.md` §3, §5. +#### The guarded update (v0.237.0 — update arc slice 4) + +**`POST /api/stacks/{name}/update` no longer updates on the spot.** It refuses what it must, starts a +job, and answers **202**. The page polls `GET /api/stacks/{name}`. + +| field | meaning | +|---|---| +| `updating` | a guarded update is in progress | +| `update_phase` / `update_phase_label` | `checking`, `backing-up`, `safety-dump`, `pinning`, `pulling`, `starting`, `verifying`, `done`, `failed` — and the Hungarian label for each | +| `update_error` | the customer sentence when the update did not complete | +| `hold_reason` | the hold's sentence while the app is held (failed update OR failed restore) | + +**Refused with 409 before anything moves:** held; a backup, restore, app-data op or quiesce holding +the app; a migration; already updating; deploying; not enough memory for the NEW template's request +(the deploy's own `memoryVerdict`, releasing the app's current request); less than **2 GB** free on +the Docker data root (a fixed floor — image sizes are not known without a registry query); and **no +restorable backup** (no openable Tier-2 recovery unit with a proven copy). + +**The sequence.** A proven copy older than `update.backup_max_age` is refreshed first +(`RunAppBackupNow`: this app's DB dump, volume dump, unit capture, Tier-2 copy). Then a database +safety dump, then the pin moves, then pull, `up`, and the health wait (`.felhom.yml` check, or 60 s of +every container running for an app with none; bounded by `update.health_timeout`). **A failed pull +puts the pin back. An app that does not become healthy is stopped and HELD** — the pin stays on the +new version, and the hold sentence names the backup it can be restored from. A successful unit +restore lifts an update hold. + +**Config (`controller.yaml`):** + +```yaml +update: + backup_max_age: 24h # a proven Tier-2 copy older than this is refreshed before the update + health_timeout: 5m # how long the new version has to become healthy before the app is held +``` + +**Crash safety:** `/update-journal.json` is written before every phase; `RecoverUpdates` (before +the boot sweep) puts a pin back or marks an interrupted update for `ResumeInterruptedUpdates`. + +**Every unattended start path honours a hold:** the boot sweep and the app-stop guard (as before), and +since v0.237.0 the drive-return gate and the nightly volume dump. The nightly capture and Tier-2 run +skip a held app so its restore point is not overwritten. + +**Not done, deliberately:** the old version is never put back automatically — whether that works is +per-app and was measured unpredictable. Reasoning: `felhom.eu/documentation/architecture/09-update-architecture.md` §6. + #### App Info Pages Each app can define rich metadata in `.felhom.yml`: diff --git a/controller/cmd/controller/main.go b/controller/cmd/controller/main.go index f21db71..d2cfd22 100644 --- a/controller/cmd/controller/main.go +++ b/controller/cmd/controller/main.go @@ -476,9 +476,20 @@ func main() { // R-171: hand the boot sweep the settings it needs to answer "is this app's drive live?" BEFORE // the goroutine starts — an unwired gate is silently the pre-v0.190.0 behaviour that started apps // onto absent drives. TestMainWiresBootDriveGate walks this file's AST for the assignment. + // --- Slice 4: guarded-update crash recovery (Scenario G) --- + // BEFORE the boot reconciler, and the order is load-bearing: an update interrupted after `up` is + // marked Updating here, and bootDriveGate refuses an Updating app — started after the sweep, the + // sweep could bring up a half-updated app with no record of why. Pin-backs for updates interrupted + // before anything ran happen here too. The resumed health wait itself is launched further down, + // once the backup side is wired, because a resumed update that fails must be able to HOLD. + if resumed := stackMgr.RecoverUpdates(); len(resumed) > 0 { + logger.Printf("[WARN] [update] %d interrupted update(s) will resume after the backup side is wired: %v", len(resumed), resumed) + } + bootDriveSettings = sett bootQuiesceLoop = quiesceLoop bootAppStopGuard = appStopGuard + bootStackMgr = stackMgr go runBootReconcile(ctx, stackMgr, logger) // --- Start CPU collector --- @@ -537,6 +548,16 @@ func main() { backupMgr.SetSharesReconciler(stackMgr.ReconcileSamba) } + // --- Slice 4: the guarded update's backup side --- + // Wired unconditionally: with backup disabled the adapter answers "no restore point" and every + // update is refused with the no-backup sentence — the precondition cannot be met, which is true. + // An UNWIRED manager also refuses (fail closed); TestSlice4_UpdateGuardsAreWiredAtStartup walks + // this file for the call, because a seam built and never wired has shipped here seven times. + stackMgr.SetUpdateGuards(&updateGuardsAdapter{b: backupMgr, q: quiesceLoop}) + if n := stackMgr.ResumeInterruptedUpdates(ctx); n > 0 { + logger.Printf("[WARN] [update] resumed %d interrupted update(s)", n) + } + // SLICE 2: the offsite apply-bridge is launched further down, AFTER the self-updater is constructed // (R-71a: the bridge's settle-gate reads the updater's floor/update-running state to defer the // consume past a managed day-0 floor-update). See "offsite apply-bridge" below. @@ -1906,6 +1927,7 @@ var ( bootDriveSettings *settings.Settings bootQuiesceLoop *quiesce.Loop bootAppStopGuard *backup.AppStopGuard + bootStackMgr *stacks.Manager ) // bootDriveGate answers bootrecon.StartGate for the real controller. It enforces §8.2: an app that @@ -1943,6 +1965,11 @@ func (g bootDriveGate) MayStart(stackName string) (bool, string) { if bootQuiesceLoop.SuppressedStacks()[stackName] { return false, "a whole-guest backup (quiesce) is holding it — the quiesce loop restarts its own stacks" } + // 1b. slice 4: a guarded update is moving this app (including one RecoverUpdates marked for + // resumption). The update job owns the bring-up and ends in healthy or held. + if bootStackMgr != nil && bootStackMgr.IsUpdating(stackName) { + return false, "a guarded update is in progress — the update brings it up and verifies it" + } // 2. an app-data operation in flight for _, held := range bootAppStopGuard.HeldStacks() { if held == stackName { @@ -1985,6 +2012,9 @@ func (g driveStartGate) MayStart(stackName string) (bool, string) { // app-stop guard's Recover reaches it directly. A hold only one path honours is not a hold. if g.sett != nil { if h, ok := g.sett.GetRestoreHold(stackName); ok { + if h.Reason == settings.HoldReasonUpdateFailed { + return false, "held after a failed update (" + h.At + ") — restore it from its backup to start it" + } return false, "held after a failed restore whose rollback also failed (" + h.At + ") — clear the hold to start it" } } @@ -2236,7 +2266,11 @@ func scanDeployedAppRunStates(mgr *stacks.Manager, q *quiesce.Loop, g *backup.Ap // `g` covers the per-app operations — the nightly volume dump, an off-site reconstitution and a // .fab export. Both are nil-safe, and the union is taken here rather than inside classifyRunStates // so that pure function keeps its single `quiesced` parameter and its existing tests. - return classifyRunStates(mgr.GetStacks(), unionSuppressed(q.SuppressedStacks(), g.SuppressedStacks()), q.FailedRestarts(), time.Now()) + // Slice 4: a THIRD mechanism moves an app on purpose — the guarded update recreates it and waits + // for health, and it ends in healthy or HELD. Counting it as dead mid-update would be R-330's false + // alarm one mechanism over. + suppressed := unionSuppressed(unionSuppressed(q.SuppressedStacks(), g.SuppressedStacks()), mgr.UpdatingStacks()) + return classifyRunStates(mgr.GetStacks(), suppressed, q.FailedRestarts(), time.Now()) } // unionSuppressed merges the suppression sets of the two mechanisms that stop apps on purpose. @@ -3313,3 +3347,61 @@ func integrityOKMsg(res backup.IntegrityResult) string { } return msg + ")" } + +// updateGuardsAdapter implements stacks.UpdateGuards over the backup manager (slice 4). The stacks +// package cannot import backup, so this is the one place the two meet. Nil-safe on b: a box with +// backup disabled has no restore point, and the update is refused for that true reason. +type updateGuardsAdapter struct { + b *backup.Manager + q *quiesce.Loop +} + +func (a *updateGuardsAdapter) HoldFor(name string) (bool, string) { + if a.b == nil { + return false, "" + } + return a.b.RestoreHoldFor(name) +} + +func (a *updateGuardsAdapter) Busy(name string) (bool, string) { + if a.q.SuppressedStacks()[name] { + return true, "a whole-guest backup (quiesce) is holding it" + } + if a.b == nil { + return false, "" + } + return a.b.UpdateBusy(name) +} + +func (a *updateGuardsAdapter) RestorePoint(name string) (stacks.UpdateRestorePoint, error) { + if a.b == nil { + return stacks.UpdateRestorePoint{}, fmt.Errorf("backup is not enabled on this box") + } + rp, err := a.b.Tier2UnitRestorePoint(name) + if err != nil { + return stacks.UpdateRestorePoint{}, err + } + at, proven := rp.ProvenCopyTime() + return stacks.UpdateRestorePoint{Restorable: rp.Restorable, Proven: proven, ProvenAt: at}, nil +} + +func (a *updateGuardsAdapter) BackupNow(ctx context.Context, name string) error { + if a.b == nil { + return fmt.Errorf("backup is not enabled on this box") + } + return a.b.RunAppBackupNow(ctx, name) +} + +func (a *updateGuardsAdapter) SafetyDump(ctx context.Context, name string) ([]string, error) { + if a.b == nil { + return nil, fmt.Errorf("backup is not enabled on this box") + } + return a.b.WriteUpdateSafetyDump(ctx, name) +} + +func (a *updateGuardsAdapter) HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error { + if a.b == nil { + return fmt.Errorf("backup is not enabled on this box — the hold cannot be recorded") + } + return a.b.HoldAfterFailedUpdate(name, at, provenCopyAt) +} diff --git a/controller/cmd/controller/slice4_wiring_test.go b/controller/cmd/controller/slice4_wiring_test.go new file mode 100644 index 0000000..f61192b --- /dev/null +++ b/controller/cmd/controller/slice4_wiring_test.go @@ -0,0 +1,104 @@ +package main + +import ( + "go/ast" + "go/parser" + "go/token" + "strings" + "testing" + + "gitea.dooplex.hu/admin/felhom-controller/internal/settings" +) + +// Update arc slice 4 — the seams are WIRED, in the order that makes them true. A seam built and never +// wired has shipped in this project seven times; each assertion here fails if a line moves or goes. + +func slice4CallLines(t *testing.T) (map[string][]int, *ast.File, *token.FileSet) { + t.Helper() + fset := token.NewFileSet() + f, err := parser.ParseFile(fset, "main.go", nil, 0) + if err != nil { + t.Fatal(err) + } + lines := map[string][]int{} + ast.Inspect(f, func(n ast.Node) bool { + if call, ok := n.(*ast.CallExpr); ok { + if sel, ok := call.Fun.(*ast.SelectorExpr); ok { + lines[sel.Sel.Name] = append(lines[sel.Sel.Name], fset.Position(call.Pos()).Line) + } else if id, ok := call.Fun.(*ast.Ident); ok { + lines[id.Name] = append(lines[id.Name], fset.Position(call.Pos()).Line) + } + } + return true + }) + return lines, f, fset +} + +func TestSlice4_UpdateGuardsAreWiredAtStartup(t *testing.T) { + lines, _, _ := slice4CallLines(t) + if len(lines["SetUpdateGuards"]) == 0 { + t.Fatal("SetUpdateGuards is never called — every update would be refused as unwired") + } + if len(lines["RecoverUpdates"]) == 0 || len(lines["runBootReconcile"]) == 0 { + t.Fatal("RecoverUpdates or the boot reconciler is missing from main.go") + } + if lines["RecoverUpdates"][0] > lines["runBootReconcile"][0] { + t.Errorf("RecoverUpdates (line %d) must run BEFORE the boot reconciler (line %d), or the sweep can start a half-updated app", + lines["RecoverUpdates"][0], lines["runBootReconcile"][0]) + } + if len(lines["ResumeInterruptedUpdates"]) == 0 || lines["ResumeInterruptedUpdates"][0] < lines["SetUpdateGuards"][0] { + t.Error("ResumeInterruptedUpdates must run AFTER SetUpdateGuards — a resumed update that fails must be able to HOLD") + } +} + +func funcBodySource(t *testing.T, f *ast.File, fset *token.FileSet, recv, name string) string { + t.Helper() + for _, d := range f.Decls { + fn, ok := d.(*ast.FuncDecl) + if !ok || fn.Name.Name != name || fn.Body == nil { + continue + } + if recv != "" { + if fn.Recv == nil { + continue + } + id, ok := fn.Recv.List[0].Type.(*ast.Ident) + if !ok || id.Name != recv { + continue + } + } + var names []string + ast.Inspect(fn.Body, func(n ast.Node) bool { + if sel, ok := n.(*ast.SelectorExpr); ok { + names = append(names, sel.Sel.Name) + } + return true + }) + return strings.Join(names, " ") + } + t.Fatalf("func %s.%s not found", recv, name) + return "" +} + +func TestSlice4_BootSweepAndDeadAppAlarmKnowAboutUpdates(t *testing.T) { + _, f, fset := slice4CallLines(t) + if !strings.Contains(funcBodySource(t, f, fset, "bootDriveGate", "MayStart"), "IsUpdating") { + t.Error("the boot sweep must refuse an app a guarded update is moving") + } + if !strings.Contains(funcBodySource(t, f, fset, "", "scanDeployedAppRunStates"), "UpdatingStacks") { + t.Error("the dead-app alarm must not count an app the update itself is recreating") + } +} + +// The shared start gate names WHICH hold it is honouring — an operator reading the boot log must not +// be told a restore failed when an update did. +func TestSlice4_DriveStartGate_NamesAnUpdateHold(t *testing.T) { + sett := holdTestSettings(t) + if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "bookstack", At: "2026-09-13T08:00:00Z", Reason: settings.HoldReasonUpdateFailed}); err != nil { + t.Fatal(err) + } + ok, why := driveStartGate{sett: sett}.MayStart("bookstack") + if ok || !strings.Contains(why, "held after a failed update") { + t.Errorf("ok=%v why=%q", ok, why) + } +} diff --git a/controller/internal/api/router.go b/controller/internal/api/router.go index e75c732..abfa19e 100644 --- a/controller/internal/api/router.go +++ b/controller/internal/api/router.go @@ -579,7 +579,11 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) { // R-379/R-380: an app held after a failed restore + failed rollback must not start from the // customer's button either. Checked BEFORE the drive gate because it applies to driveless apps, // which is the class the hold exists for. - if action == "start" || action == "restart" { + // + // R-439 (slice 4): `update` is in this list. It was not until v0.237.0, so a held app could be + // updated — the one action most likely to make a held app's data worse. Pinned by + // TestR439_UpdateOfAHeldAppIsRefused. + if action == "start" || action == "restart" || action == "update" { if held, why := r.restoreHoldFor(name); held { writeJSON(w, http.StatusConflict, apiResponse{OK: false, Error: why}) return @@ -618,6 +622,20 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) { } } + // Slice 4: every cheap refusal of an update — busy, already updating, deploying, memory, disk, and + // "no backup to return to" — BEFORE the intent below is recorded, so an update that was never going + // to happen records nothing (§8.2). Each is a 409 with the Hungarian sentence. + if action == "update" { + if ref := r.stackMgr.UpdatePreflight(name); ref != nil { + status := http.StatusConflict + if ref.Reason == "not_found" { + status = http.StatusNotFound + } + writeJSON(w, status, apiResponse{OK: false, Error: ref.Message}) + return + } + } + // R-166: THE CUSTOMER-INTENT POINT. This switch is where a human's decision about whether their // app should be running enters the system, and until v0.189.0 that decision was recorded nowhere // — so the box had to infer it from container counts, and inferred wrong for a power cut and for @@ -647,12 +665,18 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) { case "restart": err = r.stackMgr.RestartStack(name) case "update": - err = r.stackMgr.UpdateStack(name) + // Slice 4: the GUARDED update. It returns as soon as the job has started; the result is only + // ever known from GET /api/stacks/{name} (updating / update_phase / update_error). + err = r.stackMgr.StartGuardedUpdate(name) } if err != nil { r.logger.Printf("[ERROR] [api] %s failed for %s: %v", action, name, err) status := http.StatusInternalServerError + var ref *stacks.UpdateRefusal + if errors.As(err, &ref) { + status = http.StatusConflict + } if strings.Contains(err.Error(), "protected") { status = http.StatusForbidden } @@ -663,6 +687,15 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) { return } + // R-443 (slice 4): an update is NEVER reported completed here. The spike measured this line saying + // "update completed" over an app that was already crash-looping. It answers 202 — accepted, not + // finished — and "completed" exists only as update_phase=done on GET /api/stacks/{name}, which is + // written after the app's health is known. Pinned by TestR443_UpdateIsNeverReportedCompleteSynchronously. + if action == "update" { + writeJSON(w, http.StatusAccepted, apiResponse{OK: true, Message: "Frissítés elindult – az állapot a kártyán követhető", + Data: map[string]interface{}{"accepted": true, "completed": false}}) + return + } writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Stack " + name + " " + action + " completed"}) // Trigger integration lifecycle hooks after successful action diff --git a/controller/internal/api/slice4_update_test.go b/controller/internal/api/slice4_update_test.go new file mode 100644 index 0000000..2f594fa --- /dev/null +++ b/controller/internal/api/slice4_update_test.go @@ -0,0 +1,165 @@ +package api + +import ( + "context" + "encoding/json" + "fmt" + "io" + "log" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "testing" + "time" + + "gitea.dooplex.hu/admin/felhom-controller/internal/backup" + "gitea.dooplex.hu/admin/felhom-controller/internal/config" + "gitea.dooplex.hu/admin/felhom-controller/internal/settings" + "gitea.dooplex.hu/admin/felhom-controller/internal/stacks" +) + +// Update arc slice 4 through the PRODUCTION handler: actionStack → the real stacks.Manager +// (NewManager + ScanStacks) and the real backup.Manager over real settings. No docker is reached: +// every path here refuses, or fails at the pin (the app's catalog template is absent on purpose). + +type apiFakeGuards struct { + b *backup.Manager + rp stacks.UpdateRestorePoint + // blindToHolds makes the manager-side preflight NOT see holds, so a test can prove the ROUTER's + // own hold check refuses — the two layers are each pinned separately (the preflight's by + // TestSlice4_D_CheapRefusals/held). Without it, removing either layer passes inertly, because the + // other refuses with the same sentence (observed on the first run of red-proof 4, 2026-09-13). + blindToHolds bool +} + +func (g *apiFakeGuards) HoldFor(n string) (bool, string) { + if g.blindToHolds { + return false, "" + } + return g.b.RestoreHoldFor(n) +} +func (g *apiFakeGuards) Busy(string) (bool, string) { return false, "" } +func (g *apiFakeGuards) RestorePoint(string) (stacks.UpdateRestorePoint, error) { + return g.rp, nil +} +func (g *apiFakeGuards) BackupNow(context.Context, string) error { return nil } +func (g *apiFakeGuards) SafetyDump(context.Context, string) ([]string, error) { + return nil, nil +} +func (g *apiFakeGuards) HoldAfterFailedUpdate(string, time.Time, time.Time) error { return nil } + +const slice4AppYAML = "deployed: true\nenv: {}\npinned_images:\n app: nginx:1.27\n" + +func newSlice4Router(t *testing.T) (*Router, *settings.Settings, *apiFakeGuards, string) { + t.Helper() + root := t.TempDir() + dir := filepath.Join(root, "stacks", "app") + if err := os.MkdirAll(dir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dir, "docker-compose.yml"), []byte("services:\n app:\n image: nginx:1.27\n"), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dir, "app.yaml"), []byte(slice4AppYAML), 0o600); err != nil { + t.Fatal(err) + } + cfg := &config.Config{} + cfg.Paths.StacksDir = filepath.Join(root, "stacks") + cfg.Paths.DataDir = filepath.Join(root, "data") + cfg.Paths.SystemDataPath = filepath.Join(root, "sys") + cfg.Stacks.ComposeCommand = "docker compose" + lg := log.New(io.Discard, "", 0) + m, err := stacks.NewManager(cfg, lg) + if err != nil { + t.Fatal(err) + } + if err := m.ScanStacks(); err != nil { + t.Fatal(err) + } + sett, err := settings.Load(filepath.Join(root, "settings.json"), lg) + if err != nil { + t.Fatal(err) + } + b := backup.NewManager(cfg, sett, lg) + g := &apiFakeGuards{b: b, rp: stacks.UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: time.Now().Add(-time.Hour)}} + m.SetUpdateGuards(g) + return &Router{cfg: cfg, stackMgr: m, backupMgr: b, logger: lg}, sett, g, dir +} + +func postUpdate(t *testing.T, r *Router) (int, apiResponse) { + t.Helper() + w := httptest.NewRecorder() + r.actionStack(w, "update", "app") + var resp apiResponse + if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil { + t.Fatalf("non-JSON body %q: %v", w.Body.String(), err) + } + return w.Code, resp +} + +// TestR439_UpdateOfAHeldAppIsRefused — R-439 closed. +// +// COMPANION RED-PROOF 4 (REPORT.md): remove `|| action == "update"` from actionStack's hold check. The +// preflight then refuses on its own grounds with a DIFFERENT sentence, and this test fails on the +// message — which is what proves the router line is the one doing it. +func TestR439_UpdateOfAHeldAppIsRefused(t *testing.T) { + r, sett, g, dir := newSlice4Router(t) + g.rp = stacks.UpdateRestorePoint{} // the preflight's own refusal would say "no backup" — not the hold + g.blindToHolds = true // only the router's line can produce the hold's sentence + if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "app", At: "2026-09-13T08:00:00Z", Reason: settings.HoldReasonUpdateFailed, CopyDate: "2026-09-13T01:30:00Z"}); err != nil { + t.Fatal(err) + } + before, _ := os.ReadFile(filepath.Join(dir, "app.yaml")) + code, resp := postUpdate(t, r) + _, holdText := r.backupMgr.RestoreHoldFor("app") + if code != http.StatusConflict || resp.OK || resp.Error != holdText { + t.Fatalf("a HELD app's update must be refused with the hold's own sentence: code=%d ok=%v error=%q", code, resp.OK, resp.Error) + } + after, _ := os.ReadFile(filepath.Join(dir, "app.yaml")) + if string(before) != string(after) { + t.Error("a refused update must record no intent — app.yaml changed") + } +} + +func TestSlice4_Router_NoBackupIs409AndRecordsNothing(t *testing.T) { + r, _, g, dir := newSlice4Router(t) + g.rp = stacks.UpdateRestorePoint{Restorable: false} + before, _ := os.ReadFile(filepath.Join(dir, "app.yaml")) + code, resp := postUpdate(t, r) + if code != http.StatusConflict || resp.Error != fmt.Sprintf(stacks.MsgUpdateNoBackupFmt, "app") { + t.Fatalf("code=%d error=%q", code, resp.Error) + } + if after, _ := os.ReadFile(filepath.Join(dir, "app.yaml")); string(before) != string(after) { + t.Error("the preflight refusal must come BEFORE the intent write") + } +} + +// TestR443_UpdateIsNeverReportedCompleteSynchronously — R-443 closed. The handler answers 202 with +// completed:false; the job then runs (and here fails at the pin, the catalog being absent), and the +// outcome exists ONLY on GET /api/stacks/{name}. +func TestR443_UpdateIsNeverReportedCompleteSynchronously(t *testing.T) { + r, _, _, _ := newSlice4Router(t) + code, resp := postUpdate(t, r) + if code != http.StatusAccepted { + t.Fatalf("an accepted update must answer 202, got %d (%+v)", code, resp) + } + data, _ := resp.Data.(map[string]interface{}) + if data["completed"] != false || data["accepted"] != true { + t.Errorf("the body must say accepted and NOT completed, got %v", resp.Data) + } + if resp.Message == "Stack app update completed" { + t.Error("the synchronous response claimed completion — R-443") + } + deadline := time.Now().Add(5 * time.Second) + for time.Now().Before(deadline) { + if st, ok := r.stackMgr.GetStack("app"); ok && !st.Updating { + if st.UpdatePhase != stacks.UpdatePhaseFailed || st.UpdateError != stacks.MsgUpdatePinFailed { + t.Errorf("the job's truth must be on the stack: phase=%q err=%q", st.UpdatePhase, st.UpdateError) + } + return + } + time.Sleep(10 * time.Millisecond) + } + t.Fatal("the job never finished") +} diff --git a/controller/internal/backup/backup.go b/controller/internal/backup/backup.go index 9013669..2ca529d 100644 --- a/controller/internal/backup/backup.go +++ b/controller/internal/backup/backup.go @@ -680,6 +680,15 @@ func (m *Manager) runVolumeDumps() (summary []string, dumped int, allOK bool) { if m.cfg != nil && m.cfg.IsProtectedStack(stack.Name) { continue } + // Slice 4 / R-379: a HELD app is deliberately stopped, and DumpAppVolumesSafe ends in + // StartStack — so without this line the nightly backup restarted every held app, every night. + // "A hold that only one path honours is not a hold." Checked BEFORE the volume check so a held + // app is never stopped or started by this leg at all. + if m.isHeld(stack.Name) { + m.logger.Printf("[WARN] [backup] Skipping volume dump for %s — the app is HELD stopped (its restore point is preserved)", stack.Name) + summary = append(summary, fmt.Sprintf("SKIP %s volumes (held)", stack.Name)) + continue + } // Volume check FIRST — a volume-less stack must not be stopped at all (see gate-order note). if len(m.stackProvider.GetDockerVolumes(stack.Name)) == 0 { if m.isDebug() { diff --git a/controller/internal/backup/offbox_reconstitute.go b/controller/internal/backup/offbox_reconstitute.go index e4d2f28..2f91008 100644 --- a/controller/internal/backup/offbox_reconstitute.go +++ b/controller/internal/backup/offbox_reconstitute.go @@ -337,6 +337,15 @@ func (m *Manager) RestoreHoldFor(stack string) (bool, string) { if !ok { return false, "" } + // Slice 4: one storage, two reasons. An update hold names the copy it can be restored from; a + // restore hold names nothing, because the restore it refers to already consumed the copy. + if h.Reason == settings.HoldReasonUpdateFailed { + copyDate := "legutóbbi" + if h.CopyDate != "" { + copyDate = fmtHoldTime(h.CopyDate) + } + return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), copyDate) + } when := h.At if t, err := time.Parse(time.RFC3339, h.At); err == nil { when = t.Format("2006-01-02 15:04") diff --git a/controller/internal/backup/recovery_unit.go b/controller/internal/backup/recovery_unit.go index 0e69288..4134f8b 100644 --- a/controller/internal/backup/recovery_unit.go +++ b/controller/internal/backup/recovery_unit.go @@ -385,6 +385,14 @@ func (m *Manager) captureAllRecoveryUnits() { if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) { continue // drive not writable — skip, the existing unit stays as-is } + // Slice 4: a HELD app's unit is its RESTORE POINT, and the hold text names that copy's date. + // Re-capturing it would write the definition the app is held ON (after a failed update: the new + // version that would not start) into the unit, and the next Tier-2 run would mirror it over the + // copy the customer was told to restore from. The unit stays exactly as it was until the hold is + // lifted. + if m.isHeld(stack.Name) { + continue + } m.noteAttempted(stack.Name) // The reserve, checked BEFORE anything is written. Per app, and the loop continues. if !m.admitApp(stack.Name) { diff --git a/controller/internal/backup/restore_unit.go b/controller/internal/backup/restore_unit.go index fac3a27..bf1f8b3 100644 --- a/controller/internal/backup/restore_unit.go +++ b/controller/internal/backup/restore_unit.go @@ -390,5 +390,8 @@ func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRest } m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s — %d volume(s) of %d listed, %d database(s) of %d listed", stackName, res.VolumesReplayed, res.ManifestVolumes, res.DBsReplayed, res.ManifestDBs) + // Slice 4: the app is back on its unit's definition and data and was started — the route back an + // update hold names. Lift that hold (and only that kind; see clearUpdateHoldAfterRestore). + m.clearUpdateHoldAfterRestore(stackName) return res, nil } diff --git a/controller/internal/backup/slice4_update_guard_test.go b/controller/internal/backup/slice4_update_guard_test.go new file mode 100644 index 0000000..4d586d3 --- /dev/null +++ b/controller/internal/backup/slice4_update_guard_test.go @@ -0,0 +1,168 @@ +package backup + +import ( + "fmt" + "io" + "log" + "path/filepath" + "strings" + "testing" + "time" + + "gitea.dooplex.hu/admin/felhom-controller/internal/settings" +) + +// Update arc slice 4 — the backup side of the guarded update. + +// The measured case (demo-hp 2026-09-13, bookstack): the mirror's manifest said 2026-09-12T02:15:29Z +// while its database dump was written 2026-09-13T00:30Z and the Tier-2 run succeeded at 01:30Z. The +// update's age must be the proven COPY time; the manifest date would call a fresh copy stale forever. +func TestSlice4_ProvenCopyTime_IsTheLastSuccessNotTheManifestDate(t *testing.T) { + rp := restorePointFromCoverage(Tier2Coverage{ + UnitRestorable: true, UnitPackageDate: "2026-09-12T02:15:29Z", + CopyLastRun: "2026-09-13T01:30:00Z", CopyLastSuccess: "2026-09-13T01:30:00Z", + }) + at, ok := rp.ProvenCopyTime() + if !ok || !at.Equal(time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)) { + t.Fatalf("proven copy time = %v ok=%v, want the last successful copy", at, ok) + } + // The page still names the PACKAGE date (R-403) — the extraction changes nothing it shows. + if rp.CopyDate != "2026-09-12T02:15:29Z" || !rp.CopyDateProven || rp.PackagePreserved { + t.Errorf("restore point = %+v", rp) + } +} + +func TestSlice4_ProvenCopyTime_PreservedPackageUsesThePackageDate(t *testing.T) { + rp := restorePointFromCoverage(Tier2Coverage{ + UnitRestorable: true, UnitPackageDate: "2026-09-01T02:00:00Z", UnitLegPreserved: true, + CopyLastSuccess: "2026-09-13T01:30:00Z", + }) + if at, ok := rp.ProvenCopyTime(); !ok || !at.Equal(time.Date(2026, 9, 1, 2, 0, 0, 0, time.UTC)) { + t.Errorf("a PRESERVED package is as old as the package, got %v ok=%v", at, ok) + } +} + +func TestSlice4_ProvenCopyTime_NoProvenOrNoUnitIsNoRestorePoint(t *testing.T) { + for _, cov := range []Tier2Coverage{ + {UnitRestorable: true, CopyLastRun: "2026-09-13T01:30:00Z"}, // attempt, never a success (R-101) + {UnitRestorable: false, CopyLastSuccess: "2026-09-13T01:30:00Z"}, // a copy with no openable unit + {UnitRestorable: true, CopyLastSuccess: "not-a-date"}, // unparseable is unknown, never "now" + } { + if _, ok := restorePointFromCoverage(cov).ProvenCopyTime(); ok { + t.Errorf("%+v must not yield a proven copy time", cov) + } + } +} + +func slice4Settings(t *testing.T) *settings.Settings { + t.Helper() + s, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), log.New(io.Discard, "", 0)) + if err != nil { + t.Fatal(err) + } + return s +} + +func TestSlice4_UpdateHoldTextNamesTheTimeAndTheCopy(t *testing.T) { + sett := slice4Settings(t) + m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett} + at := time.Date(2026, 9, 13, 8, 0, 0, 0, time.UTC) + copyAt := time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC) + if err := m.HoldAfterFailedUpdate("bookstack", at, copyAt); err != nil { + t.Fatal(err) + } + h, ok := sett.GetRestoreHold("bookstack") + if !ok || h.Reason != settings.HoldReasonUpdateFailed || h.CopyDate != "2026-09-13T01:30:00Z" { + t.Fatalf("hold = %+v ok=%v", h, ok) + } + held, why := m.RestoreHoldFor("bookstack") + // Budapest is UTC+2 in September: 08:00Z → 10:00, 01:30Z → 03:30. + want := fmt.Sprintf(UpdateHoldFmt, "bookstack", "2026-09-13 10:00", "2026-09-13 03:30") + if !held || why != want { + t.Errorf("hold text =\n%q\nwant\n%q", why, want) + } +} + +func TestSlice4_RestoreHoldTextIsUnchanged(t *testing.T) { + sett := slice4Settings(t) + m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett} + if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "docmost", At: "2026-08-22T14:00:00Z"}); err != nil { + t.Fatal(err) + } + _, why := m.RestoreHoldFor("docmost") + if !strings.Contains(why, "visszaállítása") || !strings.Contains(why, "Vedd fel velünk a kapcsolatot") || strings.Contains(why, "frissítése") { + t.Errorf("an R-379 restore hold must keep its own sentence, got %q", why) + } +} + +func TestSlice4_ASuccessfulRestoreClearsOnlyAnUpdateHold(t *testing.T) { + sett := slice4Settings(t) + m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett} + _ = m.HoldAfterFailedUpdate("upd", time.Now(), time.Now()) + _ = sett.SetRestoreHold(settings.RestoreHold{Stack: "rst", At: "2026-08-22T14:00:00Z"}) + m.clearUpdateHoldAfterRestore("upd") + m.clearUpdateHoldAfterRestore("rst") + if _, ok := sett.GetRestoreHold("upd"); ok { + t.Error("a restore is the route back from a failed update — its hold must be lifted") + } + if _, ok := sett.GetRestoreHold("rst"); !ok { + t.Error("an R-379 restore hold stays operator-cleared") + } +} + +func TestSlice4_UpdateBusy(t *testing.T) { + m := &Manager{logger: log.New(io.Discard, "", 0)} + if busy, _ := m.UpdateBusy("app"); busy { + t.Fatal("an idle manager is not busy") + } + if err := m.acquireRunning(); err != nil { + t.Fatal(err) + } + if busy, _ := m.UpdateBusy("app"); !busy { + t.Error("a running backup/restore must make an update wait") + } + m.releaseRunning() + m.BeginRestoreOp("tier2-unit-restore", "other") + if busy, _ := m.UpdateBusy("app"); !busy { + t.Error("a restore op in flight must make an update wait") + } +} + +// "A hold that only one path honours is not a hold." The nightly legs are unattended start paths +// (DumpAppVolumesSafe ends in StartStack) and writers of the restore point the hold text names. +// +// COMPANION RED-PROOF (REPORT.md): delete the isHeld skip from runVolumeDumps — the held app is then +// stopped (and restarted) by the nightly backup, and this test fails. +func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) { + h := newAdmissionHarness(t, "held", "free") + h.m.settings = slice4Settings(t) + if err := h.m.HoldAfterFailedUpdate("held", time.Now(), time.Now()); err != nil { + t.Fatal(err) + } + h.m.runVolumeDumps() + for _, n := range append(append([]string{}, h.volDumped...), h.prov.stopped...) { + if n == "held" { + t.Fatalf("the nightly volume dump touched a HELD app (dumped=%v stopped=%v)", h.volDumped, h.prov.stopped) + } + } + if len(h.volDumped) != 1 || h.volDumped[0] != "free" { + t.Errorf("positive control: the unheld app must still be dumped, got %v", h.volDumped) + } + h.m.captureAllRecoveryUnits() + for _, n := range h.prov.infoHits { + if n == "held" { + t.Error("the capture must not rewrite a HELD app's restore point") + } + } + var mirrored []string + h.m.perAppTier2 = func(name string) error { mirrored = append(mirrored, name); return nil } + h.m.RunAllTier2() + for _, n := range mirrored { + if n == "held" { + t.Error("Tier 2 must not mirror over a HELD app's copy") + } + } + if len(mirrored) != 1 { + t.Errorf("positive control: the unheld app must still be mirrored, got %v", mirrored) + } +} diff --git a/controller/internal/backup/tier2.go b/controller/internal/backup/tier2.go index 8b29192..5e24acc 100644 --- a/controller/internal/backup/tier2.go +++ b/controller/internal/backup/tier2.go @@ -457,6 +457,12 @@ func (m *Manager) RunAllTier2() { m.settings.IsDecommissioned(m.GetAppDrivePath(stack.Name))) { continue } + // Slice 4: never mirror over a HELD app's copy — it is the restore point the hold text names. + // See the matching skip in captureAllRecoveryUnits. + if m.isHeld(stack.Name) { + m.logger.Printf("[WARN] [backup] Tier 2 skipped for %s — the app is HELD; its copy is the restore point and is preserved", stack.Name) + continue + } runOne := m.perAppTier2 if runOne == nil { runOne = m.RunTier2 diff --git a/controller/internal/backup/update_guard.go b/controller/internal/backup/update_guard.go new file mode 100644 index 0000000..468703b --- /dev/null +++ b/controller/internal/backup/update_guard.go @@ -0,0 +1,325 @@ +package backup + +import ( + "context" + "errors" + "fmt" + "path/filepath" + "time" + + "gitea.dooplex.hu/admin/felhom-controller/internal/settings" +) + +// ── The backup side of the guarded update (update arc slice 4, controller v0.237.0) ───────────── +// +// 09-update-architecture.md §3 decision 1 (operator ruling 2026-09-02): the safety copy for an update +// is a VERIFIED RECENT BACKUP as a PRECONDITION — not a new copy mechanism invented for the update +// path. So everything in this file composes machinery that already exists and is proven live: +// the Tier-2 unit restore's own predicate (R-102/R-103), the nightly legs (DB dump, volume dump, +// unit capture, Tier-2 mirror), the pre-restore safety dump (R-361) and the R-379 hold. +// +// The stacks package cannot import this one, so the update job reaches all of it through the +// stacks.UpdateGuards interface, implemented by an adapter in cmd/controller/main.go. + +// Tier2RestorePoint is the answer to "could this app be restored from its Tier-2 copy, and from +// when?" — the predicate the destructive „Teljes visszaállítás" action is gated on. +// +// EXTRACTED, NOT DUPLICATED (slice 4). Until v0.237.0 this computation lived inline in the backups +// page handler (buildAppBackupRows). The update path needs exactly the same question answered, and +// a second copy of a predicate is how this project's two copies of `namespaceRoot` came to differ +// (R-203). So there is one function, and the page and the update both call it. +type Tier2RestorePoint struct { + // Restorable — the copy holds an OPENABLE recovery unit (Tier2Coverage.CanRestoreUnit). + Restorable bool + // CopyDate — the date the unit restore NAMES: the package's own manifest date, falling back to the + // copy date (Tier2Coverage.UnitRestoreDate, R-403). RFC3339 as recorded, "" when unknown. + CopyDate string + // CopyDateProven — a copy actually SUCCEEDED (LastSuccess is set), never merely an attempt (R-101). + CopyDateProven bool + // PackagePreserved — the newest run PRESERVED an older package instead of refreshing it (R-403). + PackagePreserved bool + // CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded. + CopyLastSuccess string +} + +// restorePointFromCoverage is the pure half of the predicate. +func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint { + pkgDate, preserved := cov.UnitRestoreDate() + return Tier2RestorePoint{ + Restorable: cov.CanRestoreUnit(), + CopyDate: pkgDate, + CopyDateProven: cov.CopyLastSuccess != "", + PackagePreserved: preserved, + CopyLastSuccess: cov.CopyLastSuccess, + } +} + +// Tier2UnitRestorePoint resolves the app's recorded Tier-2 copy and returns the restore point. The +// error is the same refusal Tier2RestoreCoverage raises (no copy, drive gone, pre-v2 layout). +func (m *Manager) Tier2UnitRestorePoint(stackName string) (Tier2RestorePoint, error) { + cov, err := m.Tier2RestoreCoverage(stackName) + if err != nil { + return Tier2RestorePoint{}, err + } + return restorePointFromCoverage(cov), nil +} + +// ProvenCopyTime returns WHEN the data this copy would restore was last proven copied, and false when +// there is no proven, restorable copy at all. +// +// WHY NOT CopyDate, measured rather than assumed. CopyDate is the unit MANIFEST's created_at, and a +// capture rewrites the manifest only when the app's DEFINITION changes (compose, app.yaml, controller +// version) — a nightly DB dump keeps the same file name, so it does not move it. Measured on demo-hp +// 2026-09-13: bookstack's Tier-2 mirror held `bookstack-mariadb.sql` written 2026-09-13T00:30Z while +// its manifest still read 2026-09-12T02:15:29Z. Judging "recent" by that date would call a fresh copy +// stale — and, worse, a "back up first" run would not move it either on a quiet app, so the update +// would be refused forever. +// +// So the age is the last SUCCESSFUL copy (LastSuccess), which the Tier-2 run records only when it +// actually mirrored the unit — EXCEPT when the run preserved an older package (R-403), in which case +// the package date is the honest one, because that is what the copy really holds. +func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) { + if !p.Restorable || !p.CopyDateProven { + return time.Time{}, false + } + src := p.CopyLastSuccess + if p.PackagePreserved { + src = p.CopyDate + } + t, err := time.Parse(time.RFC3339, src) + if err != nil { + return time.Time{}, false + } + return t, true +} + +// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an +// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer +// sentence is chosen by the caller. +// +// IsRunning is box-wide, deliberately: the backup/restore single-flight is box-wide, and an update's +// "back up first" leg needs that same flag — an update started beside a running backup would either +// wait on it invisibly or fail half-way. +func (m *Manager) UpdateBusy(stackName string) (bool, string) { + if m == nil { + return false, "" + } + if m.IsRunning() { + return true, "a backup or restore is running (single-flight held)" + } + if st := m.RestoreStatus(); st.Running { + return true, fmt.Sprintf("restore op %q is running for %q", st.Op, st.Stack) + } + for _, held := range m.appStop.HeldStacks() { + if held == stackName { + return true, "an app-data operation (volume dump / export / reconstitute) is holding it" + } + } + return false, "" +} + +// ErrUpdateBackupNoUnit is returned when a "back up first" run completed but the app still has no +// openable Tier-2 unit — typically Tier 2 is switched off for the app or has no second target. +var ErrUpdateBackupNoUnit = errors.New("a frissítés előtti mentés lefutott, de nem jött létre visszaállítható másolat") + +// RunAppBackupNow runs THIS app's backup legs now, in the nightly order, and then its Tier-2 copy: +// database dump(s) → volume dump (if the app has named volumes) → recovery-unit capture → Tier-2 +// mirror. It is the "back up first" of slice 4 Scenario B. +// +// Composed, not reinvented: every leg is the one runDBDumpsInternal and RunAllTier2 already run, +// including the R-181 reserve (admitApp) before the first write and the R-166 app-stop marker inside +// DumpAppVolumesSafe. What differs is only the scope — one app instead of all of them — because an +// update must not bounce every other app on the box to back up one. +func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error { + if m.stackProvider == nil { + return fmt.Errorf("stack provider not configured") + } + if m.migrationActive() { + return fmt.Errorf("adatáthelyezés folyamatban — a mentés most nem indítható") + } + if err := m.acquireRunning(); err != nil { + return err + } + m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName) + start := time.Now() + legErr := func() error { + defer m.releaseRunning() + defer m.beginAdmissionRun()() + + drivePath := m.GetAppDrivePath(stackName) + if drivePath == "" || !filepath.IsAbs(drivePath) { + return fmt.Errorf("az alkalmazás meghajtója nem határozható meg") + } + if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) { + return fmt.Errorf("az alkalmazás meghajtója nem elérhető (%s)", drivePath) + } + if !m.admitApp(stackName) { + return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath) + } + nsRoot := m.namespaceRoot(drivePath) + + discover := m.discoverDBs + if discover == nil { + discover = func(ctx context.Context) ([]DiscoveredDB, error) { + return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames()) + } + } + dbs, err := discover(ctx) + if err != nil { + return fmt.Errorf("adatbázis-felderítés sikertelen: %w", err) + } + dumped := 0 + for _, db := range dbs { + if db.StackName != stackName { + continue + } + res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug()) + if res.Error != nil { + return fmt.Errorf("adatbázis-mentés sikertelen (%s): %w", db.ContainerName, res.Error) + } + dumped++ + m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size)) + } + + if len(m.stackProvider.GetDockerVolumes(stackName)) > 0 { + dump := m.dumpVolumesSafe + if dump == nil { + dump = m.DumpAppVolumesSafe + } + if err := dump(stackName); err != nil { + return fmt.Errorf("kötetmentés sikertelen: %w", err) + } + m.logger.Printf("[INFO] [backup] update pre-backup for %s: volume dump OK", stackName) + } + + if err := m.CaptureRecoveryUnit(stackName); err != nil { + return fmt.Errorf("a mentési egység rögzítése sikertelen: %w", err) + } + m.logger.Printf("[INFO] [backup] update pre-backup for %s: recovery unit captured (%d database dump(s))", stackName, dumped) + return nil + }() + if legErr != nil { + m.logger.Printf("[ERROR] [backup] update pre-backup for %s FAILED after %s: %v", stackName, time.Since(start).Round(time.Millisecond), legErr) + return legErr + } + + runOne := m.perAppTier2 + if runOne == nil { + runOne = m.RunTier2 + } + if err := runOne(stackName); err != nil { + m.logger.Printf("[ERROR] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v", stackName, err) + return fmt.Errorf("a másodlagos másolat elkészítése sikertelen: %w", err) + } + m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond)) + return nil +} + +// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the +// pin: "the state the customer was in a minute ago". It is writeSafetyDump (R-361) unchanged — the +// same `pre-restore-` undo naming, the same pruning to three, the same never-the-canonical-name rule +// — so it is also picked up by the same exclusions (it never enters a manifest's db_dumps). +// +// Returns the paths written; an app with no database returns (nil, nil), which is a no-op and never a +// failure (measured in writeSafetyDump: `len(mine) == 0` returns an empty set). +func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) ([]string, error) { + nsRoot := m.AppNamespaceRoot(stackName) + if nsRoot == "" { + return nil, fmt.Errorf("az alkalmazás mentési helye nem határozható meg") + } + set, err := m.writeSafetyDump(ctx, stackName, nsRoot) + if err != nil { + return nil, err + } + var paths []string + for _, f := range set.Files { + paths = append(paths, f.Path) + } + if len(paths) == 0 { + m.logger.Printf("[INFO] [backup] update safety dump for %s: the app has no database — nothing to copy (no-op)", stackName) + } else { + m.logger.Printf("[INFO] [backup] update safety dump for %s: %d file(s) %v", stackName, len(paths), paths) + } + return paths, nil +} + +// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app, +// the time of the failure, and the PROVEN date of the copy it can be restored from. One named string so +// a test asserts it verbatim instead of retyping Hungarian (R-364). +const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " + + "Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " + + "Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon." + +// holdTimeZone is where the customer-facing hold sentence renders its times. The same zone the web +// layer renders the Mentések page's copy dates in (web.getTimezone), so the date in the hold text and +// the date on the page it points at are the same string. +func holdTimeZone() *time.Location { + if loc, err := time.LoadLocation("Europe/Budapest"); err == nil { + return loc + } + return time.UTC +} + +func fmtHoldTime(rfc3339 string) string { + t, err := time.Parse(time.RFC3339, rfc3339) + if err != nil { + return rfc3339 + } + return t.In(holdTimeZone()).Format("2006-01-02 15:04") +} + +// HoldAfterFailedUpdate records that an app is held stopped because its new version did not come up +// healthy. Same storage and same gate as the R-379 hold — every start path that already refuses a +// restore hold refuses this one without being touched. +// +// It returns the error rather than only logging it, unlike holdAppAfterFailedRollback: the update job +// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app +// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the +// failure on the page. +func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time) error { + if m == nil || m.settings == nil { + return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName) + } + h := settings.RestoreHold{ + Stack: stackName, + At: at.UTC().Format(time.RFC3339), + Reason: settings.HoldReasonUpdateFailed, + } + if !copyDate.IsZero() { + h.CopyDate = copyDate.UTC().Format(time.RFC3339) + } + if err := m.settings.SetRestoreHold(h); err != nil { + return fmt.Errorf("persisting the update hold for %s: %w", stackName, err) + } + m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: %s)", stackName, h.CopyDate) + return nil +} + +// isHeld reports whether an app carries ANY hold. Used by the nightly legs to leave a held app alone. +func (m *Manager) isHeld(stackName string) bool { + held, _ := m.RestoreHoldFor(stackName) + return held +} + +// clearUpdateHoldAfterRestore lifts an UPDATE hold once a person has restored the app successfully. +// +// "A person clears it by restoring" — slice 4 Part 3. The restore just put the app back on the +// definition and data of its recovery unit and started it, which is the exact route back the hold +// text names; leaving the hold in place would refuse the next restart of an app that is now fine. +// +// A RESTORE hold (R-379) is deliberately NOT cleared here: that hold means a previous restore already +// left the database in an unknown state, and it stays operator-cleared (`-clear-restore-hold`). +func (m *Manager) clearUpdateHoldAfterRestore(stackName string) { + if m.settings == nil { + return + } + h, ok := m.settings.GetRestoreHold(stackName) + if !ok || h.Reason != settings.HoldReasonUpdateFailed { + return + } + if _, err := m.settings.ClearRestoreHold(stackName); err != nil { + m.logger.Printf("[ERROR] [backup] %s was restored, but its update hold could not be cleared: %v", stackName, err) + return + } + m.logger.Printf("[INFO] [backup] %s: restore completed — the update hold (set %s) is CLEARED", stackName, h.At) +} diff --git a/controller/internal/config/config.go b/controller/internal/config/config.go index 9927d5b..8479362 100644 --- a/controller/internal/config/config.go +++ b/controller/internal/config/config.go @@ -6,6 +6,7 @@ import ( "fmt" "os" "strings" + "time" "gopkg.in/yaml.v3" ) @@ -46,6 +47,49 @@ type Config struct { Quiesce QuiesceConfig `yaml:"quiesce"` MailRelay MailRelayConfig `yaml:"mail_relay"` Offsite OffsiteConfig `yaml:"offsite"` + Update UpdateConfig `yaml:"update"` +} + +// UpdateConfig tunes the guarded app update (update arc slice 4, v0.237.0). +// +// Both are OPERATOR knobs, not constants: "recent" and "healthy in time" are judgements the operator +// ruled defaults for (2026-09-13) and may want to move per box without a release. +type UpdateConfig struct { + // BackupMaxAge is how old the app's proven Tier-2 copy may be before an update makes a fresh one + // first. Default "24h". + BackupMaxAge string `yaml:"backup_max_age"` + // HealthTimeout bounds the wait for the new version to become healthy before the app is held. + // Default "5m". + HealthTimeout string `yaml:"health_timeout"` +} + +// DefaultUpdateBackupMaxAge and DefaultUpdateHealthTimeout are the values used when the config is +// empty or unparseable. An unparseable value falls back rather than failing the whole config load: +// a typo in a tuning knob must not stop a controller from starting. +const ( + DefaultUpdateBackupMaxAge = 24 * time.Hour + DefaultUpdateHealthTimeout = 5 * time.Minute +) + +// BackupMaxAgeDuration parses BackupMaxAge, falling back to the default on empty/invalid/non-positive. +func (u UpdateConfig) BackupMaxAgeDuration() time.Duration { + return parsePositiveDuration(u.BackupMaxAge, DefaultUpdateBackupMaxAge) +} + +// HealthTimeoutDuration parses HealthTimeout, falling back to the default on empty/invalid/non-positive. +func (u UpdateConfig) HealthTimeoutDuration() time.Duration { + return parsePositiveDuration(u.HealthTimeout, DefaultUpdateHealthTimeout) +} + +func parsePositiveDuration(s string, def time.Duration) time.Duration { + if s == "" { + return def + } + d, err := time.ParseDuration(s) + if err != nil || d <= 0 { + return def + } + return d } // MailRelayConfig tunes the in-controller SMTP shim (app email → shim → hub → Resend). @@ -384,6 +428,8 @@ func applyDefaults(cfg *Config) { d(&cfg.MailRelay.TLSListen, ":2465") d(&cfg.MailRelay.PlainNoTLSListen, ":2526") d(&cfg.MailRelay.ShimHost, "felhom-controller") + d(&cfg.Update.BackupMaxAge, "24h") + d(&cfg.Update.HealthTimeout, "5m") if len(cfg.MailRelay.FromDomains) == 0 { cfg.MailRelay.FromDomains = []string{"felhom.eu"} } diff --git a/controller/internal/settings/settings.go b/controller/internal/settings/settings.go index c33749e..a31bce0 100644 --- a/controller/internal/settings/settings.go +++ b/controller/internal/settings/settings.go @@ -1585,8 +1585,27 @@ type RestoreHold struct { ReplayError string `json:"replay_error,omitempty"` // what the restore hit RollbackErr string `json:"rollback_error,omitempty"` // what the rollback then hit SafetyDump string `json:"safety_dump,omitempty"` // basename of the undo copy that could not be applied + + // Reason (update arc slice 4, v0.237.0) says WHICH operation put the hold in place. Empty means + // HoldReasonRestoreFailed — every hold written before this field existed was a restore hold, so + // the zero value keeps their meaning and no migration is needed. + // + // ONE STORAGE, ONE GATE, TWO REASONS — deliberately not a second map. Every start path already + // consults GetRestoreHold (the customer's button, the boot sweep, the app-stop guard's Recover), + // and "a hold that only one path honours is not a hold". A second map would need every one of + // those paths found and changed again, and the one that got missed would be the next R-439. + Reason string `json:"reason,omitempty"` + // CopyDate is the RFC3339 time of the proven backup the customer is told they can restore from. + // Only set for HoldReasonUpdateFailed. + CopyDate string `json:"copy_date,omitempty"` } +// Hold reasons. See RestoreHold.Reason. +const ( + HoldReasonRestoreFailed = "" // R-379/R-380: a restore AND its rollback failed + HoldReasonUpdateFailed = "update_failed" // slice 4: the new version did not come up healthy +) + // SetRestoreHold records a hold. Modelled on SetDisconnected: a condition, plus what it is holding. func (s *Settings) SetRestoreHold(h RestoreHold) error { s.mu.Lock() diff --git a/controller/internal/stacks/deploy.go b/controller/internal/stacks/deploy.go index 4f5ac0f..428ecd7 100644 --- a/controller/internal/stacks/deploy.go +++ b/controller/internal/stacks/deploy.go @@ -4,6 +4,7 @@ import ( "crypto/rand" "encoding/base64" "encoding/hex" + "errors" "fmt" "log" "math/big" @@ -236,52 +237,12 @@ func (m *Manager) DeployStack(req DeployRequest) (string, error) { } // --- Memory validation --- - var deployWarning string - reservedMB := m.cfg.System.ReservedMemoryMB - totalMB, usedMB, memErr := system.GetMemoryMB() - // F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own - // cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the - // LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap — - // rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard - // either never fire (host total) or always fire (host used > guest cap). - if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 { - totalMB = gt - memErr = nil - } - if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil { - usedMB = committedReqMB - } - if memErr != nil { - m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr) - } else { - usableMB := totalMB - reservedMB - newReqMB := ParseMemoryMB(meta.Resources.MemRequest) - - m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB", - totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB) - - // Hard block: committed + new request exceeds usable memory - if newReqMB > 0 && usedMB+newReqMB > usableMB { - clearDeploying() - return "", fmt.Errorf( - "Nincs elég memória az alkalmazás telepítéséhez. "+ - "Szükséges: %d MB, Elérhető: %d MB "+ - "(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)", - newReqMB, - usableMB-usedMB, - totalMB, - usedMB, - reservedMB, - ) - } - - // Soft warning: limits exceed total (overcommit) - _, currentLimitMB := m.CommittedMemory() - newLimitMB := ParseMemoryMB(meta.Resources.MemLimit) - if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB { - deployWarning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " + - "Normál használat mellett ez nem okoz problémát." - } + // Slice 4: the block moved into memoryVerdict so the guarded update applies the SAME check with + // the SAME wording. Behaviour here is unchanged — same inputs, same log line, same refusal text. + refusal, deployWarning := m.memoryVerdict(ParseMemoryMB(meta.Resources.MemRequest), ParseMemoryMB(meta.Resources.MemLimit), 0, 0) + if refusal != "" { + clearDeploying() + return "", errors.New(refusal) } // Debug: log received values (redact passwords/secrets) @@ -1197,3 +1158,62 @@ func randomAlphanumeric(length int) (string, error) { } return string(result), nil } + +// memoryVerdict is the deploy's memory check, extracted so the guarded update uses it unchanged +// (slice 4). releasedReqMB/releasedLimitMB are what the act FREES before it takes the new amount — an +// update replaces the app's own current request, so counting both would refuse an update that fits. +// A deploy releases nothing and passes 0, 0. +// +// Returns the refusal (the deploy's own Hungarian wording, "" = admitted) and the soft overcommit +// warning. An unreadable memory reading admits with a WARN, exactly as the deploy always has. +func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string) { + reservedMB := m.cfg.System.ReservedMemoryMB + totalMB, usedMB, memErr := system.GetMemoryMB() + // F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own + // cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the + // LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap — + // rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard + // either never fire (host total) or always fire (host used > guest cap). + if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 { + totalMB = gt + memErr = nil + } + if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil { + usedMB = committedReqMB + } + if memErr != nil { + m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr) + return "", "" + } + usedMB -= releasedReqMB + if usedMB < 0 { + usedMB = 0 + } + usableMB := totalMB - reservedMB + + m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB", + totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB) + + // Hard block: committed + new request exceeds usable memory + if newReqMB > 0 && usedMB+newReqMB > usableMB { + return fmt.Sprintf( + "Nincs elég memória az alkalmazás telepítéséhez. "+ + "Szükséges: %d MB, Elérhető: %d MB "+ + "(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)", + newReqMB, + usableMB-usedMB, + totalMB, + usedMB, + reservedMB, + ), "" + } + + // Soft warning: limits exceed total (overcommit) + _, currentLimitMB := m.CommittedMemory() + currentLimitMB -= releasedLimitMB + if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB { + warning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " + + "Normál használat mellett ez nem okoz problémát." + } + return "", warning +} diff --git a/controller/internal/stacks/installed_test.go b/controller/internal/stacks/installed_test.go index ad57f2b..132e8bc 100644 --- a/controller/internal/stacks/installed_test.go +++ b/controller/internal/stacks/installed_test.go @@ -407,7 +407,9 @@ func TestGroupE_RestartStackReachesTheRecorder(t *testing.T) { func TestGroupE_EveryBringUpPathCallsTheRecorder(t *testing.T) { callers := map[string]bool{} // enclosing func name -> calls recordInstalledImages fset := token.NewFileSet() - for _, src := range []string{"manager.go", "deploy.go"} { + // update.go since v0.237.0: the guarded update replaced UpdateStack, and it records in + // verifyAndConclude — only after the app's health is known (slice 4). + for _, src := range []string{"manager.go", "deploy.go", "update.go"} { f, err := parser.ParseFile(fset, src, nil, 0) if err != nil { t.Fatal(err) @@ -433,7 +435,7 @@ func TestGroupE_EveryBringUpPathCallsTheRecorder(t *testing.T) { } } } - for _, want := range []string{"StartStack", "RestartStack", "UpdateStack", "runComposeDeploy"} { + for _, want := range []string{"StartStack", "RestartStack", "verifyAndConclude", "runComposeDeploy"} { if !callers[want] { t.Errorf("%s does not call recordInstalledImages — a bring-up path that records nothing leaves a stale record standing", want) } diff --git a/controller/internal/stacks/manager.go b/controller/internal/stacks/manager.go index 2d862c0..4eaa32f 100644 --- a/controller/internal/stacks/manager.go +++ b/controller/internal/stacks/manager.go @@ -2,6 +2,7 @@ package stacks import ( "bytes" + "context" "fmt" "log" "os" @@ -130,17 +131,28 @@ type HealthCheckDetail struct { // Stack represents a docker compose stack on disk. type Stack struct { - Name string `json:"name"` - Meta Metadata `json:"meta"` - ComposePath string `json:"compose_path"` - State ContainerState `json:"state"` - Deployed bool `json:"deployed"` // Has app.yaml with deployed=true - Protected bool `json:"protected"` - Orphaned bool `json:"orphaned"` // Deployed but no catalog template - Containers []ContainerInfo `json:"containers"` - AppConfig *AppConfig `json:"app_config,omitempty"` - Deploying bool `json:"deploying"` // compose up in progress - DeployError string `json:"deploy_error,omitempty"` // last async deploy error + Name string `json:"name"` + Meta Metadata `json:"meta"` + ComposePath string `json:"compose_path"` + State ContainerState `json:"state"` + Deployed bool `json:"deployed"` // Has app.yaml with deployed=true + Protected bool `json:"protected"` + Orphaned bool `json:"orphaned"` // Deployed but no catalog template + Containers []ContainerInfo `json:"containers"` + AppConfig *AppConfig `json:"app_config,omitempty"` + Deploying bool `json:"deploying"` // compose up in progress + DeployError string `json:"deploy_error,omitempty"` // last async deploy error + // Updating / UpdatePhase / UpdatePhaseLabel / UpdateError (update arc slice 4, v0.237.0) are the + // guarded update's in-memory progress, the same shape as Deploying/DeployError: the API answers + // 202 at once and the page polls GET /api/stacks/{name}. See update.go. + Updating bool `json:"updating"` + UpdatePhase string `json:"update_phase,omitempty"` + UpdatePhaseLabel string `json:"update_phase_label,omitempty"` + UpdateError string `json:"update_error,omitempty"` + // HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed + // restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page + // and the API cannot show a hold the gate has already lifted — or miss one it enforces. + HoldReason string `json:"hold_reason,omitempty"` HealthProbe *HealthProbeResult `json:"health_probe,omitempty"` // controller-side probe result LastUpdated time.Time `json:"last_updated"` // RestartingSince (C9-F2) is when this stack was FIRST observed in StateRestarting during the @@ -198,6 +210,16 @@ type Manager struct { restartPolicyCache map[string]string // execFn replaces execCommand's process boundary in tests; nil in production. execFn func(name string, args ...string) (string, error) + + // --- guarded update (slice 4, update.go) --- + updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED + updateComposeFn func(dir string, env []string, args ...string) (string, error) + updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string) + updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string) + updateDiskFreeFn func() (freeGiB float64, ok bool) + updateNowFn func() time.Time + updateJournalMu sync.Mutex + updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist // inspectRestartPolicyFn is the docker-inspect seam for the above; nil in production // (dockerRestartPolicy). Tests inject a scripted lookup and never touch docker. inspectRestartPolicyFn func(containerName string) (string, error) @@ -960,12 +982,15 @@ func aggregateState(containers []ContainerInfo, policyOf restartPolicyLookup) Co func (m *Manager) GetStacks() []Stack { m.mu.RLock() - defer m.mu.RUnlock() - result := make([]Stack, 0, len(m.stacks)) for _, s := range m.stacks { result = append(result, deepCopyStack(s)) } + g := m.updateGuards + m.mu.RUnlock() + for i := range result { + fillHoldReason(g, &result[i]) + } // Sort alphabetically by display name for consistent UI ordering sort.Slice(result, func(i, j int) bool { @@ -984,6 +1009,7 @@ func (m *Manager) GetStack(name string) (*Stack, bool) { return nil, false } cp := deepCopyStack(s) + fillHoldReason(m.updateGuards, &cp) return &cp, true } @@ -1224,54 +1250,11 @@ func (m *Manager) RestartStack(name string) error { return m.RefreshStatus() } -func (m *Manager) UpdateStack(name string) error { - stack, ok := m.GetStack(name) - if !ok { - return fmt.Errorf("stack %q not found", name) - } - - m.logger.Printf("[INFO] [stacks] Updating stack: %s", name) - start := time.Now() - dir := filepath.Dir(stack.ComposePath) - - // v0.235.0 — ADVANCE THE PIN FIRST, AND RE-RENDER BEFORE THE PULL. - // - // This is the ONE act entitled to move a version; the freeze exists so that nothing else can. - // The ordering is load-bearing, not stylistic: `compose pull` and `up -d` act on the file on - // disk, so the catalog's current definition has to BE that file before either runs. Setting the - // pin afterwards would pull the frozen version and change nothing, while reporting success — and - // a button that lies is worse than a button that refuses. - // - // A FAILED PIN WRITE REFUSES THE UPDATE, deliberately the opposite of recordInstalledImages. - // That field is an observation and a failed write is a bookkeeping gap; this one is INTENT, and - // an update whose intent could not be recorded leaves the box running a version it has no record - // of choosing — the exact ambiguity R-166 closed for desired_state, one field over. - if err := m.advancePinToCatalog(name, dir); err != nil { - m.logger.Printf("[ERROR] [stacks] Stack %s update refused: %v", name, err) - return fmt.Errorf("updating stack %s: %w", name, err) - } - - env := m.stackEnv(dir) - - if m.isDebug() { - m.checkLocalImages(name, dir) - } - - if _, err := m.composeExecCustomEnv(dir, env, "pull"); err != nil { - m.logger.Printf("[ERROR] [stacks] Stack %s update (pull) failed after %.1fs: %v", name, time.Since(start).Seconds(), err) - return fmt.Errorf("pulling images for %s: %w", name, err) - } - - if _, err := m.composeExecCustomEnv(dir, env, "up", "-d", "--remove-orphans"); err != nil { - m.logger.Printf("[ERROR] [stacks] Stack %s update (up) failed after %.1fs: %v", name, time.Since(start).Seconds(), err) - return fmt.Errorf("recreating %s: %w", name, err) - } - - m.logger.Printf("[INFO] [stacks] Stack %s updated successfully (took %.1fs)", name, time.Since(start).Seconds()) - m.recordInstalledImages(name, dir, env) - m.logPostStartStatus(name, dir, env) - return m.RefreshStatus() -} +// UpdateStack was REMOVED in v0.237.0 (update arc slice 4). It advanced the pin, pulled, ran `up -d` +// and reported success on the compose exit code — no copy first, no refusals, and HTTP 200 over a +// crash loop (R-443). Its only caller was the API, which now runs StartGuardedUpdate (update.go). +// Deleting it rather than leaving it is deliberate: an unguarded update path that still compiles is +// one caller away from being the next R-439. func (m *Manager) GetLogs(name string, lines int) (string, error) { stack, ok := m.GetStack(name) diff --git a/controller/internal/stacks/update.go b/controller/internal/stacks/update.go new file mode 100644 index 0000000..67633ee --- /dev/null +++ b/controller/internal/stacks/update.go @@ -0,0 +1,802 @@ +package stacks + +import ( + "context" + "encoding/json" + "fmt" + "os" + "path/filepath" + "time" + + "gitea.dooplex.hu/admin/felhom-controller/internal/system" +) + +// ── The guarded update (update arc slice 4, controller v0.237.0) ───────────────────────────────── +// +// WHAT IT REPLACED. `UpdateStack` advanced the pin, pulled, ran `up -d` and returned — no copy first, +// no check of memory, disk, a running backup or a held app, and a success the moment `up` returned. +// SPIKE-app-update-2026-09-01 §4 measured that as HTTP 200 over an app that was already crash-looping +// (R-443), and R-439 is that a held app could be updated at all. +// +// THE SEQUENCE, and the order is the point: +// +// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling +// → starting → verifying → done | failed +// +// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02, +// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The +// same predicate that permits the destructive „Teljes visszaállítás" permits the update — the +// route back IS that restore, so an update without it has no route back. +// 2. The safety dump is taken BEFORE the pin moves: it is "the state the customer was in a minute ago", +// and a minute later the migration may have run. +// 3. The pin moves BEFORE the pull (v0.235.0's reason: pull and up act on the file on disk). +// 4. A PULL failure puts the pin BACK — nothing ran, so reverting is safe and honest (Scenario E). +// 5. A HEALTH failure leaves the pin where it is — the new version's migration may have run, and a +// pin claiming the old version would be a record of something untrue (Scenario F). The app is +// HELD STOPPED and the customer is told which backup it can be restored from. +// +// WHAT IT DELIBERATELY DOES NOT DO: put the old version back by itself. SPIKE-upgrade-test-2026-09-06 +// measured that whether the old image starts on migrated data depends on the app (PrivateBin yes, +// Docmost and Nextcloud no) and cannot be predicted. The route back is the restore. +// +// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10), +// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at +// the next startup (Scenario G) — the AppStopGuard pattern. + +// Update phases, as recorded in the journal and served as Stack.UpdatePhase. +const ( + UpdatePhaseChecking = "checking" + UpdatePhaseBackingUp = "backing-up" + UpdatePhaseSafetyDump = "safety-dump" + UpdatePhasePinning = "pinning" + UpdatePhasePulling = "pulling" + UpdatePhaseStarting = "starting" + UpdatePhaseVerifying = "verifying" + UpdatePhaseDone = "done" + UpdatePhaseFailed = "failed" +) + +// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the +// specification — it is instantaneous and is the first step of fetching the new version, so it shares +// the pull's label rather than inventing a sentence nobody would read. +var updatePhaseLabels = map[string]string{ + UpdatePhaseChecking: "Ellenőrzés…", + UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…", + UpdatePhaseSafetyDump: "Adatbázis pillanatkép…", + UpdatePhasePinning: "Új verzió letöltése…", + UpdatePhasePulling: "Új verzió letöltése…", + UpdatePhaseStarting: "Indítás az új verzióval…", + UpdatePhaseVerifying: "Működés ellenőrzése…", + UpdatePhaseDone: "Frissítve", + UpdatePhaseFailed: "A frissítés nem sikerült", +} + +// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one. +func UpdatePhaseLabel(phase string) string { return updatePhaseLabels[phase] } + +// Customer sentences. Named so tests compare against the constant, never a retyped literal (R-364). +const ( + MsgUpdateNoGuards = "A frissítés nem indítható: a frissítés előtti biztonsági ellenőrzés nem érhető el ezen a szerveren." + MsgUpdateNotDeployed = "Az alkalmazás nincs telepítve, ezért nem frissíthető." + MsgUpdateDeployingFmt = "A(z) %s telepítése még folyamatban van — a frissítés utána indítható." + MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van." + MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött." + MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban." + MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani. Kapcsold be a 2. mentést az alkalmazás mentési beállításainál a Mentések oldalon, és várd meg az első sikeres másolatot — utána a frissítés elindítható." + MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges." + MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább." + MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább." + MsgUpdateDumpFailFmt = "A frissítés nem indult el, mert az adatbázis pillanatkép nem készült el: %v. Az alkalmazás változatlanul fut tovább." + MsgUpdatePinFailed = "A frissítés nem indult el: az új verzió leírása nem olvasható be. Az alkalmazás változatlanul fut tovább." + MsgUpdateJournalFailed = "A frissítés nem indult el: a frissítés naplója nem menthető. Az alkalmazás változatlanul fut tovább." + MsgUpdatePullFailed = "Az új verzió letöltése nem sikerült, ezért a frissítés elmaradt. Az alkalmazás a korábbi verzióval fut tovább." + MsgUpdateInterrupted = "A frissítés megszakadt, mert a vezérlő újraindult, mielőtt az új verzió elindult volna. Az alkalmazás a korábbi verzióval fut tovább." + MsgUpdateHoldUnsaved = "A frissítés nem sikerült, az alkalmazás le lett állítva, de a leállítás rögzítése nem sikerült. Ne indítsd újra — vedd fel velünk a kapcsolatot." +) + +// updateDiskFloorGiB is the free space the Docker data root must have before a pull. A FIXED FLOOR, +// stated as such: the new image set's size is not known without a registry query (the catalog +// records tags, not sizes, and §8.1 of 09 already declines registry calls on the customer box), so +// the rule is "not less than 2 GB", not "enough for these images". +const updateDiskFloorGiB = 2.0 + +// updateSettleWindow is the rule for an app with no .felhom.yml health check: every container +// running, none restarting, for this long. +const updateSettleWindow = 60 * time.Second + +// updatePollEvery is how often the health wait re-reads the stack. +const updatePollEvery = 5 * time.Second + +// UpdateRestorePoint is the precondition answer, reduced to what the update needs. +type UpdateRestorePoint struct { + Restorable bool // an openable recovery unit exists in the Tier-2 copy + Proven bool // a copy actually succeeded (never an attempt clock) + ProvenAt time.Time // when the data in that copy was last proven copied +} + +// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import +// backup, so cmd/controller/main.go wires an adapter (TestSlice4_UpdateGuardsAreWiredAtStartup). +type UpdateGuards interface { + HoldFor(name string) (bool, string) + Busy(name string) (bool, string) + RestorePoint(name string) (UpdateRestorePoint, error) + BackupNow(ctx context.Context, name string) error + SafetyDump(ctx context.Context, name string) ([]string, error) + HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error +} + +// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed): +// an update that cannot see the backup cannot promise a route back. +func (m *Manager) SetUpdateGuards(g UpdateGuards) { + m.mu.Lock() + m.updateGuards = g + m.mu.Unlock() +} + +func (m *Manager) guards() UpdateGuards { + m.mu.RLock() + defer m.mu.RUnlock() + return m.updateGuards +} + +func fillHoldReason(g UpdateGuards, st *Stack) { + if g == nil || st == nil || !st.Deployed { + return + } + if held, why := g.HoldFor(st.Name); held { + st.HoldReason = why + } +} + +// UpdateRefusal is a refusal taken before anything moved. Reason is a stable key for logs and tests; +// Message is the customer sentence. +type UpdateRefusal struct { + Reason string + Message string +} + +func (r *UpdateRefusal) Error() string { return r.Message } + +func (m *Manager) now() time.Time { + if m.updateNowFn != nil { + return m.updateNowFn() + } + return time.Now() +} + +func (m *Manager) refuseUpdate(name, reason, msg, detail string) *UpdateRefusal { + m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail) + return &UpdateRefusal{Reason: reason, Message: msg} +} + +// UpdatePreflight runs every CHEAP refusal (slice 4 Part 1 + the precondition's existence), in order, +// and returns the first. Nothing is moved and nothing is recorded by it. The router calls it before +// recording the customer's intent, so an update that was never going to happen records nothing. +func (m *Manager) UpdatePreflight(name string) *UpdateRefusal { + st, ok := m.GetStack(name) + if !ok { + return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack") + } + if !st.Deployed { + return m.refuseUpdate(name, "not_deployed", MsgUpdateNotDeployed, "not deployed") + } + g := m.guards() + if g == nil { + return m.refuseUpdate(name, "guards_unwired", MsgUpdateNoGuards, "no UpdateGuards wired — fail closed") + } + if st.Deploying { + return m.refuseUpdate(name, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, name), "a deploy is in progress") + } + if st.Updating { + return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "an update is already in progress") + } + if held, why := g.HoldFor(name); held { + return m.refuseUpdate(name, "held", why, "the app is held") + } + if busy, why := g.Busy(name); busy { + return m.refuseUpdate(name, "busy", MsgUpdateBusy, why) + } + if m.IsMigrating() { + return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running") + } + rp, err := g.RestorePoint(name) + if err != nil || !rp.Restorable || !rp.Proven { + return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name), + fmt.Sprintf("no restorable proven Tier-2 unit (restorable=%v proven=%v err=%v)", rp.Restorable, rp.Proven, err)) + } + if ref := m.updateMemoryRefusal(name, st); ref != nil { + return ref + } + free, known := m.updateDiskFree() + switch { + case !known: + m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB) + case free < updateDiskFloorGiB: + return m.refuseUpdate(name, "disk", fmt.Sprintf(MsgUpdateDiskFmt, free, updateDiskFloorGiB), + fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB)) + } + return nil +} + +// updateMemoryRefusal applies the deploy's memory check to the NEW template's request, releasing the +// app's CURRENT request first (an update replaces it). An unknown new request proceeds with a WARN. +func (m *Manager) updateMemoryRefusal(name string, st *Stack) *UpdateRefusal { + catPath := m.CatalogTemplatePath(name, ".felhom.yml") + if _, err := os.Stat(catPath); err != nil { + m.logger.Printf("[WARN] [stacks] update %s: the new template's memory request is unknown (%v) — proceeding without the memory check", name, err) + return nil + } + newMeta := LoadMetadata(filepath.Dir(catPath)) + newReq, newLim := ParseMemoryMB(newMeta.Resources.MemRequest), ParseMemoryMB(newMeta.Resources.MemLimit) + if newReq == 0 { + m.logger.Printf("[WARN] [stacks] update %s: the new template declares no memory request — proceeding without the memory check", name) + return nil + } + oldReq, oldLim := ParseMemoryMB(st.Meta.Resources.MemRequest), ParseMemoryMB(st.Meta.Resources.MemLimit) + verdict := m.updateMemoryFn + if verdict == nil { + verdict = m.memoryVerdict + } + if refusal, _ := verdict(newReq, newLim, oldReq, oldLim); refusal != "" { + return m.refuseUpdate(name, "memory", refusal, fmt.Sprintf("new_req=%dMB replacing %dMB does not fit", newReq, oldReq)) + } + return nil +} + +func (m *Manager) updateDiskFree() (float64, bool) { + if m.updateDiskFreeFn != nil { + return m.updateDiskFreeFn() + } + du := system.GetDiskUsage(system.DockerVolumePath) + if du == nil { + return 0, false + } + return du.AvailGB, true +} + +// StartGuardedUpdate re-checks the cheap refusals, claims the Updating flag atomically and launches the +// job. It returns as soon as the job has STARTED — the result arrives on GET /api/stacks/{name}. +func (m *Manager) StartGuardedUpdate(name string) error { + if ref := m.UpdatePreflight(name); ref != nil { + return ref + } + m.mu.Lock() + s, ok := m.stacks[name] + if !ok { + m.mu.Unlock() + return &UpdateRefusal{Reason: "not_found", Message: fmt.Sprintf("stack %q not found", name)} + } + // A second press between the preflight and here is the race this lock closes. + if s.Updating || s.Deploying { + m.mu.Unlock() + return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag") + } + s.Updating, s.UpdateError = true, "" + s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking) + m.mu.Unlock() + + m.logger.Printf("[INFO] [stacks] update %s: accepted — guarded update started", name) + go m.runGuardedUpdate(context.Background(), name) + return nil +} + +// IsUpdating reports whether a guarded update is in progress for the app. +func (m *Manager) IsUpdating(name string) bool { + m.mu.RLock() + defer m.mu.RUnlock() + s, ok := m.stacks[name] + return ok && s.Updating +} + +// UpdatingStacks is the set of apps an update is currently moving — for the dead-app alarm, which +// must not count an app the update itself is recreating (R-330's class, a third mechanism). +func (m *Manager) UpdatingStacks() map[string]bool { + m.mu.RLock() + defer m.mu.RUnlock() + var out map[string]bool + for name, s := range m.stacks { + if s.Updating { + if out == nil { + out = map[string]bool{} + } + out[name] = true + } + } + return out +} + +func (m *Manager) setUpdatePhase(name, phase string) { + m.mu.Lock() + if s, ok := m.stacks[name]; ok { + s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase) + } + m.mu.Unlock() +} + +// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure. +func (m *Manager) finishUpdate(name, phase, msg string) { + m.mu.Lock() + if s, ok := m.stacks[name]; ok { + s.Updating = false + s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase) + s.UpdateError = msg + } + m.mu.Unlock() +} + +func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) { + if m.updateComposeFn != nil { + return m.updateComposeFn(dir, env, args...) + } + return m.composeExecCustomEnv(dir, env, args...) +} + +func (m *Manager) updateHealth(ctx context.Context, name string, timeout time.Duration) (bool, string) { + if m.updateHealthFn != nil { + return m.updateHealthFn(ctx, name, timeout) + } + return m.waitUpdateHealthy(ctx, name, timeout) +} + +func (m *Manager) healthTimeout() time.Duration { + if m.cfg == nil { + return 5 * time.Minute + } + return m.cfg.Update.HealthTimeoutDuration() +} + +func (m *Manager) backupMaxAge() time.Duration { + if m.cfg == nil { + return 24 * time.Hour + } + return m.cfg.Update.BackupMaxAgeDuration() +} + +// pre-update copies, kept in the stack dir so they travel with it. Neither name is one the syncer +// copies (it copies exactly docker-compose.yml and .felhom.yml). +const ( + preUpdateComposeFile = "pre-update-compose.yml" + preUpdateAppliedFile = "pre-update-applied.yml" +) + +func (m *Manager) runGuardedUpdate(ctx context.Context, name string) { + start := m.now() + st, ok := m.GetStack(name) + if !ok { + m.finishUpdate(name, UpdatePhaseFailed, fmt.Sprintf("stack %q not found", name)) + return + } + dir := filepath.Dir(st.ComposePath) + g := m.guards() + entry := updateJournalEntry{StartedAt: start} + fail := func(msg, detail string) { + m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail) + m.clearJournal(name) + m.finishUpdate(name, UpdatePhaseFailed, msg) + } + + if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) { + m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateJournalFailed) + return + } + if g == nil { + fail(MsgUpdateNoGuards, "no UpdateGuards wired") + return + } + rp, err := g.RestorePoint(name) + if err != nil || !rp.Restorable || !rp.Proven { + fail(fmt.Sprintf(MsgUpdateNoBackupFmt, name), fmt.Sprintf("precondition vanished: restorable=%v proven=%v err=%v", rp.Restorable, rp.Proven, err)) + return + } + + maxAge := m.backupMaxAge() + if age := start.Sub(rp.ProvenAt); age > maxAge { + m.logger.Printf("[INFO] [stacks] update %s: the proven copy is %s old (limit %s) — backing up first", name, age.Round(time.Minute), maxAge) + if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) { + fail(MsgUpdateJournalFailed, "journal write failed") + return + } + if err := g.BackupNow(ctx, name); err != nil { + fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error()) + return + } + rp, err = g.RestorePoint(name) + if err != nil || !rp.Restorable || !rp.Proven || m.now().Sub(rp.ProvenAt) > maxAge { + fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup: restorable=%v proven=%v at=%s err=%v", rp.Restorable, rp.Proven, rp.ProvenAt.Format(time.RFC3339), err)) + return + } + } else { + m.logger.Printf("[INFO] [stacks] update %s: precondition met — proven copy from %s (%s old, limit %s)", name, rp.ProvenAt.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge) + } + entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339) + + // SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run. + if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) { + fail(MsgUpdateJournalFailed, "journal write failed") + return + } + paths, err := g.SafetyDump(ctx, name) + if err != nil { + fail(fmt.Sprintf(MsgUpdateDumpFailFmt, err), "safety dump: "+err.Error()) + return + } + m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths) + + // PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash + // at any later instant can put it back (Scenario G). + prevLive, err := os.ReadFile(st.ComposePath) + if err != nil { + fail(MsgUpdatePinFailed, "reading the live compose file: "+err.Error()) + return + } + if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil { + fail(MsgUpdateJournalFailed, "saving the pre-update compose copy: "+err.Error()) + return + } + entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile) + if applied, aerr := LoadAppliedDefinition(dir); aerr == nil { + if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil { + entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile) + } + } + if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 { + entry.PrevPin = map[string]string{} + for k, v := range cfg.PinnedImages { + entry.PrevPin[k] = v + } + } + if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) { + m.removePreUpdateCopies(dir) + fail(MsgUpdateJournalFailed, "journal write failed") + return + } + if err := m.advancePinToCatalog(name, dir); err != nil { + m.pinBack(name, dir, entry) + fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error()) + return + } + + env := m.stackEnv(dir) + if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) { + m.pinBack(name, dir, entry) + fail(MsgUpdateJournalFailed, "journal write failed") + return + } + if _, err := m.updateCompose(dir, env, "pull"); err != nil { + // Scenario E: NOTHING RAN. The containers are the old ones and still running, so the honest + // state is the old pin and the old file — put both back. + m.pinBack(name, dir, entry) + m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err) + fail(MsgUpdatePullFailed, "pull failed: "+err.Error()) + return + } + + if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) { + m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "journal write failed before up") + return + } + if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil { + // Containers may already have been recreated on the new image — something may have run. + m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "compose up failed: "+err.Error()) + return + } + m.verifyAndConclude(ctx, name, dir, env, rp.ProvenAt, start, &entry) +} + +// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is +// known, and a failure holds the app. +func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, provenAt, start time.Time, entry *updateJournalEntry) { + if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) { + m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name) + } + timeout := m.healthTimeout() + waitStart := m.now() + healthy, detail := m.updateHealth(ctx, name, timeout) + if !healthy { + m.failAndHold(ctx, name, dir, env, provenAt, "not healthy: "+detail) + return + } + m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail) + m.recordInstalledImages(name, dir, env) + _ = m.RefreshStatus() + m.clearJournal(name) + m.removePreUpdateCopies(dir) + m.finishUpdate(name, UpdatePhaseDone, "") + m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second)) +} + +// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back. +func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, provenAt time.Time, why string) { + m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why) + if _, err := m.updateCompose(dir, env, "down"); err != nil { + m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err) + } + msg := MsgUpdateHoldUnsaved + if g := m.guards(); g == nil { + m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name) + } else if err := g.HoldAfterFailedUpdate(name, m.now(), provenAt); err != nil { + m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err) + } else if _, why := g.HoldFor(name); why != "" { + msg = why + } + _ = m.RefreshStatus() + m.clearJournal(name) + m.removePreUpdateCopies(dir) + m.finishUpdate(name, UpdatePhaseFailed, msg) +} + +// pinBack restores the pin, the stored definition and the live file from the journaled copies. +func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) { + prevLive, lerr := os.ReadFile(entry.PrevCompose) + if lerr != nil { + m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr) + } + if len(entry.PrevPin) > 0 { + applied := prevLive + if entry.PrevApplied != "" { + if b, err := os.ReadFile(entry.PrevApplied); err == nil { + applied = b + } + } + if err := m.SetPin(name, dir, entry.PrevPin, applied); err != nil { + m.logger.Printf("[ERROR] [stacks] update %s: putting the pin back failed: %v", name, err) + } + } + if lerr == nil { + if err := os.WriteFile(ComposePathIn(dir), prevLive, 0o644); err != nil { + m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err) + } + } + m.removePreUpdateCopies(dir) + m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin)) +} + +func (m *Manager) removePreUpdateCopies(dir string) { + _ = os.Remove(filepath.Join(dir, preUpdateComposeFile)) + _ = os.Remove(filepath.Join(dir, preUpdateAppliedFile)) +} + +// waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the +// existing probe, or — for an app with none — every container running and none restarting for +// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line. +func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) { + deadline := m.now().Add(timeout) + var runningSince time.Time + last := "no observation yet" + for { + _ = m.RefreshStatus() + st, ok := m.GetStack(name) + switch { + case !ok: + last = "stack vanished" + runningSince = time.Time{} + case st.State == StateRunning: + if hc := st.Meta.HealthCheck; hc != nil && len(hc.Checks) > 0 { + if c := findProbeContainer(name, st.Containers); c != "" { + res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks}) + m.mu.Lock() + if s, ok := m.stacks[name]; ok { + s.HealthProbe = res + } + m.mu.Unlock() + if res.Healthy { + return true, "the app's health check passed" + } + last = "health check failing" + } else { + last = "no probe container" + } + } else { + if runningSince.IsZero() { + runningSince = m.now() + } + if m.now().Sub(runningSince) >= updateSettleWindow { + return true, fmt.Sprintf("all containers running, none restarting, for %s (no health check declared)", updateSettleWindow) + } + last = "running, settling" + } + default: + runningSince = time.Time{} + last = "state " + string(st.State) + } + if !m.now().Before(deadline) { + return false, fmt.Sprintf("not healthy within %s (last: %s)", timeout, last) + } + select { + case <-ctx.Done(): + return false, "cancelled: " + ctx.Err().Error() + case <-time.After(updatePollEvery): + } + } +} + +// ── the journal ────────────────────────────────────────────────────────────────────────────────── + +type updateJournalEntry struct { + Phase string `json:"phase"` + StartedAt time.Time `json:"started_at"` + PrevPin map[string]string `json:"prev_pin,omitempty"` + PrevCompose string `json:"prev_compose,omitempty"` + PrevApplied string `json:"prev_applied,omitempty"` + ProvenCopyAt string `json:"proven_copy_at,omitempty"` +} + +type updateJournal struct { + Updates map[string]updateJournalEntry `json:"updates"` +} + +func (m *Manager) updateJournalPath() string { + return filepath.Join(m.cfg.Paths.DataDir, "update-journal.json") +} + +func (m *Manager) readUpdateJournal() updateJournal { + j := updateJournal{Updates: map[string]updateJournalEntry{}} + data, err := os.ReadFile(m.updateJournalPath()) + if err != nil { + return j + } + if err := json.Unmarshal(data, &j); err != nil { + m.logger.Printf("[WARN] [stacks] update journal at %s is corrupt (%v) — quarantining", m.updateJournalPath(), err) + _ = os.Rename(m.updateJournalPath(), fmt.Sprintf("%s.corrupt-%d", m.updateJournalPath(), time.Now().Unix())) + return updateJournal{Updates: map[string]updateJournalEntry{}} + } + if j.Updates == nil { + j.Updates = map[string]updateJournalEntry{} + } + return j +} + +// writeUpdateJournal is atomic and fsynced (the AppStopGuard shape): the point is surviving a power cut. +func (m *Manager) writeUpdateJournal(j updateJournal) error { + p := m.updateJournalPath() + if len(j.Updates) == 0 { + if err := os.Remove(p); err != nil && !os.IsNotExist(err) { + return err + } + return nil + } + data, err := json.MarshalIndent(j, "", " ") + if err != nil { + return err + } + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + return err + } + tmp := p + ".tmp" + f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600) + if err != nil { + return err + } + if _, err := f.Write(data); err != nil { + f.Close() + os.Remove(tmp) + return err + } + if err := f.Sync(); err != nil { + f.Close() + os.Remove(tmp) + return err + } + if err := f.Close(); err != nil { + os.Remove(tmp) + return err + } + return os.Rename(tmp, p) +} + +// enterUpdatePhase journals the phase BEFORE it starts and mirrors it for the UI. False means the +// journal could not be written — the caller must not perform a mutation it could not record. +func (m *Manager) enterUpdatePhase(name string, entry *updateJournalEntry, phase string) bool { + entry.Phase = phase + m.updateJournalMu.Lock() + j := m.readUpdateJournal() + j.Updates[name] = *entry + err := m.writeUpdateJournal(j) + m.updateJournalMu.Unlock() + m.setUpdatePhase(name, phase) + if err != nil { + m.logger.Printf("[ERROR] [stacks] update %s: journal write for phase %s failed: %v", name, phase, err) + return false + } + m.logger.Printf("[INFO] [stacks] update %s: phase %s", name, phase) + return true +} + +func (m *Manager) clearJournal(name string) { + m.updateJournalMu.Lock() + defer m.updateJournalMu.Unlock() + j := m.readUpdateJournal() + delete(j.Updates, name) + if err := m.writeUpdateJournal(j); err != nil { + m.logger.Printf("[ERROR] [stacks] update %s: clearing the journal entry failed: %v", name, err) + } +} + +// RecoverUpdates reads the journal at startup (Scenario G). Call it BEFORE the boot reconciler. +// +// - interrupted BEFORE the pin moved (checking, backing-up, safety-dump): nothing moved — the entry is +// dropped and the app carries the "interrupted" sentence; +// - interrupted while pinning or pulling: nothing RAN — the pin and definition are put back (as E); +// - interrupted while starting or verifying: something may have run — the app is marked Updating +// (so the boot sweep and the dead-app alarm leave it alone) and queued for ResumeInterruptedUpdates, +// which re-runs `up -d` and the health wait, ending in A or F. +// +// It needs no backup wiring, because nothing here holds an app — that is left to the resumed job. +func (m *Manager) RecoverUpdates() []string { + m.updateJournalMu.Lock() + j := m.readUpdateJournal() + m.updateJournalMu.Unlock() + if len(j.Updates) == 0 { + return nil + } + var resumed []string + for name, e := range j.Updates { + st, ok := m.GetStack(name) + if !ok { + m.logger.Printf("[WARN] [stacks] update recovery: %s is in the journal (phase %s) but no longer exists — dropping the entry", name, e.Phase) + m.clearJournal(name) + continue + } + dir := filepath.Dir(st.ComposePath) + switch e.Phase { + case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump: + m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339)) + m.clearJournal(name) + m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted) + case UpdatePhasePinning, UpdatePhasePulling: + m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339)) + m.pinBack(name, dir, e) + m.clearJournal(name) + m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted) + case UpdatePhaseStarting, UpdatePhaseVerifying: + m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339)) + m.mu.Lock() + if s, ok := m.stacks[name]; ok { + s.Updating, s.UpdateError = true, "" + s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying) + } + m.updateResume = append(m.updateResume, name) + m.mu.Unlock() + resumed = append(resumed, name) + default: + m.logger.Printf("[WARN] [stacks] update recovery: %s has unknown phase %q — dropping the entry", name, e.Phase) + m.clearJournal(name) + } + } + return resumed +} + +// ResumeInterruptedUpdates continues the updates RecoverUpdates queued, once the backup side is wired +// (a resumed update that fails must be able to HOLD). Returns how many were resumed. +func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int { + m.mu.Lock() + names := m.updateResume + m.updateResume = nil + m.mu.Unlock() + for _, name := range names { + st, ok := m.GetStack(name) + if !ok { + m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted) + continue + } + m.updateJournalMu.Lock() + e, ok := m.readUpdateJournal().Updates[name] + m.updateJournalMu.Unlock() + if !ok { + m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted) + continue + } + provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt) + dir := filepath.Dir(st.ComposePath) + go func(name, dir string, e updateJournalEntry, provenAt time.Time) { + env := m.stackEnv(dir) + m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name) + if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil { + m.failAndHold(ctx, name, dir, env, provenAt, "resumed compose up failed: "+err.Error()) + return + } + m.verifyAndConclude(ctx, name, dir, env, provenAt, e.StartedAt, &e) + }(name, dir, e, provenAt) + } + return len(names) +} diff --git a/controller/internal/stacks/update_test.go b/controller/internal/stacks/update_test.go new file mode 100644 index 0000000..47d0d7f --- /dev/null +++ b/controller/internal/stacks/update_test.go @@ -0,0 +1,569 @@ +package stacks + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "sync" + "testing" + "time" +) + +// Update arc slice 4 — the guarded update. Scenarios A–G of the task, through the real job +// (runGuardedUpdate) with the four process boundaries injected: compose, the health wait, the backup +// side (UpdateGuards) and the clock. Every assertion reads the EFFECT back — the pin in app.yaml, the +// bytes of the live compose file, the journal on disk, Updating/UpdatePhase/UpdateError — never "no error". + +var slice4T0 = time.Date(2026, 9, 13, 10, 0, 0, 0, time.UTC) + +type fakeGuards struct { + mu sync.Mutex + calls []string + held bool + holdWhy string + busy bool + rp UpdateRestorePoint + rpErr error + rpAfterBackup *UpdateRestorePoint + backupErr error + dumpErr error + holdErr error + holdProvenAt time.Time + pinAtDump string + stackDir string +} + +func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() } +func (f *fakeGuards) callList() []string { + f.mu.Lock() + defer f.mu.Unlock() + return append([]string(nil), f.calls...) +} +func (f *fakeGuards) HoldFor(string) (bool, string) { + f.mu.Lock() + defer f.mu.Unlock() + return f.held, f.holdWhy +} +func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" } +func (f *fakeGuards) RestorePoint(string) (UpdateRestorePoint, error) { + f.note("RestorePoint") + f.mu.Lock() + defer f.mu.Unlock() + return f.rp, f.rpErr +} +func (f *fakeGuards) BackupNow(context.Context, string) error { + f.note("BackupNow") + f.mu.Lock() + defer f.mu.Unlock() + if f.backupErr == nil && f.rpAfterBackup != nil { + f.rp = *f.rpAfterBackup + } + return f.backupErr +} +func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) { + f.note("SafetyDump") + if cfg := LoadAppConfig(f.stackDir); cfg != nil { + f.mu.Lock() + f.pinAtDump = cfg.PinnedImages["web"] + f.mu.Unlock() + } + return []string{"/fake/pre-restore-x.sql"}, f.dumpErr +} +func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, provenAt time.Time) error { + f.note("HoldAfterFailedUpdate") + f.mu.Lock() + defer f.mu.Unlock() + if f.holdErr != nil { + return f.holdErr + } + f.held, f.holdWhy, f.holdProvenAt = true, "HELD-SENTENCE", provenAt + return nil +} + +type composeRec struct { + mu sync.Mutex + calls []string + fail map[string]error // first arg → error +} + +func (c *composeRec) fn(_ string, _ []string, args ...string) (string, error) { + c.mu.Lock() + defer c.mu.Unlock() + c.calls = append(c.calls, strings.Join(args, " ")) + if err := c.fail[args[0]]; err != nil { + return "", err + } + return "", nil +} +func (c *composeRec) list() []string { + c.mu.Lock() + defer c.mu.Unlock() + return append([]string(nil), c.calls...) +} + +// newSlice4Manager: a pinned nextcloud on the OLD version, the catalog offering the NEW one, fresh +// proven copy, and every boundary faked. Returns the manager, its stack dir, the guards and compose. +func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec) { + t.Helper() + m, dir := newPinManager(t, pinTplOld, pinTplNew, + "deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n") + mustWrite(t, AppliedComposePath(dir), pinTplOld) + g := &fakeGuards{rp: UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Hour)}, stackDir: dir} + c := &composeRec{fail: map[string]error{}} + m.updateGuards = g + m.updateComposeFn = c.fn + m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return true, "fake healthy" } + m.updateMemoryFn = func(int, int, int, int) (string, string) { return "", "" } + m.updateDiskFreeFn = func() (float64, bool) { return 50, true } + m.updateNowFn = func() time.Time { return slice4T0 } // R-457: the SAME clock the age check reads + m.execFn = func(string, ...string) (string, error) { return "", nil } + return m, dir, g, c +} + +func waitUpdateDone(t *testing.T, m *Manager, name string) *Stack { + t.Helper() + deadline := time.Now().Add(5 * time.Second) + for time.Now().Before(deadline) { + if st, ok := m.GetStack(name); ok && !st.Updating { + return st + } + time.Sleep(5 * time.Millisecond) + } + t.Fatal("the update never finished") + return nil +} + +func pinOf(t *testing.T, dir string) string { return readPin(t, dir).PinnedImages["web"] } + +func fileBody(t *testing.T, p string) string { + t.Helper() + b, err := os.ReadFile(p) + if err != nil { + t.Fatal(err) + } + return string(b) +} + +func journalExists(m *Manager) bool { + _, err := os.Stat(m.updateJournalPath()) + return err == nil +} + +// ── A: the happy path, and the truth ───────────────────────────────────────────────────────────── + +// TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth. The health wait BLOCKS until the test releases it; +// while it blocks, the app must read Updating=true / phase=verifying / no error — and the safety dump +// must have run while the pin still named the OLD version. +// +// COMPANION RED-PROOF 1 (REPORT.md): delete the updateHealth call from verifyAndConclude so success is +// declared on the compose exit code. This test then fails at "Updating went false before health was +// known" — which is R-443 exactly. +func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + release := make(chan struct{}) + healthCalled := make(chan struct{}, 1) + m.updateHealthFn = func(ctx context.Context, name string, timeout time.Duration) (bool, string) { + healthCalled <- struct{}{} + <-release + return true, "fake healthy" + } + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatalf("a fully-qualified update must start: %v", err) + } + select { + case <-healthCalled: + case <-time.After(5 * time.Second): + st, _ := m.GetStack("nextcloud") + t.Fatalf("the health wait was never reached; state: updating=%v phase=%s err=%q", st.Updating, st.UpdatePhase, st.UpdateError) + } + st, _ := m.GetStack("nextcloud") + if !st.Updating { + t.Fatal("Updating went false before health was known — success reported on the compose exit code (R-443)") + } + if st.UpdatePhase != UpdatePhaseVerifying || st.UpdatePhaseLabel != "Működés ellenőrzése…" { + t.Errorf("while waiting for health the phase must be verifying, got %q / %q", st.UpdatePhase, st.UpdatePhaseLabel) + } + if st.UpdateError != "" { + t.Errorf("no error may be shown while verifying, got %q", st.UpdateError) + } + if !journalExists(m) { + t.Error("the journal must exist while the update is in flight (Scenario G depends on it)") + } + if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" { + t.Errorf("by verifying, the pin must have advanced; got %q", got) + } + close(release) + st = waitUpdateDone(t, m, "nextcloud") + if st.UpdatePhase != UpdatePhaseDone || st.UpdateError != "" || st.UpdatePhaseLabel != "Frissítve" { + t.Fatalf("after health the update is done: phase=%q label=%q err=%q", st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError) + } + if journalExists(m) { + t.Error("a completed update must clear its journal entry") + } + if _, err := os.Stat(filepath.Join(dir, preUpdateComposeFile)); err == nil { + t.Error("the pre-update copy must be removed after success") + } + if g.pinAtDump != "nextcloud:31.0.14-apache" { + t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump) + } + if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want { + t.Errorf("compose calls = %q, want %q", got, want) + } + // RestorePoint twice by design: once in the preflight (the refusal), once inside the job (the + // precondition must still hold when the job actually starts). + if got := strings.Join(g.callList(), ","); got != "RestorePoint,RestorePoint,SafetyDump" { + t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got) + } +} + +// ── B: the proven copy is too old ───────────────────────────────────────────────────────────────── + +func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) { + m, dir, g, _ := newSlice4Manager(t) + g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default + g.rpAfterBackup = &UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Minute)} + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + st := waitUpdateDone(t, m, "nextcloud") + if st.UpdatePhase != UpdatePhaseDone { + t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError) + } + calls := strings.Join(g.callList(), ",") + if !strings.HasPrefix(calls, "RestorePoint,RestorePoint,BackupNow,RestorePoint,SafetyDump") { + t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls) + } + if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" { + t.Errorf("pin = %q", got) + } +} + +func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) + g.backupErr = errors.New("disk full") + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + st := waitUpdateDone(t, m, "nextcloud") + if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want { + t.Errorf("UpdateError = %q, want the backup's own error in the sentence %q", st.UpdateError, want) + } + if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" { + t.Errorf("a failed backup must move nothing; pin = %q", got) + } + if len(c.list()) != 0 { + t.Errorf("a failed backup must reach no compose call; got %v", c.list()) + } + if journalExists(m) { + t.Error("the journal must be cleared on a refusal") + } +} + +func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: rpAfterBackup nil + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + st := waitUpdateDone(t, m, "nextcloud") + if st.UpdateError != MsgUpdateBackupNoUnit { + t.Errorf("UpdateError = %q", st.UpdateError) + } + if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 { + t.Error("nothing may move when the backup did not produce a fresh restorable copy") + } +} + +// ── C: no backup exists that could restore this app ─────────────────────────────────────────────── + +// COMPANION RED-PROOF 2 (REPORT.md): make the precondition in UpdatePreflight proceed when +// !rp.Restorable. This test then fails with the update started. +func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + // A PROVEN, FRESH copy whose unit cannot be opened — the realistic half-copied mirror. Proven and + // fresh on purpose: a fixture that is also unproven would be refused by the proven check alone, + // and a red-proof that drops the restorable check would then pass inertly (observed on the first + // run of red-proof 2, 2026-09-13). + g.rp = UpdateRestorePoint{Restorable: false, Proven: true, ProvenAt: slice4T0.Add(-time.Hour)} + err := m.StartGuardedUpdate("nextcloud") + var ref *UpdateRefusal + if !errors.As(err, &ref) || ref.Reason != "no_backup" { + t.Fatalf("an app with no restorable copy must be REFUSED (no_backup), got %v", err) + } + if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want { + t.Errorf("message = %q", ref.Message) + } + time.Sleep(50 * time.Millisecond) + if st, _ := m.GetStack("nextcloud"); st.Updating { + t.Error("a refused update must not set Updating") + } + if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 { + t.Error("a refused update must move nothing") + } + // A copy that exists but was never PROVEN is not a copy (R-101). + g.rp = UpdateRestorePoint{Restorable: true, Proven: false} + if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" { + t.Errorf("an unproven copy must refuse too, got %v", ref) + } +} + +// ── D: the cheap refusals, each one ───────────────────────────────────────────────────────────────── + +func TestSlice4_D_CheapRefusals(t *testing.T) { + cases := []struct { + name string + setup func(m *Manager, g *fakeGuards, dir string) + reason string + msg string + }{ + {"held", func(m *Manager, g *fakeGuards, _ string) { g.held, g.holdWhy = true, "THE HOLD TEXT" }, "held", "THE HOLD TEXT"}, + {"busy", func(m *Manager, g *fakeGuards, _ string) { g.busy = true }, "busy", MsgUpdateBusy}, + {"already updating", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Updating = true }, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, "nextcloud")}, + {"deploying", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Deploying = true }, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, "nextcloud")}, + {"memory", func(m *Manager, _ *fakeGuards, _ string) { + catDir := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates", "nextcloud") + if err := os.WriteFile(filepath.Join(catDir, ".felhom.yml"), []byte("resources:\n mem_request: 900M\n"), 0o644); err != nil { + panic(err) + } + m.updateMemoryFn = func(newReq, _, _, _ int) (string, string) { + return fmt.Sprintf("Nincs elég memória (%d MB)", newReq), "" + } + }, "memory", "Nincs elég memória (900 MB)"}, + {"disk", func(m *Manager, _ *fakeGuards, _ string) { + m.updateDiskFreeFn = func() (float64, bool) { return 1.0, true } + }, "disk", fmt.Sprintf(MsgUpdateDiskFmt, 1.0, updateDiskFloorGiB)}, + {"guards unwired", func(m *Manager, _ *fakeGuards, _ string) { m.updateGuards = nil }, "guards_unwired", MsgUpdateNoGuards}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + tc.setup(m, g, dir) + ref := m.UpdatePreflight("nextcloud") + if ref == nil || ref.Reason != tc.reason { + t.Fatalf("want refusal %q, got %+v", tc.reason, ref) + } + if ref.Message != tc.msg { + t.Errorf("message = %q, want %q", ref.Message, tc.msg) + } + if err := m.StartGuardedUpdate("nextcloud"); err == nil { + t.Fatal("StartGuardedUpdate must refuse the same") + } + time.Sleep(20 * time.Millisecond) + if len(c.list()) != 0 || pinOf(t, dir) != "nextcloud:31.0.14-apache" { + t.Errorf("a cheap refusal reached the act: compose=%v pin=%s", c.list(), pinOf(t, dir)) + } + }) + } +} + +// ── E: the pull fails ──────────────────────────────────────────────────────────────────────────── + +func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + c.fail["pull"] = errors.New("exit code 1\nstderr: manifest unknown") + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + st := waitUpdateDone(t, m, "nextcloud") + if st.UpdateError != MsgUpdatePullFailed { + t.Errorf("UpdateError = %q, want the Hungarian sentence and never raw stderr", st.UpdateError) + } + if strings.Contains(st.UpdateError, "manifest unknown") { + t.Error("raw Docker stderr leaked into the customer sentence") + } + if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" { + t.Errorf("the pin must be PUT BACK after a failed pull, got %q", got) + } + if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld { + t.Errorf("the live file must be re-rendered to the old version:\n%s", got) + } + if got := fileBody(t, AppliedComposePath(dir)); got != pinTplOld { + t.Errorf("the stored definition must be the old one again:\n%s", got) + } + if got := strings.Join(c.list(), " | "); got != "pull" { + t.Errorf("after a failed pull nothing else runs; compose calls = %q", got) + } + for _, call := range g.callList() { + if call == "HoldAfterFailedUpdate" { + t.Error("a failed PULL ran nothing and must not hold the app") + } + } +} + +// ── F: the new version does not come up ───────────────────────────────────────────────────────── + +// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test +// then fails: no hold, and the customer is not told the route back. +func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" } + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + st := waitUpdateDone(t, m, "nextcloud") + if st.UpdatePhase != UpdatePhaseFailed { + t.Errorf("phase = %q", st.UpdatePhase) + } + held, _ := g.HoldFor("nextcloud") + if !held { + t.Fatal("an app that did not come up must be HELD") + } + if !g.holdProvenAt.Equal(g.rp.ProvenAt) { + t.Errorf("the hold must name the PROVEN copy date %s, got %s", g.rp.ProvenAt, g.holdProvenAt) + } + if st.UpdateError != "HELD-SENTENCE" { + t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError) + } + if st.HoldReason != "HELD-SENTENCE" { + t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason) + } + if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" { + t.Errorf("the pin must STAY on the new version (its migration may have run), got %q", got) + } + if got := strings.Join(c.list(), " | "); got != "pull | up -d --remove-orphans | down" { + t.Errorf("the failed app must be stopped; compose calls = %q", got) + } + if journalExists(m) { + t.Error("the journal is cleared once the hold (the durable record) is written") + } +} + +func TestSlice4_F_UnsavedHoldSaysSo(t *testing.T) { + m, _, g, _ := newSlice4Manager(t) + m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" } + g.holdErr = errors.New("settings.json read-only") + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + if st := waitUpdateDone(t, m, "nextcloud"); st.UpdateError != MsgUpdateHoldUnsaved { + t.Errorf("an unrecorded hold must be disclosed, got %q", st.UpdateError) + } +} + +// ── G: the controller restarts mid-update ──────────────────────────────────────────────────────── + +func writeTestJournal(t *testing.T, m *Manager, name string, e updateJournalEntry) { + t.Helper() + if err := m.writeUpdateJournal(updateJournal{Updates: map[string]updateJournalEntry{name: e}}); err != nil { + t.Fatal(err) + } +} + +// simulateAdvanced puts the stack in the state a crash AFTER the pin moved would leave: pin, live file +// and stored definition all new, the pre-update copy on disk. +func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry { + t.Helper() + mustWrite(t, filepath.Join(dir, preUpdateComposeFile), pinTplOld) + mustWrite(t, filepath.Join(dir, preUpdateAppliedFile), pinTplOld) + if err := m.advancePinToCatalog("nextcloud", dir); err != nil { + t.Fatal(err) + } + return updateJournalEntry{ + StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"}, + PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile), + ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339), + } +} + +func TestSlice4_G_InterruptedBeforeUpIsPutBack(t *testing.T) { + m, dir, _, c := newSlice4Manager(t) + e := simulateAdvanced(t, m, dir) + e.Phase = UpdatePhasePulling + writeTestJournal(t, m, "nextcloud", e) + + if resumed := m.RecoverUpdates(); len(resumed) != 0 { + t.Fatalf("an update interrupted before `up` is not resumed, got %v", resumed) + } + if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" { + t.Errorf("the pin must be put back, got %q", got) + } + if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld { + t.Errorf("the live file must be the old definition again:\n%s", got) + } + st, _ := m.GetStack("nextcloud") + if st.Updating || st.UpdateError != MsgUpdateInterrupted { + t.Errorf("updating=%v err=%q", st.Updating, st.UpdateError) + } + if journalExists(m) || len(c.list()) != 0 { + t.Error("recovery of a pre-up interruption runs nothing and clears the journal") + } +} + +func TestSlice4_G_InterruptedBeforeThePinIsDroppedUntouched(t *testing.T) { + m, dir, _, _ := newSlice4Manager(t) + writeTestJournal(t, m, "nextcloud", updateJournalEntry{Phase: UpdatePhaseSafetyDump, StartedAt: slice4T0}) + m.RecoverUpdates() + if pinOf(t, dir) != "nextcloud:31.0.14-apache" || journalExists(m) { + t.Error("an update interrupted before the pin moved must leave the pin and clear the journal") + } +} + +func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) { + m, dir, g, c := newSlice4Manager(t) + e := simulateAdvanced(t, m, dir) + e.Phase = UpdatePhaseVerifying + writeTestJournal(t, m, "nextcloud", e) + guards := m.updateGuards + m.updateGuards = nil // at RecoverUpdates time the backup side is NOT wired yet (main.go order) + + resumed := m.RecoverUpdates() + if len(resumed) != 1 || !m.IsUpdating("nextcloud") || !m.UpdatingStacks()["nextcloud"] { + t.Fatalf("an update interrupted after `up` must be marked Updating for the boot sweep; resumed=%v", resumed) + } + if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" { + t.Errorf("after `up` the pin is NOT put back — something may have run; got %q", got) + } + + m.updateGuards = guards + m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "still broken" } + if n := m.ResumeInterruptedUpdates(context.Background()); n != 1 { + t.Fatalf("resumed %d, want 1", n) + } + st := waitUpdateDone(t, m, "nextcloud") + if held, _ := g.HoldFor("nextcloud"); !held || st.UpdatePhase != UpdatePhaseFailed { + t.Errorf("a resumed update that is still unhealthy must end HELD; held=%v phase=%q", held, st.UpdatePhase) + } + if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | down" { + t.Errorf("resumption re-runs `up` then stops the failed app; compose calls = %q", got) + } + if !g.holdProvenAt.Equal(slice4T0.Add(-time.Hour)) { + t.Errorf("the resumed hold must name the journaled proven copy date, got %s", g.holdProvenAt) + } +} + +// ── the page reads the hold from the ONE store ────────────────────────────────────────────────── + +func TestSlice4_GetStacksCarriesTheHoldText(t *testing.T) { + m, _, g, _ := newSlice4Manager(t) + g.held, g.holdWhy = true, "HOLD" + for _, st := range m.GetStacks() { + if st.Name == "nextcloud" && st.HoldReason != "HOLD" { + t.Errorf("GetStacks HoldReason = %q", st.HoldReason) + } + } + g.held = false + if st, _ := m.GetStack("nextcloud"); st.HoldReason != "" { + t.Errorf("a lifted hold must disappear on the next read, got %q", st.HoldReason) + } +} + +func TestSlice4_PhaseLabelsAreTheSpecifiedCopy(t *testing.T) { + want := map[string]string{ + UpdatePhaseChecking: "Ellenőrzés…", + UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…", + UpdatePhaseSafetyDump: "Adatbázis pillanatkép…", + UpdatePhasePulling: "Új verzió letöltése…", + UpdatePhaseStarting: "Indítás az új verzióval…", + UpdatePhaseVerifying: "Működés ellenőrzése…", + UpdatePhaseDone: "Frissítve", + } + for p, l := range want { + if got := UpdatePhaseLabel(p); got != l { + t.Errorf("label(%s) = %q, want %q", p, got, l) + } + } +} diff --git a/controller/internal/web/handlers.go b/controller/internal/web/handlers.go index bdd0f81..5f0f7c6 100644 --- a/controller/internal/web/handlers.go +++ b/controller/internal/web/handlers.go @@ -1429,12 +1429,15 @@ func (s *Server) buildAppBackupRows(status *backup.FullBackupStatus) []AppBackup // a disconnected destination, a pre-v2 layout — so an offer is never rendered for a // copy the action would refuse. On any refusal the action is simply not offered; the // row keeps rendering everything else it already showed. - if cov, covErr := s.backupMgr.Tier2RestoreCoverage(app.StackName); covErr == nil { - row.Tier2UnitRestorable = cov.CanRestoreUnit() + // + // Slice 4: the computation lives in backup.Tier2UnitRestorePoint, because the guarded + // update asks the same question and a second copy of a predicate drifts (R-203). + if rp, rpErr := s.backupMgr.Tier2UnitRestorePoint(app.StackName); rpErr == nil { + row.Tier2UnitRestorable = rp.Restorable // R-403: the UNIT action names the PACKAGE's date, not the run's. After a // preserved leg those are different dates and the run's is the flattering one. - pkgDate, stale := cov.UnitRestoreDate() - row.Tier2CopyDate, row.Tier2CopyDateProven = pkgDate, cov.CopyLastSuccess != "" + pkgDate, stale := rp.CopyDate, rp.PackagePreserved + row.Tier2CopyDate, row.Tier2CopyDateProven = pkgDate, rp.CopyDateProven row.Tier2UnitConfirm = tier2UnitConfirmWithStaleness(pkgDate, row.Tier2CopyDateProven, stale) if stale && pkgDate != "" { row.Tier2UnitStaleNotice = fmt.Sprintf(tier2UnitStaleNoticeFmt, fmtRFC3339Local(pkgDate)) diff --git a/controller/internal/web/intermediary.go b/controller/internal/web/intermediary.go index a64a7e2..395b987 100644 --- a/controller/internal/web/intermediary.go +++ b/controller/internal/web/intermediary.go @@ -219,6 +219,12 @@ func (s *Server) stopAppsOnPath(storagePath string) []string { // restartStacks starts each named stack (the gate-stopped set on drive return). Best-effort per app. func (s *Server) restartStacks(names []string) { for _, name := range names { + // Slice 4 / R-379: the drive-return gate starts apps UNATTENDED, so it must honour a hold + // exactly as the customer's button does. Until v0.237.0 it did not — a held app whose drive + // blinked would have been started again. "A hold that only one path honours is not a hold." + if s.appHeld(name) { + continue + } if err := s.stackMgr.StartStack(name); err != nil { s.logger.Printf("[WARN] [gate] restart %s: %v", name, err) } @@ -454,6 +460,9 @@ func (s *Server) processGuestBootChange() { } recreate := func(bs bootStack) { s.logger.Printf("[INFO] [gate] boot %s: live bind confirmed — recreating drive-backed app %s (state=%s) onto %s", resp.GuestBootID, bs.name, bs.state, bs.hdd) + if s.appHeld(bs.name) { + return + } _ = s.stackMgr.StopStack(bs.name) if serr := s.stackMgr.StartStack(bs.name); serr != nil { s.logger.Printf("[WARN] [gate] boot recreate %s: %v", bs.name, serr) @@ -694,3 +703,16 @@ func (s *Server) notifyDriveReturned(path string, isTarget map[string]bool) { } s.notifier.NotifyStorageReconnected(label) } + +// appHeld reports whether an app carries a hold (failed update or failed restore) and logs the skip. +// Nil-safe: no backup manager means no hold store, so nothing is held. +func (s *Server) appHeld(name string) bool { + if s.backupMgr == nil { + return false + } + held, _ := s.backupMgr.RestoreHoldFor(name) + if held { + s.logger.Printf("[WARN] [gate] NOT starting %s — the app is HELD (a failed update or restore); a person releases it", name) + } + return held +} diff --git a/controller/internal/web/slice4_update_test.go b/controller/internal/web/slice4_update_test.go new file mode 100644 index 0000000..aeb2d5d --- /dev/null +++ b/controller/internal/web/slice4_update_test.go @@ -0,0 +1,74 @@ +package web + +import ( + "os" + "path/filepath" + "testing" + "time" + + "gitea.dooplex.hu/admin/felhom-controller/internal/backup" + "gitea.dooplex.hu/admin/felhom-controller/internal/settings" +) + +// Slice 4 — the predicate moved from this package into backup.Tier2UnitRestorePoint. The page must +// render IDENTICALLY: the row's four unit-restore fields are compared against the coverage computed +// the old way (the inline expression that used to live in buildAppBackupRows). +func TestSlice4_BackupRowUnitFieldsAreUnchangedByTheExtraction(t *testing.T) { + s, sett, m := newOffboxWebServer(t) + dest := t.TempDir() + if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash"}); err != nil { + t.Fatal(err) + } + if err := sett.SetCrossDriveConfig("app", &settings.CrossDriveBackup{ + Enabled: true, Method: "rsync", DestinationPath: dest, + LastRun: "2026-09-13T01:30:00Z", LastStatus: "ok", LastSuccess: "2026-09-13T01:30:00Z", SuccessTracked: true, + }); err != nil { + t.Fatal(err) + } + unit := filepath.Join(dest, "backups", "secondary", "app", "recovery-unit") + if err := os.MkdirAll(unit, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(unit, "manifest.json"), []byte(`{"schema_version":2,"app_name":"app","created_at":"2026-09-12T02:15:29Z"}`), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dest, "backups", "secondary", "app", ".felhom-tier2-layout"), []byte("2"), 0o644); err != nil { + t.Fatal(err) + } + + cov, err := m.Tier2RestoreCoverage("app") + if err != nil { + t.Fatal(err) + } + wantDate, wantStale := cov.UnitRestoreDate() + rows := s.buildAppBackupRows(&backup.FullBackupStatus{AppDataInfo: []backup.AppBackupInfo{{StackName: "app", DisplayName: "App"}}}) + row := findRow(rows, "app") + if row == nil { + t.Fatal("no row") + } + if !cov.CanRestoreUnit() { + t.Fatal("fixture: the copy must hold an openable unit") + } + if row.Tier2UnitRestorable != cov.CanRestoreUnit() || row.Tier2CopyDate != wantDate || + row.Tier2CopyDateProven != (cov.CopyLastSuccess != "") || + row.Tier2UnitConfirm != tier2UnitConfirmWithStaleness(wantDate, cov.CopyLastSuccess != "", wantStale) { + t.Errorf("the row changed: restorable=%v date=%q proven=%v confirm=%q", row.Tier2UnitRestorable, row.Tier2CopyDate, row.Tier2CopyDateProven, row.Tier2UnitConfirm) + } +} + +// The drive-return gate starts apps UNATTENDED. A held app must be skipped. +// +// COMPANION RED-PROOF (REPORT.md): delete the appHeld check from restartStacks. The server has NO stack +// manager here on purpose, so the unguarded StartStack call panics and this test fails. +func TestSlice4_DriveReturnGateSkipsAHeldApp(t *testing.T) { + s, _, m := newOffboxWebServer(t) + if err := m.HoldAfterFailedUpdate("held", time.Now(), time.Now()); err != nil { + t.Fatal(err) + } + defer func() { + if r := recover(); r != nil { + t.Fatalf("the drive-return gate tried to START a held app: %v", r) + } + }() + s.restartStacks([]string{"held"}) +}