package stacks import ( "context" "encoding/json" "fmt" "os" "path/filepath" "strings" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/i18n" "gitea.dooplex.hu/admin/felhom-controller/internal/system" "gitea.dooplex.hu/admin/felhom-controller/internal/util" ) // ── The guarded update (update arc slice 4, controller v0.237.0) ───────────────────────────────── // // WHAT IT REPLACED. `UpdateStack` advanced the pin, pulled, ran `up -d` and returned — no copy first, // no check of memory, disk, a running backup or a held app, and a success the moment `up` returned. // SPIKE-app-update-2026-09-01 §4 measured that as HTTP 200 over an app that was already crash-looping // (R-443), and R-439 is that a held app could be updated at all. // // THE SEQUENCE, and the order is the point: // // checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling // → copying → starting → verifying → done | (undoing → undone | failed) // // 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02, // 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The // same predicate that permits the destructive „Teljes visszaállítás" permits the update — the // route back IS that restore, so an update without it has no route back. // 2. The safety dump is taken BEFORE the pin moves: it is "the state the customer was in a minute ago", // and a minute later the migration may have run. // 3. The pin moves BEFORE the pull (v0.235.0's reason: pull and up act on the file on disk). // 4. A PULL failure puts the pin BACK — nothing ran, so reverting is safe and honest (Scenario E). // 5. A HEALTH failure leaves the pin where it is — the new version's migration may have run, and a // pin claiming the old version would be a record of something untrue (Scenario F). The app is // HELD STOPPED and the customer is told which backup it can be restored from. // // SINCE v0.263.0 IT PUTS THE OLD VERSION BACK ITSELF — with the data from before the update (09 §3 // decision 15, undo.go). Until then it deliberately did not: SPIKE-upgrade-test-2026-09-06 measured // that the old image alone refuses data a new one migrated (Docmost, Nextcloud). The undo does not put // the old image on migrated data; it puts the pre-migration data back as well. A HOLD now happens only // when that undo fails too. // // CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10), // so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at // the next startup (Scenario G) — the AppStopGuard pattern. // Update phases, as recorded in the journal and served as Stack.UpdatePhase. const ( UpdatePhaseChecking = "checking" UpdatePhaseBackingUp = "backing-up" UpdatePhaseSafetyDump = "safety-dump" UpdatePhasePinning = "pinning" UpdatePhasePulling = "pulling" UpdatePhaseStarting = "starting" UpdatePhaseVerifying = "verifying" UpdatePhaseDone = "done" UpdatePhaseFailed = "failed" // UpdatePhaseCopying, UpdatePhaseUndoing and UpdatePhaseUndone live in undo.go. ) // updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the // specification — it is instantaneous and is the first step of fetching the new version, so it shares // the pull's label rather than inventing a sentence nobody would read. var updatePhaseLabels = map[string]string{ UpdatePhaseChecking: "Ellenőrzés…", UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…", UpdatePhaseSafetyDump: "Adatbázis pillanatkép…", UpdatePhasePinning: "Új verzió letöltése…", UpdatePhasePulling: "Új verzió letöltése…", UpdatePhaseStarting: "Indítás az új verzióval…", UpdatePhaseVerifying: "Működés ellenőrzése…", UpdatePhaseDone: "Frissítve", UpdatePhaseFailed: "A frissítés nem sikerült", // v0.263.0 — born as bundle keys (update.phase.*); TestUndo_PhaseLabelsMatchTheBundle pins them equal. UpdatePhaseCopying: "Az adatok másolása a frissítés előtt…", UpdatePhaseUndoing: "Visszaállítás az előző változatra…", UpdatePhaseUndone: "Visszaállítva az előző változatra", } // UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one. func UpdatePhaseLabel(phase string) string { return updatePhaseLabels[phase] } // Customer sentences. Named so tests compare against the constant, never a retyped literal (R-364). const ( MsgUpdateNoGuards = "A frissítés nem indítható: a frissítés előtti biztonsági ellenőrzés nem érhető el ezen a szerveren." MsgUpdateNotDeployed = "Az alkalmazás nincs telepítve, ezért nem frissíthető." MsgUpdateDeployingFmt = "A(z) %s telepítése még folyamatban van — a frissítés utána indítható." MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van." MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött." MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban." MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani, és most új mentés sem készíthető róla. Ellenőrizd a Mentések oldalon, hogy az alkalmazás meghajtója elérhető-e — utána a frissítés elindítható." MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges." MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább." MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább." MsgUpdateDumpFailFmt = "A frissítés nem indult el, mert az adatbázis pillanatkép nem készült el: %v. Az alkalmazás változatlanul fut tovább." MsgUpdatePinFailed = "A frissítés nem indult el: az új verzió leírása nem olvasható be. Az alkalmazás változatlanul fut tovább." MsgUpdateJournalFailed = "A frissítés nem indult el: a frissítés naplója nem menthető. Az alkalmazás változatlanul fut tovább." MsgUpdatePullFailed = "Az új verzió letöltése nem sikerült, ezért a frissítés elmaradt. Az alkalmazás a korábbi verzióval fut tovább." MsgUpdateInterrupted = "A frissítés megszakadt, mert a vezérlő újraindult, mielőtt az új verzió elindult volna. Az alkalmazás a korábbi verzióval fut tovább." MsgUpdateHoldUnsaved = "A frissítés nem sikerült, az alkalmazás le lett állítva, de a leállítás rögzítése nem sikerült. Ne indítsd újra — vedd fel velünk a kapcsolatot." ) // updateDiskFloorGiB is the free space the Docker data root must have before a pull. A FIXED FLOOR, // stated as such: the new image set's size is not known without a registry query (the catalog // records tags, not sizes, and §8.1 of 09 already declines registry calls on the customer box), so // the rule is "not less than 2 GB", not "enough for these images". const updateDiskFloorGiB = 2.0 // updateSettleWindow is the rule for an app with no .felhom.yml health check: every container // running, none restarting, for this long. const updateSettleWindow = 60 * time.Second // updatePollEvery is how often the health wait re-reads the stack. const updatePollEvery = 5 * time.Second // Backup tiers (R-475), mirroring backup.UpdateTier* — stacks cannot import backup, so // TestR475_TierConstantsAgree (cmd/controller) pins the two sets equal. const ( UpdateTierLocal = 1 // the app's own recovery unit UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive UpdateTierOffsite = 3 // off-site ) // UpdateRestorePoint is one proven, restorable copy the update may lean on. The backup side returns // only proven, restorable copies (never an attempt clock, never an unopenable unit), so there is no // "maybe" field here to forget to check. type UpdateRestorePoint struct { Tier int // UpdateTierSecondDrive / UpdateTierLocal / UpdateTierOffsite ProvenAt time.Time // when the data in that copy was last proven written } func updateTierName(tier int) string { switch tier { case UpdateTierSecondDrive: return "Tier 2 (second drive)" case UpdateTierLocal: return "Tier 1 (own recovery unit)" case UpdateTierOffsite: return "Tier 3 (off-site)" } return fmt.Sprintf("tier %d", tier) } // freshRestorePoint is THE age rule (backup_max_age), applied to whichever tier is being considered // — R-475 Scenario M: a stale copy on ANY tier is stale. // // COMPANION RED-PROOF M (REPORT.md): check the age only for Tier 2 (let any other tier through // whatever its age). TestR475_M_TheAgeRuleAppliesToTheChosenTier then fails: a 30-hour-old copy of // the app's own unit carries the update with no backup first. func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePoint) bool { return func(p UpdateRestorePoint) bool { return !p.ProvenAt.IsZero() && now.Sub(p.ProvenAt) <= maxAge } } // usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's // deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp // 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update // at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also // rewrites deployed_at, so the update after a restore backs up first — slower, never less safe. // COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails. func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool { fresh := freshRestorePoint(now, maxAge) return func(p UpdateRestorePoint) bool { if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) { return false } return fresh(p) } } // currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable. func (m *Manager) currentDeployTime(name string) time.Time { st, ok := m.GetStack(name) if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" { return time.Time{} } t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt) if err != nil { return time.Time{} } return t } // ClearUpdateState wipes everything a finished-or-abandoned update left on a stack. R-614: a fresh // install of an app that had previously failed an update showed the OLD phase — „Frissítve" on a // deploy that had just happened — because a remove cleared the directory and not the in-memory // record. The name is the only thing the next install shares with the last one, so the record has to // go when the app does. func (m *Manager) ClearUpdateState(name string) { m.mu.Lock() defer m.mu.Unlock() s, ok := m.stacks[name] if !ok { return } s.Updating = false s.UpdatePhase = "" s.UpdatePhaseLabel = "" s.UpdateError = "" s.updateHeld = false s.HealthProbe = nil } func (m *Manager) markUpdateHeld(name string) { m.mu.Lock() if s, ok := m.stacks[name]; ok { s.updateHeld = true } m.mu.Unlock() } func fmtDeployTime(t time.Time) string { if t.IsZero() { return "unknown" } return t.UTC().Format(time.RFC3339) } func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string { if len(pts) == 0 { return "none" } parts := make([]string, 0, len(pts)) for _, p := range pts { parts = append(parts, fmt.Sprintf("%s at %s (%s old)", updateTierName(p.Tier), p.ProvenAt.UTC().Format(time.RFC3339), now.Sub(p.ProvenAt).Round(time.Minute))) } return strings.Join(parts, "; ") } // UpdateGuards is everything the update needs from the backup side. The stacks package cannot import // backup, so cmd/controller/main.go wires an adapter (TestSlice4_UpdateGuardsAreWiredAtStartup). type UpdateGuards interface { HoldFor(name string) (bool, string) Busy(name string) (bool, string) // RestorePoints walks the tiers in preference order (2, 1, 3) and returns the first copy accept // admits (nil = any), whether one was found, and every copy looked at (R-475). RestorePoints(ctx context.Context, name string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint) // CanBackUp reports whether "back up first" can run for this app now (R-475 Scenario L). CanBackUp(name string) (bool, string) BackupNow(ctx context.Context, name string) error SafetyDump(ctx context.Context, name string) ([]string, error) // HoldAfterFailedUpdate records the hold. undoState (v0.263.0) says what a FAILED undo left the data // as (UndoState*), "" when no undo was attempted; the hold sentence says so. HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint, undoState string) error } // SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed): // an update that cannot see the backup cannot promise a route back. func (m *Manager) SetUpdateGuards(g UpdateGuards) { m.mu.Lock() m.updateGuards = g m.mu.Unlock() } // SetSelfUpdatingCheck wires the OTHER half of the v0.261.0 lock: is the CONTROLLER swapping itself? // // A plain callback rather than a member of UpdateGuards, for two reasons. UpdateGuards is the BACKUP // side's interface and this has nothing to do with backups; and `stacks` must never import // `selfupdate` (selfupdate reaches the agent, and the import would run the wrong way), so the // dependency is inverted here and satisfied in main.go with `updater.IsUpdateRunning`. // // Nil is safe and means the pre-v0.261.0 behaviour: no self-update gate. func (m *Manager) SetSelfUpdatingCheck(fn func() bool) { m.mu.Lock() m.selfUpdating = fn m.mu.Unlock() } func (m *Manager) selfUpdatingNow() bool { m.mu.RLock() fn := m.selfUpdating m.mu.RUnlock() return fn != nil && fn() } func (m *Manager) guards() UpdateGuards { m.mu.RLock() defer m.mu.RUnlock() return m.updateGuards } func fillHoldReason(g UpdateGuards, st *Stack) { if st == nil { return } held := false if g != nil && st.Deployed { if h, why := g.HoldFor(st.Name); h { st.HoldReason, held = why, true } } // R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold // is lifted — a successful restore — or the app is removed, the sentence says a running (or absent) // app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore. // A failure that held nothing (a pull failure) keeps its sentence: it is still true. // COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails. if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) { st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", "" } } // UpdateRefusal is a refusal taken before anything moved. Reason is a stable key for logs and tests; // Message is the customer sentence. type UpdateRefusal struct { Reason string Message string // Cause carries the refusal as an ERROR when the producer made one (v0.253.0, R-557). A // util.MsgError there knows its bundle key, so `api.Router.errText` renders the refusal in the // household's language; Message stays the Hungarian fallback for every refusal built from a // literal. Unwrap is what lets errors.Is and util.AsMsg see through this wrapper. Cause error } func (r *UpdateRefusal) Error() string { if r.Cause != nil { return r.Cause.Error() } return r.Message } func (r *UpdateRefusal) Unwrap() error { return r.Cause } func (m *Manager) now() time.Time { if m.updateNowFn != nil { return m.updateNowFn() } return time.Now() } func (m *Manager) refuseUpdate(name, reason, msg, detail string) *UpdateRefusal { m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail) return &UpdateRefusal{Reason: reason, Message: msg} } // refuseUpdateErr is refuseUpdate for a refusal the producer already built as an error — it keeps the // error whole, so its kind and its message key both survive to the API. func (m *Manager) refuseUpdateErr(name, reason string, cause error, detail string) *UpdateRefusal { m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail) return &UpdateRefusal{Reason: reason, Message: cause.Error(), Cause: cause} } // UpdatePreflight runs every CHEAP refusal (slice 4 Part 1 + the precondition's existence), in order, // and returns the first. Nothing is moved and nothing is recorded by it. The router calls it before // recording the customer's intent, so an update that was never going to happen records nothing. func (m *Manager) UpdatePreflight(name string) *UpdateRefusal { st, ok := m.GetStack(name) if !ok { return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack") } if !st.Deployed { return m.refuseUpdateErr(name, "not_deployed", util.MsgError("update.refusal.not_deployed"), "not deployed") } g := m.guards() if g == nil { return m.refuseUpdateErr(name, "guards_unwired", util.MsgError("update.error.no_guards"), "no UpdateGuards wired — fail closed") } if st.Deploying { return m.refuseUpdateErr(name, "deploying", util.MsgError("update.refusal.deploying", name), "a deploy is in progress") } if st.Updating { return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "an update is already in progress") } if held, why := g.HoldFor(name); held { return m.refuseUpdate(name, "held", why, "the app is held") } if busy, why := g.Busy(name); busy { return m.refuseUpdateErr(name, "busy", util.MsgError("update.refusal.busy"), why) } if m.IsMigrating() { return m.refuseUpdateErr(name, "migrating", util.MsgError("update.refusal.migrating"), "a data migration is running") } // v0.261.0 — the other half of the self-update lock. The controller's swap restarts this process; // starting an app update into that is how an update loses its own supervisor mid-flight. TRANSIENT: // the household is told to try again in a few minutes, and the caller in `09` §6.2 reads the // machine-readable `downgrade`-style reason and retries rather than giving up. if m.selfUpdatingNow() { return m.refuseUpdateErr(name, "self_updating", util.MsgError("err.stacks.update_self_updating"), "the controller is swapping itself — refusing to start an app update into a restart") } // R-524 — THE PIN NEVER MOVES BACKWARDS WITHOUT THE OPERATOR. MEASURED 2026-09-15 (BIGNIGHT // Phase 6): privatebin was updated 2.0.5 → 2.0.6, the catalog was reverted to 2.0.5, and the // „Frissítés" button behind the badge would have advanced the pin to the OLDER image — on a // datadir the newer version may already have migrated, with §4's ruling saying that cannot be // undone. A catalog revert is an operator act on our side; it must never become a data event on // the customer's side by itself. // // It refuses ONLY the provable case (stacks.CatalogOrder's Ahead arm: every differing service // orderable and newer). Anything unorderable, mixed or equal falls through to the behaviour that // shipped in v0.237.0 — this gate can block an update, so it errs towards letting one run. // // COMPANION RED-PROOF (REPORT.md): make CatalogOrder's Ahead arm return Behind. // TestR524_PreflightRefusesDowngrade then fails — the update is allowed to move the pin back. if CatalogOrder(*st) == UpdateOrderAhead { return m.refuseUpdateErr(name, "downgrade", util.MsgError("err.stacks.update_downgrade"), fmt.Sprintf("installed is provably NEWER than the catalog on every differing service (installed=%v catalog=%v)", st.AppConfig.InstalledImages, st.CatalogImages)) } // R-475: any tier counts, and an app with no copy at all is backed up first by the job. So the only // refusal left here is Scenario L — no copy on any tier AND no way to make one now. (With a copy // but no way to back up, the job still applies the age rule and refuses then if the copy is stale.) // Tier 3 is looked at only on this branch, so an ordinary update never waits on the network here. // // COMPANION RED-PROOF (REPORT.md): drop the `!found` condition. TestR475_L then fails — an app // with a copy but no way to back up is refused too. if canBackUp, why := g.CanBackUp(name); !canBackUp { if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found { return m.refuseUpdateErr(name, "no_backup", util.MsgError("update.refusal.no_backup", name), fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why)) } m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why) } if ref := m.updateMemoryRefusal(name, st); ref != nil { return ref } free, known := m.updateDiskFree() switch { case !known: m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB) case free < updateDiskFloorGiB: return m.refuseUpdateErr(name, "disk", util.MsgError("update.refusal.disk", free, updateDiskFloorGiB), fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB)) } return nil } // updateMemoryRefusal applies the deploy's memory check to the NEW template's request, releasing the // app's CURRENT request first (an update replaces it). An unknown new request proceeds with a WARN. func (m *Manager) updateMemoryRefusal(name string, st *Stack) *UpdateRefusal { catPath := m.CatalogTemplatePath(name, ".felhom.yml") if _, err := os.Stat(catPath); err != nil { m.logger.Printf("[WARN] [stacks] update %s: the new template's memory request is unknown (%v) — proceeding without the memory check", name, err) return nil } newMeta := LoadMetadata(filepath.Dir(catPath)) newReq, newLim := ParseMemoryMB(newMeta.Resources.MemRequest), ParseMemoryMB(newMeta.Resources.MemLimit) if newReq == 0 { m.logger.Printf("[WARN] [stacks] update %s: the new template declares no memory request — proceeding without the memory check", name) return nil } oldReq, oldLim := ParseMemoryMB(st.Meta.Resources.MemRequest), ParseMemoryMB(st.Meta.Resources.MemLimit) verdict := m.updateMemoryFn if verdict == nil { verdict = m.memoryVerdict } if refusal, _ := verdict(newReq, newLim, oldReq, oldLim); refusal != nil { return m.refuseUpdateErr(name, "memory", refusal, fmt.Sprintf("new_req=%dMB replacing %dMB does not fit", newReq, oldReq)) } return nil } func (m *Manager) updateDiskFree() (float64, bool) { if m.updateDiskFreeFn != nil { return m.updateDiskFreeFn() } du := system.GetDiskUsage(system.DockerVolumePath) if du == nil { return 0, false } return du.AvailGB, true } // StartGuardedUpdate re-checks the cheap refusals, claims the Updating flag atomically and launches the // job. It returns as soon as the job has STARTED — the result arrives on GET /api/stacks/{name}. func (m *Manager) StartGuardedUpdate(name string) error { if ref := m.UpdatePreflight(name); ref != nil { return ref } m.mu.Lock() s, ok := m.stacks[name] if !ok { m.mu.Unlock() return &UpdateRefusal{Reason: "not_found", Message: fmt.Sprintf("stack %q not found", name)} } // A second press between the preflight and here is the race this lock closes. if s.Updating || s.Deploying { m.mu.Unlock() return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "lost the race for the Updating flag") } s.Updating, s.UpdateError, s.updateHeld = true, "", false s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking) m.mu.Unlock() m.logger.Printf("[INFO] [stacks] update %s: accepted — guarded update started", name) go m.runGuardedUpdate(context.Background(), name) return nil } // IsUpdating reports whether a guarded update is in progress for the app. func (m *Manager) IsUpdating(name string) bool { m.mu.RLock() defer m.mu.RUnlock() s, ok := m.stacks[name] return ok && s.Updating } // AnyUpdating reports whether a guarded update is in flight for ANY app. // // ── WHY IT EXISTS (v0.261.0) ──────────────────────────────────────────────────────────────────── // // The CONTROLLER updates itself too — daily at `self_update.auto_update_time` (04:30 by default) and, // once the hub serves a floor above this box, after any report, at any hour. That swap restarts the // controller container. Until v0.261.0 its ONLY busy gate was `backupRunning`, so a self-update could // land in the middle of a guarded app update. // // MEASURED, and the gap is narrower than it looks but real: the update's `backing-up` phase takes the // backup single-flight (`RunAppBackupNow` → `acquireRunning`), so `backupMgr.IsRunning()` ALREADY // covered that one phase. It covers none of the others — `checking`, `safety-dump`, `pinning`, // `pulling`, `starting`, `verifying` — and `starting`/`verifying` are exactly where the new version // may already have touched the app's data. // // ⚠ IT MUST ANSWER FALSE FOR A HELD APP. `Stack.Updating` is cleared by `finishUpdate` on done, // failed AND held, so a held app does not hold this lock — otherwise one app that cannot come up // would block the controller's own updates for ever, which is a worse failure than the one this // prevents. TestR608_LockReleasesAfterHold pins that. func (m *Manager) AnyUpdating() bool { m.mu.RLock() defer m.mu.RUnlock() for _, s := range m.stacks { if s.Updating { return true } } return false } // UpdatingStacks is the set of apps an update is currently moving — for the dead-app alarm, which // must not count an app the update itself is recreating (R-330's class, a third mechanism). func (m *Manager) UpdatingStacks() map[string]bool { m.mu.RLock() defer m.mu.RUnlock() var out map[string]bool for name, s := range m.stacks { if s.Updating { if out == nil { out = map[string]bool{} } out[name] = true } } return out } func (m *Manager) setUpdatePhase(name, phase string) { m.mu.Lock() if s, ok := m.stacks[name]; ok { s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase) } m.mu.Unlock() } // finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure — a // finished one (the hold's own sentence, or none). A sentence the job owns goes through finishUpdateKey. func (m *Manager) finishUpdate(name, phase, msg string) { m.finishUpdateKey(name, phase, "", msg) } // finishUpdateKey is finishUpdate with the sentence as a bundle KEY (v0.264.0, R-606): UpdateError // keeps the Hungarian (byte-identical to the MsgUpdate* literal it replaced), and the key + args ride // beside it for the page. key "" = `plain` is a finished sentence and is stored as it is. func (m *Manager) finishUpdateKey(name, phase, key, plain string, args ...interface{}) { msg := plain if key != "" { msg = util.Text(i18n.Default, key, args...) } m.mu.Lock() if s, ok := m.stacks[name]; ok { s.Updating = false s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase) s.UpdateError, s.UpdateErrorKey, s.UpdateErrorArgs = msg, key, args } m.mu.Unlock() } // UpdatePhaseLabelIn is a phase's label in lang (v0.264.0, R-606). Hungarian is the updatePhaseLabels // map itself; another language reads `update.phase.` and falls back to the Hungarian. func UpdatePhaseLabelIn(lang, phase string) string { if lang == i18n.Default || phase == "" { return UpdatePhaseLabel(phase) } if b, err := i18n.Shared(); err == nil && b.Has(lang, "update.phase."+phase) { return b.Msg(lang, "update.phase."+phase) } return UpdatePhaseLabel(phase) } // UpdateErrorIn is the stack's update sentence in lang (v0.264.0, R-606). func (s Stack) UpdateErrorIn(lang string) string { if s.UpdateErrorKey == "" || lang == i18n.Default { return s.UpdateError } return util.Text(lang, s.UpdateErrorKey, s.UpdateErrorArgs...) } func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) { if m.updateComposeFn != nil { return m.updateComposeFn(dir, env, args...) } return m.composeExecCustomEnv(dir, env, args...) } func (m *Manager) updateHealth(ctx context.Context, name string, timeout time.Duration) (bool, string) { if m.updateHealthFn != nil { return m.updateHealthFn(ctx, name, timeout) } return m.waitUpdateHealthy(ctx, name, timeout) } func (m *Manager) healthTimeout() time.Duration { if m.cfg == nil { return 5 * time.Minute } return m.cfg.Update.HealthTimeoutDuration() } func (m *Manager) backupMaxAge() time.Duration { if m.cfg == nil { return 24 * time.Hour } return m.cfg.Update.BackupMaxAgeDuration() } // pre-update copies, kept in the stack dir so they travel with it. Neither name is one the syncer // copies (it copies exactly docker-compose.yml and .felhom.yml). const ( preUpdateComposeFile = "pre-update-compose.yml" preUpdateAppliedFile = "pre-update-applied.yml" ) func (m *Manager) runGuardedUpdate(ctx context.Context, name string) { start := m.now() st, ok := m.GetStack(name) if !ok { m.finishUpdate(name, UpdatePhaseFailed, fmt.Sprintf("stack %q not found", name)) return } dir := filepath.Dir(st.ComposePath) g := m.guards() entry := updateJournalEntry{StartedAt: start} fail := func(key, detail string, args ...interface{}) { m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail) m.clearJournal(name) m.finishUpdateKey(name, UpdatePhaseFailed, key, "", args...) } if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) { m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.journal_failed", "") return } if g == nil { fail("update.error.no_guards", "no UpdateGuards wired") return } // R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule // applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a // stale second-drive mirror never forces a backup while the app's own unit is minutes old. maxAge := m.backupMaxAge() deployedAt := m.currentDeployTime(name) rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt)) if ok { m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge) } else { m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen)) if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) { fail("update.error.journal_failed", "journal write failed") return } if err := g.BackupNow(ctx, name); err != nil { fail("update.error.backup_failed", "pre-update backup: "+err.Error(), err.Error()) return } now := m.now() rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt)) if !ok { fail("update.error.backup_no_unit", fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen))) return } m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339)) } entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339) entry.ProvenTier = rp.Tier // SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run. if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) { fail("update.error.journal_failed", "journal write failed") return } paths, err := g.SafetyDump(ctx, name) if err != nil { fail("update.error.dump_failed", "safety dump: "+err.Error(), err.Error()) return } m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths) // v0.263.0 — the undo's copy is PLANNED here, before anything moves: its size against the disk // floor (decision 19's limit). The copy itself is taken after the pull, where the app stops anyway. undoVols, perr := m.planUndoCopies(name) if perr != nil { if se, ok := perr.(*undoSpaceError); ok { fail("err.stacks.update_undo_space", "undo copy: "+perr.Error(), se.need, se.free, updateDiskFloorGiB) } else { fail("err.stacks.update_undo_copy_failed", "undo copy plan: "+perr.Error()) } return } // PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash // at any later instant can put it back (Scenario G). prevLive, err := os.ReadFile(st.ComposePath) if err != nil { fail("update.error.pin_failed", "reading the live compose file: "+err.Error()) return } if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil { fail("update.error.journal_failed", "saving the pre-update compose copy: "+err.Error()) return } entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile) // R-639: the OLD .felhom.yml too — its probe is the one the old version answers. md, merr := savePreUpdateMeta(dir) switch { case md == "": m.logger.Printf("[WARN] [stacks] update %s: could not keep the previous .felhom.yml (%v) — an undo would check health with the current one", name, merr) case merr != nil: m.logger.Printf("[WARN] [stacks] update %s: %v", name, merr) entry.PrevMeta = md default: entry.PrevMeta = md } if applied, aerr := LoadAppliedDefinition(dir); aerr == nil { if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil { entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile) } } if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 { entry.PrevPin = map[string]string{} for k, v := range cfg.PinnedImages { entry.PrevPin[k] = v } } if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) { m.removePreUpdateCopies(dir) fail("update.error.journal_failed", "journal write failed") return } if err := m.advancePinToCatalog(name, dir); err != nil { m.pinBack(name, dir, entry) fail("update.error.pin_failed", "advancing the pin: "+err.Error()) return } if cfg := LoadAppConfig(dir); cfg != nil { entry.NewPin = cfg.PinnedImages } env := m.stackEnv(dir) if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) { m.pinBack(name, dir, entry) fail("update.error.journal_failed", "journal write failed") return } if _, err := m.updateCompose(dir, env, "pull"); err != nil { // Scenario E: NOTHING RAN. The containers are the old ones and still running, so the honest // state is the old pin and the old file — put both back. m.pinBack(name, dir, entry) m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err) fail("update.error.pull_failed", "pull failed: "+err.Error()) return } // COPYING (v0.263.0) — the app is stopped here anyway to be recreated, so the extra downtime is the // copy alone. A failed copy moves nothing: the copies go, the pin goes back, the old containers // start again. if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) { m.pinBack(name, dir, entry) fail("update.error.journal_failed", "journal write failed") return } if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: the undo copy failed: %v — removing it, putting the pin back and starting the previous version", name, err) m.removeUndoCopies(name, entry.UndoCopies) m.pinBack(name, dir, entry) if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil { m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr) } fail("err.stacks.update_undo_copy_failed", "undo copy: "+err.Error()) return } if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) { m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up", &entry) return } if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil { // Containers may already have been recreated on the new image — something may have run. m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error(), &entry) return } m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry) } // verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is // known, and a failure holds the app. func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, start time.Time, entry *updateJournalEntry) { if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) { m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name) } timeout := m.healthTimeout() waitStart := m.now() healthy, detail := m.updateHealth(ctx, name, timeout) if !healthy { m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail, entry) return } m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail) m.recordInstalledImages(name, dir, env) _ = m.RefreshStatus() m.removeUndoCopies(name, entry.UndoCopies) m.recordUpdateUndone(name, dir, nil) // a successful update ends the "undone" note m.clearJournal(name) m.removePreUpdateCopies(dir) m.finishUpdate(name, UpdatePhaseDone, "") m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second)) } // holdLogTailLines is how much of each service's log the hold keeps. 400 lines is enough to hold a // migration run and a startup failure, and small enough that a hold never fills a disk. const holdLogTailLines = "400" // captureHoldLogs writes each service's log into /hold-logs// before the app is // stopped. Best-effort by design (R-621): the hold itself must happen either way. func (m *Manager) captureHoldLogs(name, dir string, env []string) { ts := m.now().UTC().Format("20060102T150405Z") outDir := filepath.Join(dir, "hold-logs", ts) if err := os.MkdirAll(outDir, 0o755); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: cannot create the hold-log directory %s: %v — the hold still proceeds", name, outDir, err) return } out, err := m.updateCompose(dir, env, "logs", "--no-color", "--tail", holdLogTailLines) if err != nil { m.logger.Printf("[WARN] [stacks] update %s: `compose logs` before the hold failed: %v — writing what came back anyway", name, err) } path := filepath.Join(outDir, "compose-logs.txt") if werr := os.WriteFile(path, []byte(out), 0o644); werr != nil { m.logger.Printf("[ERROR] [stacks] update %s: could not write %s: %v", name, path, werr) return } m.logger.Printf("[INFO] [stacks] update %s: kept %d bytes of the app's own log at %s before stopping it (R-621)", name, len(out), path) } // failAndHold is Scenario F. Since v0.263.0 it first UNDOES (decision 15): when the job took its // last-second copy, the old version goes back with the data from before the update, and the app is // HELD only when that undo fails too. Without a copy (an update journaled by an older controller, // resumed after an upgrade) it holds as it always did. func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string, entry *updateJournalEntry) { m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s", name, why) // R-621: capture the app's own logs BEFORE the `down`, because the `down` destroys them. Two // drill nights lost the only evidence of WHY an update failed this way — `adventurelog` ran nine // migrations and then never bound its port, and the log that would have said so was gone by the // time anyone looked. The capture is bounded and best-effort: a hold must never fail because its // evidence could not be written. m.captureHoldLogs(name, dir, env) undoState := "" if entry != nil && entry.Copied { if undoState = m.tryUndo(ctx, name, dir, why, entry); undoState == "" { return // undone: the previous version runs on the pre-update data } m.logger.Printf("[ERROR] [stacks] update %s: the UNDO failed too (%s) — HOLDING the app; the undo copies are kept: %v", name, undoState, entry.UndoCopies) } else { m.logger.Printf("[ERROR] [stacks] update %s: no undo copy for this update — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name) if _, err := m.updateCompose(dir, env, "down"); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err) } } holdWhy := "" if g := m.guards(); g == nil { m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name) } else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err) } else if _, w := g.HoldFor(name); w != "" { holdWhy = w m.markUpdateHeld(name) } _ = m.RefreshStatus() m.clearJournal(name) m.removePreUpdateCopies(dir) if holdWhy == "" { m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.hold_unsaved", "") } else { // The hold's own sentence (the page renders it per reader through RestoreHoldForLang). m.finishUpdate(name, UpdatePhaseFailed, holdWhy) } m.emitUpdateEvent(UpdateEventHeld, name, entry, rp, holdWhy != "") } // pinBack restores the pin, the stored definition and the live file from the journaled copies, and // then removes the copies. func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) { m.restoreDefinition(name, dir, entry) m.removePreUpdateCopies(dir) m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin)) } // restoreDefinition is pinBack WITHOUT removing the copies — the undo's form, so a power cut after it // can run it again (RecoverUpdates → undoing) and find the copies still there. It also puts the pinned // version's .felhom.yml record back (v0.263.2), so the NEXT update's undo probes the right version. func (m *Manager) restoreDefinition(name, dir string, entry updateJournalEntry) { if entry.PrevMeta != "" { m.storeAppliedMetaFrom(name, dir, filepath.Join(entry.PrevMeta, ".felhom.yml")) } prevLive, lerr := os.ReadFile(entry.PrevCompose) if lerr != nil { m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr) } if len(entry.PrevPin) > 0 { applied := prevLive if entry.PrevApplied != "" { if b, err := os.ReadFile(entry.PrevApplied); err == nil { applied = b } } if err := m.SetPin(name, dir, entry.PrevPin, applied); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: putting the pin back failed: %v", name, err) } } if lerr == nil { if err := os.WriteFile(ComposePathIn(dir), prevLive, 0o644); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err) } } } func (m *Manager) removePreUpdateCopies(dir string) { _ = os.Remove(filepath.Join(dir, preUpdateComposeFile)) _ = os.Remove(filepath.Join(dir, preUpdateAppliedFile)) _ = os.RemoveAll(filepath.Join(dir, preUpdateMetaDir)) } // settleReason names WHY the wait fell back to container state, so the journal and the log do not // have to be read together to tell "this app declares no check" from "this app declares one that // resolves to no container" (R-630). func settleReason(meta Metadata) string { if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 { return "no probe container — settled on container state" } return "no health check declared" } // waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the // existing probe, or — for an app with none — every container running and none restarting for // updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line. func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) { return m.waitUpdateHealthyMeta(ctx, name, timeout, nil) } // waitUpdateHealthyMeta is the health wait with the probe taken from `override` instead of the app's // current .felhom.yml — the undo's form (the old version is judged by the old probe). nil = current. func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeout time.Duration, override *Metadata) (bool, string) { deadline := m.now().Add(timeout) var runningSince time.Time warnedNoProbe := false last := "no observation yet" for { _ = m.RefreshStatus() st, ok := m.GetStack(name) // v0.263.0 — THE UNDO'S PROBE MUST BE ALLOWED TO RUN. The periodic health probe judges the app // with the CURRENT .felhom.yml and flips a running app to StateUnhealthy when that check fails — // which is exactly the undo's situation when the new version brought a probe the old one does // not answer. Gating on StateRunning alone meant the old probe was never asked and the undo was // reported "did not start" after the full timeout. MEASURED LIVE on 9202 2026-09-23 (docmost, // `last: state unhealthy` for 90 s while the old version served). So with an override that // declares checks, an Unhealthy app is PROBED — and the override's own check decides. It never // settles an Unhealthy app on container state (below: the settle path requires StateRunning). // Pinned by TestUndo_OldProbeRunsOnAnAppTheCurrentProbeMarkedUnhealthy. undoProbe := ok && override != nil && st.State == StateUnhealthy && override.HealthCheck != nil && len(override.HealthCheck.Checks) > 0 switch { case !ok: last = "stack vanished" runningSince = time.Time{} case st.State == StateRunning || undoProbe: // A declared health check is only usable if it resolves to a container. When it does // not, the app is judged the same way an app with NO declared check is judged — // settling on container state — and the log says so. // // R-630, and this `else` is the whole defect: the old code set `last = "no probe // container"` and looped, so `verifying` spent the FULL `update.health_timeout` and // `failAndHold` then STOPPED an app whose containers were all healthy. Measured on // paperless-ngx 2026-09-22: `done` was never reachable, `failed` at +313.0 s, front door // 404 afterwards. A stack with no probe is not "healthy" and it is not "failing" — it is // SETTLED ON CONTAINER STATE (`09` §3), and never a reason to stop a running app. usable := false meta := st.Meta if override != nil { meta = *override } if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 { c, candidates := findProbeContainerMeta(name, &meta, st.Containers) if c != "" { usable = true res := m.probeRun(probeTarget{stackName: name, containerName: c, checks: hc.Checks}) m.mu.Lock() if s, ok := m.stacks[name]; ok { s.HealthProbe = res } m.mu.Unlock() if res.Healthy { return true, "the app's health check passed" } last = "health check failing" } else if !warnedNoProbe { warnedNoProbe = true m.logger.Printf("[WARN] [stacks] update %s: no probe container for %s — settling on container state instead; candidates: %v", name, name, candidates) } } if !usable && st.State != StateRunning { last = "unhealthy, and the check resolves to no container" } else if !usable { if runningSince.IsZero() { runningSince = m.now() } if m.now().Sub(runningSince) >= updateSettleWindow { return true, fmt.Sprintf("all containers running, none restarting, for %s (%s)", updateSettleWindow, settleReason(meta)) } last = "running, settling" } default: runningSince = time.Time{} last = "state " + string(st.State) } if !m.now().Before(deadline) { return false, fmt.Sprintf("not healthy within %s (last: %s)", timeout, last) } select { case <-ctx.Done(): return false, "cancelled: " + ctx.Err().Error() case <-time.After(updatePollEvery): } } } // ── the journal ────────────────────────────────────────────────────────────────────────────────── type updateJournalEntry struct { Phase string `json:"phase"` StartedAt time.Time `json:"started_at"` PrevPin map[string]string `json:"prev_pin,omitempty"` PrevCompose string `json:"prev_compose,omitempty"` PrevApplied string `json:"prev_applied,omitempty"` ProvenCopyAt string `json:"proven_copy_at,omitempty"` // ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names // the right copy. 0 in a journal written by v0.238.1 or older. ProvenTier int `json:"proven_tier,omitempty"` // v0.263.0 — the undo (undo.go). UndoCopies is journaled BEFORE each copy starts; Copied is true // only once every copy finished (it may be true with zero copies: an app with no named volume). // PrevMeta is the directory holding the previous .felhom.yml; NewPin the pin the update moved to. UndoCopies []undoCopy `json:"undo_copies,omitempty"` Copied bool `json:"copied,omitempty"` PrevMeta string `json:"prev_meta,omitempty"` NewPin map[string]string `json:"new_pin,omitempty"` } type updateJournal struct { Updates map[string]updateJournalEntry `json:"updates"` } func (m *Manager) updateJournalPath() string { return filepath.Join(m.cfg.Paths.DataDir, "update-journal.json") } func (m *Manager) readUpdateJournal() updateJournal { j := updateJournal{Updates: map[string]updateJournalEntry{}} data, err := os.ReadFile(m.updateJournalPath()) if err != nil { return j } if err := json.Unmarshal(data, &j); err != nil { m.logger.Printf("[WARN] [stacks] update journal at %s is corrupt (%v) — quarantining", m.updateJournalPath(), err) _ = os.Rename(m.updateJournalPath(), fmt.Sprintf("%s.corrupt-%d", m.updateJournalPath(), time.Now().Unix())) return updateJournal{Updates: map[string]updateJournalEntry{}} } if j.Updates == nil { j.Updates = map[string]updateJournalEntry{} } return j } // writeUpdateJournal is atomic and fsynced (the AppStopGuard shape): the point is surviving a power cut. func (m *Manager) writeUpdateJournal(j updateJournal) error { p := m.updateJournalPath() if len(j.Updates) == 0 { if err := os.Remove(p); err != nil && !os.IsNotExist(err) { return err } return nil } data, err := json.MarshalIndent(j, "", " ") if err != nil { return err } if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { return err } tmp := p + ".tmp" f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600) if err != nil { return err } if _, err := f.Write(data); err != nil { f.Close() os.Remove(tmp) return err } if err := f.Sync(); err != nil { f.Close() os.Remove(tmp) return err } if err := f.Close(); err != nil { os.Remove(tmp) return err } return os.Rename(tmp, p) } // enterUpdatePhase journals the phase BEFORE it starts and mirrors it for the UI. False means the // journal could not be written — the caller must not perform a mutation it could not record. func (m *Manager) enterUpdatePhase(name string, entry *updateJournalEntry, phase string) bool { entry.Phase = phase m.updateJournalMu.Lock() j := m.readUpdateJournal() j.Updates[name] = *entry err := m.writeUpdateJournal(j) m.updateJournalMu.Unlock() m.setUpdatePhase(name, phase) if err != nil { m.logger.Printf("[ERROR] [stacks] update %s: journal write for phase %s failed: %v", name, phase, err) return false } m.logger.Printf("[INFO] [stacks] update %s: phase %s", name, phase) return true } func (m *Manager) clearJournal(name string) { m.updateJournalMu.Lock() defer m.updateJournalMu.Unlock() j := m.readUpdateJournal() delete(j.Updates, name) if err := m.writeUpdateJournal(j); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: clearing the journal entry failed: %v", name, err) } } // RecoverUpdates reads the journal at startup (Scenario G). Call it BEFORE the boot reconciler. // // - interrupted BEFORE the pin moved (checking, backing-up, safety-dump): nothing moved — the entry is // dropped and the app carries the "interrupted" sentence; // - interrupted while pinning or pulling: nothing RAN — the pin and definition are put back (as E); // - interrupted while starting or verifying: something may have run — the app is marked Updating // (so the boot sweep and the dead-app alarm leave it alone) and queued for ResumeInterruptedUpdates, // which re-runs `up -d` and the health wait, ending in A or F. // // It needs no backup wiring, because nothing here holds an app — that is left to the resumed job. func (m *Manager) RecoverUpdates() []string { m.updateJournalMu.Lock() j := m.readUpdateJournal() m.updateJournalMu.Unlock() if len(j.Updates) == 0 { return nil } var resumed []string for name, e := range j.Updates { st, ok := m.GetStack(name) if !ok { m.logger.Printf("[WARN] [stacks] update recovery: %s is in the journal (phase %s) but no longer exists — dropping the entry", name, e.Phase) m.clearJournal(name) continue } dir := filepath.Dir(st.ComposePath) switch e.Phase { case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump: m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339)) m.clearJournal(name) m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "") case UpdatePhasePinning, UpdatePhasePulling: m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339)) m.pinBack(name, dir, e) m.clearJournal(name) m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "") case UpdatePhaseCopying: // v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the // pin goes back, and the previous version is started again. m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while copying its data (started %s) — nothing new ran; removing the partial copy, putting the pin back and starting the previous version", name, e.StartedAt.Format(time.RFC3339)) m.removeUndoCopies(name, e.UndoCopies) m.pinBack(name, dir, e) if _, err := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); err != nil { m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err) } m.clearJournal(name) m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "") case UpdatePhaseUndoing: // v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from // the copies (still there: they are removed only after the undo succeeded) and then probes. // Never "done": what ran last was a failed new version. m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while UNDOING (started %s) — marking it Updating and RESUMING the undo", name, e.StartedAt.Format(time.RFC3339)) m.mu.Lock() if s, ok := m.stacks[name]; ok { s.Updating, s.UpdateError, s.updateHeld = true, "", false s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseUndoing, UpdatePhaseLabel(UpdatePhaseUndoing) } m.updateResume = append(m.updateResume, name) m.mu.Unlock() resumed = append(resumed, name) case UpdatePhaseStarting, UpdatePhaseVerifying: m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339)) m.mu.Lock() if s, ok := m.stacks[name]; ok { s.Updating, s.UpdateError, s.updateHeld = true, "", false s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying) } m.updateResume = append(m.updateResume, name) m.mu.Unlock() resumed = append(resumed, name) default: m.logger.Printf("[WARN] [stacks] update recovery: %s has unknown phase %q — dropping the entry", name, e.Phase) m.clearJournal(name) } } return resumed } // ResumeInterruptedUpdates continues the updates RecoverUpdates queued, once the backup side is wired // (a resumed update that fails must be able to HOLD). Returns how many were resumed. func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int { m.mu.Lock() names := m.updateResume m.updateResume = nil m.mu.Unlock() for _, name := range names { st, ok := m.GetStack(name) if !ok { m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "") continue } m.updateJournalMu.Lock() e, ok := m.readUpdateJournal().Updates[name] m.updateJournalMu.Unlock() if !ok { m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "") continue } provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt) rp := UpdateRestorePoint{Tier: e.ProvenTier, ProvenAt: provenAt} dir := filepath.Dir(st.ComposePath) go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) { env := m.stackEnv(dir) if e.Phase == UpdatePhaseUndoing { m.logger.Printf("[INFO] [stacks] update %s: resuming the UNDO after a controller restart", name) m.failAndHold(ctx, name, dir, env, rp, "resumed after a restart during the undo", &e) return } m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name) if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil { m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error(), &e) return } m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e) }(name, dir, e, rp) } return len(names) } // probeRun is the network half of the update's health wait — its own seam, so a test can drive the // REAL wait loop (the state gate, the container resolution, the settle rule) with only the HTTP/TCP // probe faked. nil ⇒ runChecks. func (m *Manager) probeRun(t probeTarget) *HealthProbeResult { if m.probeRunFn != nil { return m.probeRunFn(t) } return m.runChecks(t) }