package backup import ( "context" "errors" "fmt" "gitea.dooplex.hu/admin/felhom-controller/internal/i18n" "gitea.dooplex.hu/admin/felhom-controller/internal/util" "os" "path/filepath" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // ── The backup side of the guarded update (update arc slice 4, controller v0.237.0) ───────────── // // 09-update-architecture.md §3 decision 1 (operator ruling 2026-09-02): the safety copy for an update // is a VERIFIED RECENT BACKUP as a PRECONDITION — not a new copy mechanism invented for the update // path. So everything in this file composes machinery that already exists and is proven live: // the Tier-2 unit restore's own predicate (R-102/R-103), the nightly legs (DB dump, volume dump, // unit capture, Tier-2 mirror), the pre-restore safety dump (R-361) and the R-379 hold. // // The stacks package cannot import this one, so the update job reaches all of it through the // stacks.UpdateGuards interface, implemented by an adapter in cmd/controller/main.go. // Tier2RestorePoint is the answer to "could this app be restored from its Tier-2 copy, and from // when?" — the predicate the destructive „Teljes visszaállítás" action is gated on. // // EXTRACTED, NOT DUPLICATED (slice 4). Until v0.237.0 this computation lived inline in the backups // page handler (buildAppBackupRows). The update path needs exactly the same question answered, and // a second copy of a predicate is how this project's two copies of `namespaceRoot` came to differ // (R-203). So there is one function, and the page and the update both call it. type Tier2RestorePoint struct { // Restorable — the copy holds an OPENABLE recovery unit (Tier2Coverage.CanRestoreUnit). Restorable bool // CopyDate — the date the unit restore NAMES: the package's own manifest date, falling back to the // copy date (Tier2Coverage.UnitRestoreDate, R-403). RFC3339 as recorded, "" when unknown. CopyDate string // CopyDateProven — a copy actually SUCCEEDED (LastSuccess is set), never merely an attempt (R-101). CopyDateProven bool // PackagePreserved — the newest run PRESERVED an older package instead of refreshing it (R-403). PackagePreserved bool // CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded. CopyLastSuccess string // DataDate (v0.275.0, R-696) — the mirrored unit's DATA time (unitNewestArtifact on the mirror): when // the data the copy holds was written, which a mirror run copies but never makes newer. "" = unknown. DataDate string } // restorePointFromCoverage is the pure half of the predicate. func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint { pkgDate, preserved := cov.UnitRestoreDate() return Tier2RestorePoint{ Restorable: cov.CanRestoreUnit(), CopyDate: pkgDate, CopyDateProven: cov.CopyLastSuccess != "", PackagePreserved: preserved, CopyLastSuccess: cov.CopyLastSuccess, DataDate: cov.UnitDataDate, } } // Tier2UnitRestorePoint resolves the app's recorded Tier-2 copy and returns the restore point. The // error is the same refusal Tier2RestoreCoverage raises (no copy, drive gone, pre-v2 layout). func (m *Manager) Tier2UnitRestorePoint(stackName string) (Tier2RestorePoint, error) { cov, err := m.Tier2RestoreCoverage(stackName) if err != nil { return Tier2RestorePoint{}, err } return restorePointFromCoverage(cov), nil } // ProvenCopyTime returns WHEN the data this copy would restore was last proven copied, and false when // there is no proven, restorable copy at all. // // WHY NOT CopyDate, measured rather than assumed. CopyDate is the unit MANIFEST's created_at, and a // capture rewrites the manifest only when the app's DEFINITION changes (compose, app.yaml, controller // version) — a nightly DB dump keeps the same file name, so it does not move it. Measured on demo-hp // 2026-09-13: bookstack's Tier-2 mirror held `bookstack-mariadb.sql` written 2026-09-13T00:30Z while // its manifest still read 2026-09-12T02:15:29Z. Judging "recent" by that date would call a fresh copy // stale — and, worse, a "back up first" run would not move it either on a quiet app, so the update // would be refused forever. // // So the age is the last SUCCESSFUL copy (LastSuccess), which the Tier-2 run records only when it // actually mirrored the unit — EXCEPT when the run preserved an older package (R-403), in which case // the package date is the honest one, because that is what the copy really holds. func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) { if !p.Restorable || !p.CopyDateProven { return time.Time{}, false } src := p.CopyLastSuccess if p.PackagePreserved { src = p.CopyDate } t, err := time.Parse(time.RFC3339, src) if err != nil { return time.Time{}, false } // v0.275.0 (R-696): a mirror run copies the unit's data; it never makes the data newer. When the // mirror's data time is known and older than the copy, the data time is the copy's age — a mirror // taken right after an update, of a unit whose dump is from before it, is as old as that dump. if d, derr := time.Parse(time.RFC3339, p.DataDate); derr == nil && d.Before(t) { return d, true } return t, true } // ── R-475: any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0) ──── // // Until v0.239.0 the update's precondition was Tier2UnitRestorePoint alone, so an app with no second // drive could never be updated — even with a fresh recovery unit on its own drive and an off-site // snapshot from last night. The ruling: every backup counts. Tier2UnitRestorePoint itself is NOT // changed; the backups page still calls it for the „Teljes visszaállítás" action, which really does // restore from the second drive only. // Backup tiers, as the update precondition and the hold sentence name them. const ( UpdateTierLocal = 1 // the app's own recovery unit on its drive — „helyi" on the restore page UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive UpdateTierOffsite = 3 // the off-site restic repository ) // updateTierOrder is the preference order the ruling set: the second drive, then the app's own unit, // then off-site. The first tier holding a copy the caller ACCEPTS is chosen. var updateTierOrder = []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite} // updateTierOrderBindData (R-479, operator ruling 2026-09-13, v0.241.0) is the order for an app whose // DATA lives in bind-mounted files outside its recovery unit: second drive, OFF-SITE, own unit. The own // unit then holds the definition and the database dumps but not the files, so a route back that names // it would restore settings and not data — measured on demo-hp with gokapi (v0.239.0: „a beállítások // visszaálltak … adatot nem"). Off-site carries the mandatory file legs; it comes before the unit. var updateTierOrderBindData = []int{UpdateTierSecondDrive, UpdateTierOffsite, UpdateTierLocal} // DataOutsideUnit reports whether the app keeps data in bind-mounted files that the recovery unit does // not hold — i.e. the app has classified binds. Nil provider or no binds → false (the unit holds the // data: named volumes and database dumps). Pinned by TestR479_. func (m *Manager) DataOutsideUnit(stackName string) bool { if m == nil || m.stackProvider == nil { return false } binds, has := m.stackProvider.GetStackClassifiedBinds(stackName) return has && len(binds) > 0 } // UpdateTierOrderFor is the tier order the update walks for this app (R-475 / R-479). func (m *Manager) UpdateTierOrderFor(stackName string) []int { if m.DataOutsideUnit(stackName) { return updateTierOrderBindData } return updateTierOrder } // UpdateCopyHolds is the customer phrase for what a copy on `tier` holds for this app — the second half // of the R-479 ruling: the hold sentence names WHAT the chosen copy holds, not only where it is. // // v0.264.0 (R-606): the phrase is a bundle key — it is a PROMISE ABOUT WHETHER THE HOUSEHOLD'S FILES // COME BACK and must reach an English household in English. The value returned (and stored in the // hold) is still the Hungarian, byte for byte; copyHoldsIn maps it back to its key at render time, so // a hold written by any earlier version localises too. func (m *Manager) UpdateCopyHolds(stackName string, tier int) string { return util.Text(i18n.Default, updateCopyHoldsKey(m.DataOutsideUnit(stackName), tier)) } // UpdateCopyHoldsKey is the same verdict as UpdateCopyHolds, as its bundle KEY (R-647, v0.265.0). The // app_update_held event carries the key, not the Hungarian phrase: the hub prints the raw details as // the mail's `Note:` line, and an English household read „a beállításokat, …" there. func (m *Manager) UpdateCopyHoldsKey(stackName string, tier int) string { return updateCopyHoldsKey(m.DataOutsideUnit(stackName), tier) } func updateCopyHoldsKey(outside bool, tier int) string { switch tier { case UpdateTierLocal: if outside { return "hold.copy_holds.db_only" } return "hold.copy_holds.volumes" case UpdateTierSecondDrive, UpdateTierOffsite: if outside { return "hold.copy_holds.files" } return "hold.copy_holds.volumes" } return "" } // copyHoldKeys are the phrases a hold may have stored; the stored value is the Hungarian. var copyHoldKeys = []string{"hold.copy_holds.db_only", "hold.copy_holds.volumes", "hold.copy_holds.files"} // copyHoldsIn renders a STORED copy-holds phrase in lang. An unknown phrase is returned as stored — // never an empty clause (a hold must not lose the sentence that says what the copy holds). func copyHoldsIn(lang, stored string) string { for _, k := range copyHoldKeys { if util.Text(i18n.Default, k) == stored { return util.Text(lang, k) } } return stored } // UpdateTierLabel is a tier's name in the customer's hold sentence (Hungarian). "" for an unknown tier. func UpdateTierLabel(tier int) string { return UpdateTierLabelIn(i18n.Default, tier) } // UpdateTierLabelIn is UpdateTierLabel in lang (v0.264.0, R-606). func UpdateTierLabelIn(lang string, tier int) string { switch tier { case UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite: return util.Text(lang, fmt.Sprintf("hold.update.tier.%d", tier)) } return "" } // updateOffsiteCheckTimeout bounds the off-site lookup. An update must not stall on an unreachable // Storage Box: past this the off-site copy counts as ABSENT (with a WARN), and the update carries on // with backing up first. A var only so a test can shorten it. var updateOffsiteCheckTimeout = 15 * time.Second // UpdateTierPoint is one proven, restorable copy of an app on one tier. type UpdateTierPoint struct { Tier int // At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime (capped by the // mirror's data time), Tier 1 the unit's DATA time (ListRestorePoints → unitNewestArtifact), Tier 3 // the newest snapshot, capped by the data time the box recorded when it pushed it (v0.275.0, R-696). At time.Time } // UpdateRestorePoints walks the tiers in preference order (2, 1, 3) and returns the FIRST copy that // accept admits (nil accepts any), whether one was found, and every copy it looked at on the way. // // It stops at the first accepted copy, so a box with a fresh second-drive copy never touches the // network. The AGE rule is the caller's (stacks applies backup_max_age through accept) — that is what // makes "the age rule applies to whichever tier is chosen" one rule, not three (R-475 Scenario M). func (m *Manager) UpdateRestorePoints(ctx context.Context, stackName string, accept func(UpdateTierPoint) bool) (UpdateTierPoint, bool, []UpdateTierPoint) { var seen []UpdateTierPoint for _, tier := range m.UpdateTierOrderFor(stackName) { p, ok := m.updateTierPoint(ctx, stackName, tier) if !ok { continue } seen = append(seen, p) if accept == nil || accept(p) { return p, true, seen } } return UpdateTierPoint{}, false, seen } func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier int) (UpdateTierPoint, bool) { switch tier { case UpdateTierSecondDrive: get := m.updateTier2PointFn if get == nil { get = m.Tier2UnitRestorePoint } rp, err := get(stackName) if err != nil { if m.isDebug() { m.logger.Printf("[DEBUG] [backup] update precondition for %s: no Tier-2 copy (%v)", stackName, err) } return UpdateTierPoint{}, false } at, ok := rp.ProvenCopyTime() return UpdateTierPoint{Tier: tier, At: at}, ok case UpdateTierLocal: list := m.updateTier1PointsFn if list == nil { list = m.ListRestorePoints } pts, _ := list(stackName) for _, rp := range pts { if at, err := time.Parse(time.RFC3339, rp.Time); err == nil { return UpdateTierPoint{Tier: tier, At: at}, true } } return UpdateTierPoint{}, false case UpdateTierOffsite: times := m.updateOffsiteTimesFn if times == nil { if m.settings == nil || !m.OffboxConfigured() { return UpdateTierPoint{}, false } times = m.OffsiteSnapshotTimes } cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout) defer cancel() got, err := times(cctx) if err != nil { if !errors.Is(err, errNoOffsiteTarget) { m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err) } return UpdateTierPoint{}, false } if at, ok := got[stackName]; ok && !at.IsZero() { return UpdateTierPoint{Tier: tier, At: m.offsiteDataTime(stackName, at)}, true } } return UpdateTierPoint{}, false } // CanBackUpApp reports whether "back up first" can run for this app at all right now — the second // half of R-475 Scenario L: an app with no copy anywhere is refused only when this is false too. // Cheap and read-only; RunAppBackupNow re-checks everything when it actually runs. func (m *Manager) CanBackUpApp(stackName string) (bool, string) { if m == nil { return false, "backup is not enabled on this box" } if m.stackProvider == nil { return false, "stack provider not configured" } if m.migrationActive() { return false, "a data migration is running" } drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { return false, "the app's drive cannot be resolved" } if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) { return false, fmt.Sprintf("the app's drive %s is not available", drivePath) } return true, "" } // UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an // update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer // sentence is chosen by the caller. // // IsRunning is box-wide, deliberately: the backup/restore single-flight is box-wide, and an update's // "back up first" leg needs that same flag — an update started beside a running backup would either // wait on it invisibly or fail half-way. func (m *Manager) UpdateBusy(stackName string) (bool, string) { if m == nil { return false, "" } if m.IsRunning() { return true, "a backup or restore is running (single-flight held)" } if st := m.RestoreStatus(); st.Running { return true, fmt.Sprintf("restore op %q is running for %q", st.Op, st.Stack) } for _, held := range m.appStop.HeldStacks() { if held == stackName { return true, "an app-data operation (volume dump / export / reconstitute) is holding it" } } return false, "" } // ErrUpdateBackupNoUnit is returned when a "back up first" run completed but the app still has no // openable Tier-2 unit — typically Tier 2 is switched off for the app or has no second target. var ErrUpdateBackupNoUnit = util.MsgError("err.backup.a_frissites_elotti_mentes_lefutott_de") // RunAppBackupNow runs THIS app's backup legs now, in the nightly order, and then its Tier-2 copy: // database dump(s) → volume dump (if the app has named volumes) → recovery-unit capture → Tier-2 // mirror. It is the "back up first" of slice 4 Scenario B. // // Composed, not reinvented: every leg is the one runDBDumpsInternal and RunAllTier2 already run, // including the R-181 reserve (admitApp) before the first write and the R-166 app-stop marker inside // DumpAppVolumesSafe. What differs is only the scope — one app instead of all of them — because an // update must not bounce every other app on the box to back up one. func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error { if m.stackProvider == nil { return fmt.Errorf("stack provider not configured") } if m.migrationActive() { return util.MsgError("err.backup.adatathelyezes_folyamatban_a_mentes_most_nem") } if err := m.acquireRunning(); err != nil { return err } m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName) start := time.Now() var nsRoot string legErr := func() error { defer m.releaseRunning() defer m.beginAdmissionRun()() drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { return util.MsgError("err.backup.az_alkalmazas_meghajtoja_nem_hatarozhato_meg") } if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) { return util.MsgError("err.backup.az_alkalmazas_meghajtoja_nem_elerheto", drivePath) } if !m.admitApp(stackName) { return util.MsgError("err.backup.nincs_eleg_szabad_hely_a_menteshez", drivePath) } nsRoot = m.namespaceRoot(drivePath) discover := m.discoverDBs if discover == nil { discover = func(ctx context.Context) ([]DiscoveredDB, error) { return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames()) } } dbs, err := discover(ctx) if err != nil { return util.MsgError("err.backup.adatbazis_felderites_sikertelen", err) } dumped := 0 for _, db := range dbs { if db.StackName != stackName { continue } res := m.dumpOneOrDefault(ctx, db, AppDBDumpPath(nsRoot, stackName)) if res.Error != nil { return util.MsgError("err.backup.adatbazis_mentes_sikertelen", db.ContainerName, res.Error) } dumped++ m.stampDataFile(stackName, RecoveryUnitPath(nsRoot, stackName), "db-dumps/"+filepath.Base(res.FilePath)) m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size)) } if len(m.stackProvider.GetDockerVolumes(stackName)) > 0 { dump := m.dumpVolumesSafe if dump == nil { dump = m.DumpAppVolumesSafe } if err := dump(stackName); err != nil { return util.MsgError("err.backup.kotetmentes_sikertelen", err) } m.logger.Printf("[INFO] [backup] update pre-backup for %s: volume dump OK", stackName) } if err := m.CaptureRecoveryUnit(stackName); err != nil { return util.MsgError("err.backup.a_mentesi_egyseg_rogzitese_sikertelen", err) } m.logger.Printf("[INFO] [backup] update pre-backup for %s: recovery unit captured (%d database dump(s))", stackName, dumped) return nil }() if legErr != nil { m.logger.Printf("[ERROR] [backup] update pre-backup for %s FAILED after %s: %v", stackName, time.Since(start).Round(time.Millisecond), legErr) return legErr } m.updatePreBackupTail(stackName, nsRoot, time.Now()) m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond)) return nil } // updatePreBackupTail is what "back up first" does after the capture succeeded (R-475). // // 1. It marks the app's OWN unit as proven current NOW. CaptureRecoveryUnit leaves the manifest alone // when nothing changed (the checksum skip), and Tier 1's age is the newest artifact's mtime — so on // an app with no database and no named volume a fresh "back up first" would leave Tier 1 as old as // its last definition change, and the update would be refused forever. That is the trap // ProvenCopyTime documents for Tier 2, one tier down. The capture has just compared the unit with // the live definition, so "current as of now" is exactly what it established. // 2. It runs the Tier-2 copy, and a Tier-2 failure is a WARN, not a failure: the update may lean on // any tier, and the Tier-1 unit it can lean on was just written. Pinned by // TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh. func (m *Manager) updatePreBackupTail(stackName, nsRoot string, now time.Time) { if nsRoot != "" { if err := os.Chtimes(RecoveryUnitManifestPath(nsRoot, stackName), now, now); err != nil { m.logger.Printf("[WARN] [backup] update pre-backup for %s: could not mark the recovery unit as proven current (%v) — its own-unit copy may read older than it is", stackName, err) } } runOne := m.perAppTier2 if runOne == nil { runOne = m.RunTier2 } if err := runOne(stackName); err != nil { m.logger.Printf("[WARN] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v — not fatal: the app's own recovery unit was just captured, and an update may lean on any tier (R-475)", stackName, err) } } // WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the // pin: "the state the customer was in a minute ago". It is writeSafetyDump (R-361) unchanged — the // same `pre-restore-` undo naming, the same pruning to three, the same never-the-canonical-name rule // — so it is also picked up by the same exclusions (it never enters a manifest's db_dumps). // // Returns the paths written; an app with no database returns (nil, nil), which is a no-op and never a // failure (measured in writeSafetyDump: `len(mine) == 0` returns an empty set). func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) ([]string, error) { nsRoot := m.AppNamespaceRoot(stackName) if nsRoot == "" { return nil, util.MsgError("err.backup.az_alkalmazas_mentesi_helye_nem_hatarozhato") } set, err := m.writeSafetyDump(ctx, stackName, nsRoot) if err != nil { return nil, err } var paths []string for _, f := range set.Files { paths = append(paths, f.Path) } if len(paths) == 0 { m.logger.Printf("[INFO] [backup] update safety dump for %s: the app has no database — nothing to copy (no-op)", stackName) } else { m.logger.Printf("[INFO] [backup] update safety dump for %s: %d file(s) %v", stackName, len(paths), paths) } return paths, nil } // UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app, // the time of the failure, the TIER of the copy it can be restored from (UpdateTierLabel), and that // copy's PROVEN date. One named string so a test asserts it verbatim instead of retyping Hungarian // (R-364). Since v0.239.0 (R-475) it names the tier: the copy may be on any of three, and each is // restored from a different place on the Mentések page. const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " + "Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " + "Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s — ez a másolat %s." // UpdateHoldTierFmt is the v0.239.0–v0.240.0 sentence, kept for a hold that recorded a tier but not // what the copy holds (CopyHolds empty). const UpdateHoldTierFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " + "Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " + "Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s." // UpdateHoldLegacyFmt is the v0.237.0–v0.238.1 sentence, kept for a hold written before the tier was // recorded (CopyTier 0) — every such hold named a Tier-2 copy, but it did not SAY so, and rewriting // it now would state a fact the record does not hold. const UpdateHoldLegacyFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " + "Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " + "Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon." // holdTimeZone is where the customer-facing hold sentence renders its times. The same zone the web // layer renders the Mentések page's copy dates in (web.getTimezone), so the date in the hold text and // the date on the page it points at are the same string. func holdTimeZone() *time.Location { if loc, err := time.LoadLocation("Europe/Budapest"); err == nil { return loc } return time.UTC } func fmtHoldTime(rfc3339 string) string { t, err := time.Parse(time.RFC3339, rfc3339) if err != nil { return rfc3339 } return t.In(holdTimeZone()).Format("2006-01-02 15:04") } // HoldAfterFailedUpdate records that an app is held stopped because its new version did not come up // healthy. Same storage and same gate as the R-379 hold — every start path that already refuses a // restore hold refuses this one without being touched. // // It returns the error rather than only logging it, unlike holdAppAfterFailedRollback: the update job // has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app // that the next restart button will quietly start again. The caller logs it at ERROR and keeps the // failure on the page. func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error { return m.HoldAfterFailedUpdateHolding(stackName, at, copyDate, copyTier, "", "") } // HoldAfterFailedUpdateHolding is HoldAfterFailedUpdate with the R-479 phrase for what the copy holds; // "" records none (the tier-only sentence). The adapter in main.go computes the phrase with // UpdateCopyHolds at hold time. // // undoState (v0.263.0) is what a FAILED undo left the data as (stacks.UndoState*), "" when no undo was // attempted. RestoreHoldFor puts it in front of the sentence, so the household reads that the box // already tried to put the app back, and in what state that left the data. func (m *Manager) HoldAfterFailedUpdateHolding(stackName string, at time.Time, copyDate time.Time, copyTier int, copyHolds, undoState string) error { if m == nil || m.settings == nil { return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName) } h := settings.RestoreHold{ Stack: stackName, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUpdateFailed, UndoState: undoState, } if !copyDate.IsZero() { h.CopyDate = copyDate.UTC().Format(time.RFC3339) h.CopyTier = copyTier h.CopyHolds = copyHolds } if err := m.settings.SetRestoreHold(h); err != nil { return fmt.Errorf("persisting the update hold for %s: %w", stackName, err) } m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s; holds: %q; undo: %q)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate, h.CopyHolds, h.UndoState) return nil } // isHeld reports whether the nightly legs must leave an app alone: it carries ANY hold, OR a guarded // update is moving it right now. // // THE SECOND HALF WAS FOUND LIVE, v0.238.0 Scenario F on demo-hp 2026-09-13. During the update's // 5-minute health wait the app is not yet held, and the periodic capture ran at 10:17:09 and wrote // the NEW definition (alpine:3.20, which never started) into the app's PRIMARY unit, 53 s before the // hold landed at 10:18:02. The Tier-2 mirror the hold names was intact only because the Tier-2 run is // daily — a nightly Tier-2 falling inside a verify window would have mirrored the broken definition // over the very copy the customer is told to restore from. An app mid-update has a restore point that // must not move, exactly like a held one. func (m *Manager) isHeld(stackName string) bool { held, _ := m.RestoreHoldFor(stackName) if held { return true } return m.updatingCheck != nil && m.updatingCheck(stackName) } // SetUpdatingCheck wires the "is a guarded update moving this app" question (stacks.Manager.IsUpdating). // INIT-ONLY, in main.go — pinned by TestSlice4_UpdatingCheckIsWiredAtStartup. The backup package cannot // import stacks, which is why it is a seam. func (m *Manager) SetUpdatingCheck(fn func(stackName string) bool) { m.updatingCheck = fn } // clearUpdateHoldAfterRestore lifts an UPDATE hold once a person has restored the app successfully. // // "A person clears it by restoring" — slice 4 Part 3. The restore just put the app back on the // definition and data of its recovery unit and started it, which is the exact route back the hold // text names; leaving the hold in place would refuse the next restart of an app that is now fine. // // A RESTORE hold (R-379) is deliberately NOT cleared here: that hold means a previous restore already // left the database in an unknown state, and it stays operator-cleared (`-clear-restore-hold`). func (m *Manager) clearUpdateHoldAfterRestore(stackName string) { if m.settings == nil { return } h, ok := m.settings.GetRestoreHold(stackName) if !ok || h.Reason != settings.HoldReasonUpdateFailed { return } if _, err := m.settings.ClearRestoreHold(stackName); err != nil { m.logger.Printf("[ERROR] [backup] %s was restored, but its update hold could not be cleared: %v", stackName, err) return } m.logger.Printf("[INFO] [backup] %s: restore completed — the update hold (set %s) is CLEARED", stackName, h.At) // R-671 (v0.272.0): the undo copies the hold kept describe the state this restore just replaced. They // were kept so the hold's data stayed recoverable; once the app is restored whole they are dead weight // (measured 2026-09-24 on 9202: three nextcloud copies, ~0.9 GiB, outlived the hold and nothing named // them). Removed here, and only here — never for a restore hold (R-379), never while an update moves the // app. Pinned by TestR671_RestoreThatClearsAnUpdateHoldRemovesItsUndoCopies. if m.undoCopyRemover != nil && (m.updatingCheck == nil || !m.updatingCheck(stackName)) { n := m.undoCopyRemover(stackName) m.logger.Printf("[INFO] [backup] %s: removed %d undo cop(y/ies) the lifted update hold had kept (R-671)", stackName, n) } } // SetUndoCopyRemover wires the undo-copy cleanup (R-671): stacks.Manager.RemoveUndoCopies. INIT-ONLY, main.go — // pinned by TestUndoCopyRemoverIsWiredAtStartup (an AST walk). The backup package cannot import stacks. func (m *Manager) SetUndoCopyRemover(fn func(stackName string) int) { m.undoCopyRemover = fn } // UpdateHeldStacks is the set of apps held stopped after a failed update (R-660, v0.268.0) — the // FOURTH way the product stops an app on purpose, and until v0.268.0 the one `classifyRunStates` did // not know: each hold's `app_update_held` was followed ~11 s later by an `app_start_failed` for the // same app (chaos rounds 8 and 11, 2026-09-23 night). A RESTORE hold (R-379) is not in the set: it // has no event of its own, so the app-down alarm stays its only voice. Nil-safe. func (m *Manager) UpdateHeldStacks() map[string]bool { if m == nil || m.settings == nil { return nil } var out map[string]bool for _, h := range m.settings.ListRestoreHolds() { // v0.269.0 (decision 28): an app the box stopped for a crash loop / OOM storm is stopped BY THE // PRODUCT and has its own event (app_stopped_unhealthy) — the same class as an update hold. if h.Reason != settings.HoldReasonUpdateFailed && h.Reason != settings.HoldReasonUnhealthyStop { continue } if out == nil { out = map[string]bool{} } out[h.Stack] = true } return out } // ── R-659 (v0.268.0): the hold names only a copy that can bring the app back WHOLE ──────────────── // // MEASURED 2026-09-24 00:00 on 9202 (chaos round 11): nextcloud's update and its undo both failed; the // hold named „saját meghajtó" (the precondition copy — decision 8 lets an update lean on any tier); // the household pressed exactly that restore and was REFUSED, because the unit holds no copy of the // app's files on the drive (R-538). The box had no other copy, so nothing on any page brought the app // back. Operator ruling 2026-09-24 (`09` §3 decision 25, option A): the hold names only a copy that // brings the app back whole; with none, it says so, says support is informed, and support is told. // // THE TRUTH TABLE, read from the restores' OWN refusals (measured from source, v0.267.0), not from // what each tier stores: // // app own unit (1) second drive (2) off-site (3) // no declared drive files whole (unit restore) whole („Teljes visszaállítás") whole (full restore) // declared drive files NOT — refused (R-538) NOT — its unit restore is refused whole („Teljes // (DeclaredDriveFileLegs) by the same guard; its file restore visszaállítás (fájlok // only ADDS missing files, no database + adatbázis)") // // So the question is asked of the SAME predicate the refusal uses (DeclaredDriveFileLegs), and a test // pins that the two cannot drift (TestR659_TruthTableAgreesWithTheRestoresRefusal). Tier 2 holds a // file app's files AND its unit, but no single action brings the app back whole from it — R-661. // WholeOnTier reports whether a copy on `tier` can bring this app back WHOLE through the restore the // Mentések page offers for that tier. func (m *Manager) WholeOnTier(stackName string, tier int) bool { switch tier { case UpdateTierOffsite: return true case UpdateTierLocal: return !m.HasDriveFileLegs(stackName) case UpdateTierSecondDrive: if !m.HasDriveFileLegs(stackName) { return true } // v0.269.0 (decision 26): a file app is whole on the second drive when the mirror holds BOTH an // openable unit and its file legs — RestoreTier2Whole brings back both. cov, err := m.Tier2RestoreCoverage(stackName) return err == nil && cov.CanRestoreUnit() && cov.CanRestore() } return false } // HoldCopies walks EVERY tier (not only until the first acceptable one, as the update does) and // returns the newest copy that brings the app back whole, whether there is one, and every copy seen. func (m *Manager) HoldCopies(ctx context.Context, stackName string) (UpdateTierPoint, bool, []UpdateTierPoint) { var seen []UpdateTierPoint var best UpdateTierPoint found := false for _, tier := range []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite} { p, ok := m.updateTierPoint(ctx, stackName, tier) if !ok { continue } seen = append(seen, p) if m.WholeOnTier(stackName, tier) && (!found || p.At.After(best.At)) { best, found = p, true } } return best, found, seen } // HoldAfterFailedUpdateWhole records the update hold naming the newest WHOLE copy, or — with none — // a hold that names nothing and says support is informed (NoWholeCopy). The copies seen are recorded // either way. Returns whether no whole copy exists. func (m *Manager) HoldAfterFailedUpdateWhole(ctx context.Context, stackName string, at time.Time, undoState string) (bool, error) { best, found, seen := m.HoldCopies(ctx, stackName) var seenS []string for _, p := range seen { seenS = append(seenS, fmt.Sprintf("tier %d at %s", p.Tier, p.At.UTC().Format(time.RFC3339))) } if !found { if m == nil || m.settings == nil { return true, fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName) } h := settings.RestoreHold{Stack: stackName, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUpdateFailed, UndoState: undoState, NoWholeCopy: true, CopiesSeen: seenS} if err := m.settings.SetRestoreHold(h); err != nil { return true, fmt.Errorf("persisting the update hold for %s: %w", stackName, err) } m.logger.Printf("[ERROR] [backup] %s is HELD STOPPED after a failed update and NO copy on this box brings it back whole (seen: %v; drive files declared: %v; undo: %q) — support must act (R-659)", stackName, seenS, m.HasDriveFileLegs(stackName), undoState) return true, nil } if err := m.HoldAfterFailedUpdateHolding(stackName, at, best.At, best.Tier, m.UpdateCopyHolds(stackName, best.Tier), undoState); err != nil { return false, err } if h, ok := m.settings.GetRestoreHold(stackName); ok { h.CopiesSeen = seenS _ = m.settings.SetRestoreHold(h) } return false, nil } // FreshWholeCopy answers decision 13's `files_may_change` mark for the automatic update leg (v0.271.0): // is there a copy on this box, younger than maxAge, that brings the app back WHOLE — the SAME truth // table the hold uses (WholeOnTier, decisions 25 and 26), so the leg and the hold cannot disagree about // what "whole" means. The string says why, for the leg's log. func (m *Manager) FreshWholeCopy(ctx context.Context, stackName string, maxAge time.Duration, now time.Time) (bool, string) { best, found, seen := m.HoldCopies(ctx, stackName) if !found { return false, fmt.Sprintf("no copy on this box brings it back whole (%d copies seen; drive files declared: %v)", len(seen), m.HasDriveFileLegs(stackName)) } if age := now.Sub(best.At); age > maxAge { return false, fmt.Sprintf("the newest whole copy (tier %d, %s) is %s old, limit %s", best.Tier, best.At.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge) } return true, fmt.Sprintf("tier %d copy from %s", best.Tier, best.At.UTC().Format(time.RFC3339)) } // HoldNoWholeCopy reports whether the app's hold names no copy (R-659) — the page then offers no // restore button for it. func (m *Manager) HoldNoWholeCopy(stackName string) bool { if m == nil || m.settings == nil { return false } h, ok := m.settings.GetRestoreHold(stackName) return ok && h.Reason == settings.HoldReasonUpdateFailed && h.NoWholeCopy } // UpdateHold returns the stored update hold, for the operator event (R-659). func (m *Manager) UpdateHold(stackName string) (settings.RestoreHold, bool) { if m == nil || m.settings == nil { return settings.RestoreHold{}, false } h, ok := m.settings.GetRestoreHold(stackName) if !ok || h.Reason != settings.HoldReasonUpdateFailed { return settings.RestoreHold{}, false } return h, true } // ── Decision 28 (v0.269.0): the box stops an app in a crash loop or an out-of-memory storm ───────── // UnhealthyRepeatWindow is how soon a second stop counts as a repeat: the sentence then says support is // informed. const UnhealthyRepeatWindow = 24 * time.Hour // HoldUnhealthy records that the box stopped `stack` (kind "crash_loop" or "oom_storm") and returns the // trip number: 1, or 2 when the previous stop was less than 24 h ago. Same store as every hold, so no // start path — the boot sweep, the drive gate, the nightly legs — revives it silently. func (m *Manager) HoldUnhealthy(stack, kind string, at time.Time) (int, error) { if m == nil || m.settings == nil { return 0, fmt.Errorf("no settings wired — the unhealthy stop of %s cannot be recorded", stack) } trip := 1 if last, ok := m.settings.LastUnhealthyStop(stack); ok && at.Sub(last) < UnhealthyRepeatWindow { trip = 2 } h := settings.RestoreHold{Stack: stack, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUnhealthyStop, UnhealthyKind: kind, Trip: trip} if err := m.settings.SetRestoreHold(h); err != nil { return trip, fmt.Errorf("persisting the unhealthy stop of %s: %w", stack, err) } if err := m.settings.RecordUnhealthyStop(stack, at); err != nil { m.logger.Printf("[WARN] [backup] %s: recording the stop time failed: %v", stack, err) } m.logger.Printf("[WARN] [backup] %s is STOPPED by the box: %s (trip %d within %s) — Start gives it one more try (decision 28)", stack, kind, trip, UnhealthyRepeatWindow) return trip, nil } // LiftUnhealthyStop is the Start button's half: an unhealthy-stop hold is lifted, any other kind stays. func (m *Manager) LiftUnhealthyStop(stack string) bool { if m == nil || m.settings == nil { return false } ok, err := m.settings.ClearUnhealthyStopHold(stack) if err != nil { m.logger.Printf("[ERROR] [backup] lifting the unhealthy stop of %s failed: %v", stack, err) return false } return ok } // HoldKind names the kind of hold in force ("" when none): update_failed, unhealthy_stop, or "restore". func (m *Manager) HoldKind(stack string) string { if m == nil || m.settings == nil { return "" } h, ok := m.settings.GetRestoreHold(stack) if !ok { return "" } if h.Reason == "" { return "restore" } return h.Reason }