Files
felhom-controller/controller/internal/backup/update_guard.go
T
admin 7fcda8f1a4
gates / gates (push) Successful in 25s
R-699: a unit that holds no data is never an update-precondition copy (found live on 9202)
A just-installed app's unit, captured by the status refresh before any backup, satisfied
the precondition on its manifest time; tandoor's PostgreSQL was converted with no backup
of its database. Listed still; never a copy on Tier 1 or Tier 2. Red-proof RP6.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-27 12:04:03 +02:00

858 lines
40 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package backup
import (
"context"
"errors"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"os"
"path/filepath"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// ── The backup side of the guarded update (update arc slice 4, controller v0.237.0) ─────────────
//
// 09-update-architecture.md §3 decision 1 (operator ruling 2026-09-02): the safety copy for an update
// is a VERIFIED RECENT BACKUP as a PRECONDITION — not a new copy mechanism invented for the update
// path. So everything in this file composes machinery that already exists and is proven live:
// the Tier-2 unit restore's own predicate (R-102/R-103), the nightly legs (DB dump, volume dump,
// unit capture, Tier-2 mirror), the pre-restore safety dump (R-361) and the R-379 hold.
//
// The stacks package cannot import this one, so the update job reaches all of it through the
// stacks.UpdateGuards interface, implemented by an adapter in cmd/controller/main.go.
// Tier2RestorePoint is the answer to "could this app be restored from its Tier-2 copy, and from
// when?" — the predicate the destructive „Teljes visszaállítás" action is gated on.
//
// EXTRACTED, NOT DUPLICATED (slice 4). Until v0.237.0 this computation lived inline in the backups
// page handler (buildAppBackupRows). The update path needs exactly the same question answered, and
// a second copy of a predicate is how this project's two copies of `namespaceRoot` came to differ
// (R-203). So there is one function, and the page and the update both call it.
type Tier2RestorePoint struct {
// Restorable — the copy holds an OPENABLE recovery unit (Tier2Coverage.CanRestoreUnit).
Restorable bool
// CopyDate — the date the unit restore NAMES: the package's own manifest date, falling back to the
// copy date (Tier2Coverage.UnitRestoreDate, R-403). RFC3339 as recorded, "" when unknown.
CopyDate string
// CopyDateProven — a copy actually SUCCEEDED (LastSuccess is set), never merely an attempt (R-101).
CopyDateProven bool
// PackagePreserved — the newest run PRESERVED an older package instead of refreshing it (R-403).
PackagePreserved bool
// CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded.
CopyLastSuccess string
// DataDate (v0.275.0, R-696) — the mirrored unit's DATA time (unitNewestArtifact on the mirror): when
// the data the copy holds was written, which a mirror run copies but never makes newer. "" = unknown.
DataDate string
// DataUnproven (R-699) — the mirrored unit is only a captured definition (no `data`, no data file):
// never a copy the update may lean on.
DataUnproven bool
}
// restorePointFromCoverage is the pure half of the predicate.
func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint {
pkgDate, preserved := cov.UnitRestoreDate()
return Tier2RestorePoint{
Restorable: cov.CanRestoreUnit(),
CopyDate: pkgDate,
CopyDateProven: cov.CopyLastSuccess != "",
PackagePreserved: preserved,
CopyLastSuccess: cov.CopyLastSuccess,
DataDate: cov.UnitDataDate,
DataUnproven: cov.UnitDataUnproven,
}
}
// Tier2UnitRestorePoint resolves the app's recorded Tier-2 copy and returns the restore point. The
// error is the same refusal Tier2RestoreCoverage raises (no copy, drive gone, pre-v2 layout).
func (m *Manager) Tier2UnitRestorePoint(stackName string) (Tier2RestorePoint, error) {
cov, err := m.Tier2RestoreCoverage(stackName)
if err != nil {
return Tier2RestorePoint{}, err
}
return restorePointFromCoverage(cov), nil
}
// ProvenCopyTime returns WHEN the data this copy would restore was last proven copied, and false when
// there is no proven, restorable copy at all.
//
// WHY NOT CopyDate, measured rather than assumed. CopyDate is the unit MANIFEST's created_at, and a
// capture rewrites the manifest only when the app's DEFINITION changes (compose, app.yaml, controller
// version) — a nightly DB dump keeps the same file name, so it does not move it. Measured on demo-hp
// 2026-09-13: bookstack's Tier-2 mirror held `bookstack-mariadb.sql` written 2026-09-13T00:30Z while
// its manifest still read 2026-09-12T02:15:29Z. Judging "recent" by that date would call a fresh copy
// stale — and, worse, a "back up first" run would not move it either on a quiet app, so the update
// would be refused forever.
//
// So the age is the last SUCCESSFUL copy (LastSuccess), which the Tier-2 run records only when it
// actually mirrored the unit — EXCEPT when the run preserved an older package (R-403), in which case
// the package date is the honest one, because that is what the copy really holds.
func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
if !p.Restorable || !p.CopyDateProven || p.DataUnproven {
return time.Time{}, false
}
src := p.CopyLastSuccess
if p.PackagePreserved {
src = p.CopyDate
}
t, err := time.Parse(time.RFC3339, src)
if err != nil {
return time.Time{}, false
}
// v0.275.0 (R-696): a mirror run copies the unit's data; it never makes the data newer. When the
// mirror's data time is known and older than the copy, the data time is the copy's age — a mirror
// taken right after an update, of a unit whose dump is from before it, is as old as that dump.
if d, derr := time.Parse(time.RFC3339, p.DataDate); derr == nil && d.Before(t) {
return d, true
}
return t, true
}
// ── R-475: any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0) ────
//
// Until v0.239.0 the update's precondition was Tier2UnitRestorePoint alone, so an app with no second
// drive could never be updated — even with a fresh recovery unit on its own drive and an off-site
// snapshot from last night. The ruling: every backup counts. Tier2UnitRestorePoint itself is NOT
// changed; the backups page still calls it for the „Teljes visszaállítás" action, which really does
// restore from the second drive only.
// Backup tiers, as the update precondition and the hold sentence name them.
const (
UpdateTierLocal = 1 // the app's own recovery unit on its drive — „helyi" on the restore page
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
UpdateTierOffsite = 3 // the off-site restic repository
)
// updateTierOrder is the preference order the ruling set: the second drive, then the app's own unit,
// then off-site. The first tier holding a copy the caller ACCEPTS is chosen.
var updateTierOrder = []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite}
// updateTierOrderBindData (R-479, operator ruling 2026-09-13, v0.241.0) is the order for an app whose
// DATA lives in bind-mounted files outside its recovery unit: second drive, OFF-SITE, own unit. The own
// unit then holds the definition and the database dumps but not the files, so a route back that names
// it would restore settings and not data — measured on demo-hp with gokapi (v0.239.0: „a beállítások
// visszaálltak … adatot nem"). Off-site carries the mandatory file legs; it comes before the unit.
var updateTierOrderBindData = []int{UpdateTierSecondDrive, UpdateTierOffsite, UpdateTierLocal}
// DataOutsideUnit reports whether the app keeps data in bind-mounted files that the recovery unit does
// not hold — i.e. the app has classified binds. Nil provider or no binds → false (the unit holds the
// data: named volumes and database dumps). Pinned by TestR479_.
func (m *Manager) DataOutsideUnit(stackName string) bool {
if m == nil || m.stackProvider == nil {
return false
}
binds, has := m.stackProvider.GetStackClassifiedBinds(stackName)
return has && len(binds) > 0
}
// UpdateTierOrderFor is the tier order the update walks for this app (R-475 / R-479).
func (m *Manager) UpdateTierOrderFor(stackName string) []int {
if m.DataOutsideUnit(stackName) {
return updateTierOrderBindData
}
return updateTierOrder
}
// UpdateCopyHolds is the customer phrase for what a copy on `tier` holds for this app — the second half
// of the R-479 ruling: the hold sentence names WHAT the chosen copy holds, not only where it is.
//
// v0.264.0 (R-606): the phrase is a bundle key — it is a PROMISE ABOUT WHETHER THE HOUSEHOLD'S FILES
// COME BACK and must reach an English household in English. The value returned (and stored in the
// hold) is still the Hungarian, byte for byte; copyHoldsIn maps it back to its key at render time, so
// a hold written by any earlier version localises too.
func (m *Manager) UpdateCopyHolds(stackName string, tier int) string {
return util.Text(i18n.Default, updateCopyHoldsKey(m.DataOutsideUnit(stackName), tier))
}
// UpdateCopyHoldsKey is the same verdict as UpdateCopyHolds, as its bundle KEY (R-647, v0.265.0). The
// app_update_held event carries the key, not the Hungarian phrase: the hub prints the raw details as
// the mail's `Note:` line, and an English household read „a beállításokat, …" there.
func (m *Manager) UpdateCopyHoldsKey(stackName string, tier int) string {
return updateCopyHoldsKey(m.DataOutsideUnit(stackName), tier)
}
func updateCopyHoldsKey(outside bool, tier int) string {
switch tier {
case UpdateTierLocal:
if outside {
return "hold.copy_holds.db_only"
}
return "hold.copy_holds.volumes"
case UpdateTierSecondDrive, UpdateTierOffsite:
if outside {
return "hold.copy_holds.files"
}
return "hold.copy_holds.volumes"
}
return ""
}
// copyHoldKeys are the phrases a hold may have stored; the stored value is the Hungarian.
var copyHoldKeys = []string{"hold.copy_holds.db_only", "hold.copy_holds.volumes", "hold.copy_holds.files"}
// copyHoldsIn renders a STORED copy-holds phrase in lang. An unknown phrase is returned as stored —
// never an empty clause (a hold must not lose the sentence that says what the copy holds).
func copyHoldsIn(lang, stored string) string {
for _, k := range copyHoldKeys {
if util.Text(i18n.Default, k) == stored {
return util.Text(lang, k)
}
}
return stored
}
// UpdateTierLabel is a tier's name in the customer's hold sentence (Hungarian). "" for an unknown tier.
func UpdateTierLabel(tier int) string { return UpdateTierLabelIn(i18n.Default, tier) }
// UpdateTierLabelIn is UpdateTierLabel in lang (v0.264.0, R-606).
func UpdateTierLabelIn(lang string, tier int) string {
switch tier {
case UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite:
return util.Text(lang, fmt.Sprintf("hold.update.tier.%d", tier))
}
return ""
}
// updateOffsiteCheckTimeout bounds the off-site lookup. An update must not stall on an unreachable
// Storage Box: past this the off-site copy counts as ABSENT (with a WARN), and the update carries on
// with backing up first. A var only so a test can shorten it.
var updateOffsiteCheckTimeout = 15 * time.Second
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
type UpdateTierPoint struct {
Tier int
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime (capped by the
// mirror's data time), Tier 1 the unit's DATA time (ListRestorePoints → unitNewestArtifact), Tier 3
// the newest snapshot, capped by the data time the box recorded when it pushed it (v0.275.0, R-696).
At time.Time
}
// UpdateRestorePoints walks the tiers in preference order (2, 1, 3) and returns the FIRST copy that
// accept admits (nil accepts any), whether one was found, and every copy it looked at on the way.
//
// It stops at the first accepted copy, so a box with a fresh second-drive copy never touches the
// network. The AGE rule is the caller's (stacks applies backup_max_age through accept) — that is what
// makes "the age rule applies to whichever tier is chosen" one rule, not three (R-475 Scenario M).
func (m *Manager) UpdateRestorePoints(ctx context.Context, stackName string, accept func(UpdateTierPoint) bool) (UpdateTierPoint, bool, []UpdateTierPoint) {
var seen []UpdateTierPoint
for _, tier := range m.UpdateTierOrderFor(stackName) {
p, ok := m.updateTierPoint(ctx, stackName, tier)
if !ok {
continue
}
seen = append(seen, p)
if accept == nil || accept(p) {
return p, true, seen
}
}
return UpdateTierPoint{}, false, seen
}
func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier int) (UpdateTierPoint, bool) {
switch tier {
case UpdateTierSecondDrive:
get := m.updateTier2PointFn
if get == nil {
get = m.Tier2UnitRestorePoint
}
rp, err := get(stackName)
if err != nil {
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] update precondition for %s: no Tier-2 copy (%v)", stackName, err)
}
return UpdateTierPoint{}, false
}
at, ok := rp.ProvenCopyTime()
return UpdateTierPoint{Tier: tier, At: at}, ok
case UpdateTierLocal:
list := m.updateTier1PointsFn
if list == nil {
list = m.ListRestorePoints
}
pts, _ := list(stackName)
for _, rp := range pts {
if !rp.DataProven {
// R-699: a unit that is only a captured definition is not a copy of the app's data.
m.logger.Printf("[INFO] [backup] update precondition for %s: the own unit holds no data yet (no backup run has confirmed it) — not a copy", stackName)
continue
}
if at, err := time.Parse(time.RFC3339, rp.Time); err == nil {
return UpdateTierPoint{Tier: tier, At: at}, true
}
}
return UpdateTierPoint{}, false
case UpdateTierOffsite:
times := m.updateOffsiteTimesFn
if times == nil {
if m.settings == nil || !m.OffboxConfigured() {
return UpdateTierPoint{}, false
}
times = m.OffsiteSnapshotTimes
}
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
defer cancel()
got, err := times(cctx)
if err != nil {
if !errors.Is(err, errNoOffsiteTarget) {
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
}
return UpdateTierPoint{}, false
}
if at, ok := got[stackName]; ok && !at.IsZero() {
return UpdateTierPoint{Tier: tier, At: m.offsiteDataTime(stackName, at)}, true
}
}
return UpdateTierPoint{}, false
}
// CanBackUpApp reports whether "back up first" can run for this app at all right now — the second
// half of R-475 Scenario L: an app with no copy anywhere is refused only when this is false too.
// Cheap and read-only; RunAppBackupNow re-checks everything when it actually runs.
func (m *Manager) CanBackUpApp(stackName string) (bool, string) {
if m == nil {
return false, "backup is not enabled on this box"
}
if m.stackProvider == nil {
return false, "stack provider not configured"
}
if m.migrationActive() {
return false, "a data migration is running"
}
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" || !filepath.IsAbs(drivePath) {
return false, "the app's drive cannot be resolved"
}
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
return false, fmt.Sprintf("the app's drive %s is not available", drivePath)
}
return true, ""
}
// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an
// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer
// sentence is chosen by the caller.
//
// IsRunning is box-wide, deliberately: the backup/restore single-flight is box-wide, and an update's
// "back up first" leg needs that same flag — an update started beside a running backup would either
// wait on it invisibly or fail half-way.
func (m *Manager) UpdateBusy(stackName string) (bool, string) {
if m == nil {
return false, ""
}
if m.IsRunning() {
return true, "a backup or restore is running (single-flight held)"
}
if st := m.RestoreStatus(); st.Running {
return true, fmt.Sprintf("restore op %q is running for %q", st.Op, st.Stack)
}
for _, held := range m.appStop.HeldStacks() {
if held == stackName {
return true, "an app-data operation (volume dump / export / reconstitute) is holding it"
}
}
return false, ""
}
// ErrUpdateBackupNoUnit is returned when a "back up first" run completed but the app still has no
// openable Tier-2 unit — typically Tier 2 is switched off for the app or has no second target.
var ErrUpdateBackupNoUnit = util.MsgError("err.backup.a_frissites_elotti_mentes_lefutott_de")
// RunAppBackupNow runs THIS app's backup legs now, in the nightly order, and then its Tier-2 copy:
// database dump(s) → volume dump (if the app has named volumes) → recovery-unit capture → Tier-2
// mirror. It is the "back up first" of slice 4 Scenario B.
//
// Composed, not reinvented: every leg is the one runDBDumpsInternal and RunAllTier2 already run,
// including the R-181 reserve (admitApp) before the first write and the R-166 app-stop marker inside
// DumpAppVolumesSafe. What differs is only the scope — one app instead of all of them — because an
// update must not bounce every other app on the box to back up one.
func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
if m.stackProvider == nil {
return fmt.Errorf("stack provider not configured")
}
if m.migrationActive() {
return util.MsgError("err.backup.adatathelyezes_folyamatban_a_mentes_most_nem")
}
if err := m.acquireRunning(); err != nil {
return err
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName)
start := time.Now()
var nsRoot string
legErr := func() error {
defer m.releaseRunning()
defer m.beginAdmissionRun()()
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" || !filepath.IsAbs(drivePath) {
return util.MsgError("err.backup.az_alkalmazas_meghajtoja_nem_hatarozhato_meg")
}
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
return util.MsgError("err.backup.az_alkalmazas_meghajtoja_nem_elerheto", drivePath)
}
if !m.admitApp(stackName) {
return util.MsgError("err.backup.nincs_eleg_szabad_hely_a_menteshez", drivePath)
}
nsRoot = m.namespaceRoot(drivePath)
discover := m.discoverDBs
if discover == nil {
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
}
}
dbs, err := discover(ctx)
if err != nil {
return util.MsgError("err.backup.adatbazis_felderites_sikertelen", err)
}
dumped := 0
for _, db := range dbs {
if db.StackName != stackName {
continue
}
res := m.dumpOneOrDefault(ctx, db, AppDBDumpPath(nsRoot, stackName))
if res.Error != nil {
return util.MsgError("err.backup.adatbazis_mentes_sikertelen", db.ContainerName, res.Error)
}
dumped++
m.stampDataFile(stackName, RecoveryUnitPath(nsRoot, stackName), "db-dumps/"+filepath.Base(res.FilePath))
m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size))
}
if len(m.stackProvider.GetDockerVolumes(stackName)) > 0 {
dump := m.dumpVolumesSafe
if dump == nil {
dump = m.DumpAppVolumesSafe
}
if err := dump(stackName); err != nil {
return util.MsgError("err.backup.kotetmentes_sikertelen", err)
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: volume dump OK", stackName)
}
if err := m.CaptureRecoveryUnit(stackName); err != nil {
return util.MsgError("err.backup.a_mentesi_egyseg_rogzitese_sikertelen", err)
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: recovery unit captured (%d database dump(s))", stackName, dumped)
return nil
}()
if legErr != nil {
m.logger.Printf("[ERROR] [backup] update pre-backup for %s FAILED after %s: %v", stackName, time.Since(start).Round(time.Millisecond), legErr)
return legErr
}
m.updatePreBackupTail(stackName, nsRoot, time.Now())
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
return nil
}
// updatePreBackupTail is what "back up first" does after the capture succeeded (R-475).
//
// 1. It marks the app's OWN unit as proven current NOW. CaptureRecoveryUnit leaves the manifest alone
// when nothing changed (the checksum skip), and Tier 1's age is the newest artifact's mtime — so on
// an app with no database and no named volume a fresh "back up first" would leave Tier 1 as old as
// its last definition change, and the update would be refused forever. That is the trap
// ProvenCopyTime documents for Tier 2, one tier down. The capture has just compared the unit with
// the live definition, so "current as of now" is exactly what it established.
// 2. It runs the Tier-2 copy, and a Tier-2 failure is a WARN, not a failure: the update may lean on
// any tier, and the Tier-1 unit it can lean on was just written. Pinned by
// TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh.
func (m *Manager) updatePreBackupTail(stackName, nsRoot string, now time.Time) {
if nsRoot != "" {
if err := os.Chtimes(RecoveryUnitManifestPath(nsRoot, stackName), now, now); err != nil {
m.logger.Printf("[WARN] [backup] update pre-backup for %s: could not mark the recovery unit as proven current (%v) — its own-unit copy may read older than it is", stackName, err)
}
}
runOne := m.perAppTier2
if runOne == nil {
runOne = m.RunTier2
}
if err := runOne(stackName); err != nil {
m.logger.Printf("[WARN] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v — not fatal: the app's own recovery unit was just captured, and an update may lean on any tier (R-475)", stackName, err)
}
}
// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the
// pin: "the state the customer was in a minute ago". It is writeSafetyDump (R-361) unchanged — the
// same `pre-restore-` undo naming, the same pruning to three, the same never-the-canonical-name rule
// — so it is also picked up by the same exclusions (it never enters a manifest's db_dumps).
//
// Returns the paths written; an app with no database returns (nil, nil), which is a no-op and never a
// failure (measured in writeSafetyDump: `len(mine) == 0` returns an empty set).
func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) ([]string, error) {
nsRoot := m.AppNamespaceRoot(stackName)
if nsRoot == "" {
return nil, util.MsgError("err.backup.az_alkalmazas_mentesi_helye_nem_hatarozhato")
}
set, err := m.writeSafetyDump(ctx, stackName, nsRoot)
if err != nil {
return nil, err
}
var paths []string
for _, f := range set.Files {
paths = append(paths, f.Path)
}
if len(paths) == 0 {
m.logger.Printf("[INFO] [backup] update safety dump for %s: the app has no database — nothing to copy (no-op)", stackName)
} else {
m.logger.Printf("[INFO] [backup] update safety dump for %s: %d file(s) %v", stackName, len(paths), paths)
}
return paths, nil
}
// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app,
// the time of the failure, the TIER of the copy it can be restored from (UpdateTierLabel), and that
// copy's PROVEN date. One named string so a test asserts it verbatim instead of retyping Hungarian
// (R-364). Since v0.239.0 (R-475) it names the tier: the copy may be on any of three, and each is
// restored from a different place on the Mentések page.
const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s — ez a másolat %s."
// UpdateHoldTierFmt is the v0.239.0–v0.240.0 sentence, kept for a hold that recorded a tier but not
// what the copy holds (CopyHolds empty).
const UpdateHoldTierFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s."
// UpdateHoldLegacyFmt is the v0.237.0–v0.238.1 sentence, kept for a hold written before the tier was
// recorded (CopyTier 0) — every such hold named a Tier-2 copy, but it did not SAY so, and rewriting
// it now would state a fact the record does not hold.
const UpdateHoldLegacyFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon."
// holdTimeZone is where the customer-facing hold sentence renders its times. The same zone the web
// layer renders the Mentések page's copy dates in (web.getTimezone), so the date in the hold text and
// the date on the page it points at are the same string.
func holdTimeZone() *time.Location {
if loc, err := time.LoadLocation("Europe/Budapest"); err == nil {
return loc
}
return time.UTC
}
func fmtHoldTime(rfc3339 string) string {
t, err := time.Parse(time.RFC3339, rfc3339)
if err != nil {
return rfc3339
}
return t.In(holdTimeZone()).Format("2006-01-02 15:04")
}
// HoldAfterFailedUpdate records that an app is held stopped because its new version did not come up
// healthy. Same storage and same gate as the R-379 hold — every start path that already refuses a
// restore hold refuses this one without being touched.
//
// It returns the error rather than only logging it, unlike holdAppAfterFailedRollback: the update job
// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
// failure on the page.
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error {
return m.HoldAfterFailedUpdateHolding(stackName, at, copyDate, copyTier, "", "")
}
// HoldAfterFailedUpdateHolding is HoldAfterFailedUpdate with the R-479 phrase for what the copy holds;
// "" records none (the tier-only sentence). The adapter in main.go computes the phrase with
// UpdateCopyHolds at hold time.
//
// undoState (v0.263.0) is what a FAILED undo left the data as (stacks.UndoState*), "" when no undo was
// attempted. RestoreHoldFor puts it in front of the sentence, so the household reads that the box
// already tried to put the app back, and in what state that left the data.
func (m *Manager) HoldAfterFailedUpdateHolding(stackName string, at time.Time, copyDate time.Time, copyTier int, copyHolds, undoState string) error {
if m == nil || m.settings == nil {
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
}
h := settings.RestoreHold{
Stack: stackName,
At: at.UTC().Format(time.RFC3339),
Reason: settings.HoldReasonUpdateFailed,
UndoState: undoState,
}
if !copyDate.IsZero() {
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
h.CopyTier = copyTier
h.CopyHolds = copyHolds
}
if err := m.settings.SetRestoreHold(h); err != nil {
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
}
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s; holds: %q; undo: %q)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate, h.CopyHolds, h.UndoState)
return nil
}
// isHeld reports whether the nightly legs must leave an app alone: it carries ANY hold, OR a guarded
// update is moving it right now.
//
// THE SECOND HALF WAS FOUND LIVE, v0.238.0 Scenario F on demo-hp 2026-09-13. During the update's
// 5-minute health wait the app is not yet held, and the periodic capture ran at 10:17:09 and wrote
// the NEW definition (alpine:3.20, which never started) into the app's PRIMARY unit, 53 s before the
// hold landed at 10:18:02. The Tier-2 mirror the hold names was intact only because the Tier-2 run is
// daily — a nightly Tier-2 falling inside a verify window would have mirrored the broken definition
// over the very copy the customer is told to restore from. An app mid-update has a restore point that
// must not move, exactly like a held one.
func (m *Manager) isHeld(stackName string) bool {
held, _ := m.RestoreHoldFor(stackName)
if held {
return true
}
return m.updatingCheck != nil && m.updatingCheck(stackName)
}
// SetUpdatingCheck wires the "is a guarded update moving this app" question (stacks.Manager.IsUpdating).
// INIT-ONLY, in main.go — pinned by TestSlice4_UpdatingCheckIsWiredAtStartup. The backup package cannot
// import stacks, which is why it is a seam.
func (m *Manager) SetUpdatingCheck(fn func(stackName string) bool) {
m.updatingCheck = fn
}
// clearUpdateHoldAfterRestore lifts an UPDATE hold once a person has restored the app successfully.
//
// "A person clears it by restoring" — slice 4 Part 3. The restore just put the app back on the
// definition and data of its recovery unit and started it, which is the exact route back the hold
// text names; leaving the hold in place would refuse the next restart of an app that is now fine.
//
// A RESTORE hold (R-379) is deliberately NOT cleared here: that hold means a previous restore already
// left the database in an unknown state, and it stays operator-cleared (`-clear-restore-hold`).
func (m *Manager) clearUpdateHoldAfterRestore(stackName string) {
if m.settings == nil {
return
}
h, ok := m.settings.GetRestoreHold(stackName)
if !ok || h.Reason != settings.HoldReasonUpdateFailed {
return
}
if _, err := m.settings.ClearRestoreHold(stackName); err != nil {
m.logger.Printf("[ERROR] [backup] %s was restored, but its update hold could not be cleared: %v", stackName, err)
return
}
m.logger.Printf("[INFO] [backup] %s: restore completed — the update hold (set %s) is CLEARED", stackName, h.At)
// R-671 (v0.272.0): the undo copies the hold kept describe the state this restore just replaced. They
// were kept so the hold's data stayed recoverable; once the app is restored whole they are dead weight
// (measured 2026-09-24 on 9202: three nextcloud copies, ~0.9 GiB, outlived the hold and nothing named
// them). Removed here, and only here — never for a restore hold (R-379), never while an update moves the
// app. Pinned by TestR671_RestoreThatClearsAnUpdateHoldRemovesItsUndoCopies.
if m.undoCopyRemover != nil && (m.updatingCheck == nil || !m.updatingCheck(stackName)) {
n := m.undoCopyRemover(stackName)
m.logger.Printf("[INFO] [backup] %s: removed %d undo cop(y/ies) the lifted update hold had kept (R-671)", stackName, n)
}
}
// SetUndoCopyRemover wires the undo-copy cleanup (R-671): stacks.Manager.RemoveUndoCopies. INIT-ONLY, main.go —
// pinned by TestUndoCopyRemoverIsWiredAtStartup (an AST walk). The backup package cannot import stacks.
func (m *Manager) SetUndoCopyRemover(fn func(stackName string) int) {
m.undoCopyRemover = fn
}
// UpdateHeldStacks is the set of apps held stopped after a failed update (R-660, v0.268.0) — the
// FOURTH way the product stops an app on purpose, and until v0.268.0 the one `classifyRunStates` did
// not know: each hold's `app_update_held` was followed ~11 s later by an `app_start_failed` for the
// same app (chaos rounds 8 and 11, 2026-09-23 night). A RESTORE hold (R-379) is not in the set: it
// has no event of its own, so the app-down alarm stays its only voice. Nil-safe.
func (m *Manager) UpdateHeldStacks() map[string]bool {
if m == nil || m.settings == nil {
return nil
}
var out map[string]bool
for _, h := range m.settings.ListRestoreHolds() {
// v0.269.0 (decision 28): an app the box stopped for a crash loop / OOM storm is stopped BY THE
// PRODUCT and has its own event (app_stopped_unhealthy) — the same class as an update hold.
if h.Reason != settings.HoldReasonUpdateFailed && h.Reason != settings.HoldReasonUnhealthyStop {
continue
}
if out == nil {
out = map[string]bool{}
}
out[h.Stack] = true
}
return out
}
// ── R-659 (v0.268.0): the hold names only a copy that can bring the app back WHOLE ────────────────
//
// MEASURED 2026-09-24 00:00 on 9202 (chaos round 11): nextcloud's update and its undo both failed; the
// hold named „saját meghajtó" (the precondition copy — decision 8 lets an update lean on any tier);
// the household pressed exactly that restore and was REFUSED, because the unit holds no copy of the
// app's files on the drive (R-538). The box had no other copy, so nothing on any page brought the app
// back. Operator ruling 2026-09-24 (`09` §3 decision 25, option A): the hold names only a copy that
// brings the app back whole; with none, it says so, says support is informed, and support is told.
//
// THE TRUTH TABLE, read from the restores' OWN refusals (measured from source, v0.267.0), not from
// what each tier stores:
//
// app own unit (1) second drive (2) off-site (3)
// no declared drive files whole (unit restore) whole („Teljes visszaállítás") whole (full restore)
// declared drive files NOT — refused (R-538) NOT — its unit restore is refused whole („Teljes
// (DeclaredDriveFileLegs) by the same guard; its file restore visszaállítás (fájlok
// only ADDS missing files, no database + adatbázis)")
//
// So the question is asked of the SAME predicate the refusal uses (DeclaredDriveFileLegs), and a test
// pins that the two cannot drift (TestR659_TruthTableAgreesWithTheRestoresRefusal). Tier 2 holds a
// file app's files AND its unit, but no single action brings the app back whole from it — R-661.
// WholeOnTier reports whether a copy on `tier` can bring this app back WHOLE through the restore the
// Mentések page offers for that tier.
func (m *Manager) WholeOnTier(stackName string, tier int) bool {
switch tier {
case UpdateTierOffsite:
return true
case UpdateTierLocal:
return !m.HasDriveFileLegs(stackName)
case UpdateTierSecondDrive:
if !m.HasDriveFileLegs(stackName) {
return true
}
// v0.269.0 (decision 26): a file app is whole on the second drive when the mirror holds BOTH an
// openable unit and its file legs — RestoreTier2Whole brings back both.
cov, err := m.Tier2RestoreCoverage(stackName)
return err == nil && cov.CanRestoreUnit() && cov.CanRestore()
}
return false
}
// HoldCopies walks EVERY tier (not only until the first acceptable one, as the update does) and
// returns the newest copy that brings the app back whole, whether there is one, and every copy seen.
func (m *Manager) HoldCopies(ctx context.Context, stackName string) (UpdateTierPoint, bool, []UpdateTierPoint) {
var seen []UpdateTierPoint
var best UpdateTierPoint
found := false
for _, tier := range []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite} {
p, ok := m.updateTierPoint(ctx, stackName, tier)
if !ok {
continue
}
seen = append(seen, p)
if m.WholeOnTier(stackName, tier) && (!found || p.At.After(best.At)) {
best, found = p, true
}
}
return best, found, seen
}
// HoldAfterFailedUpdateWhole records the update hold naming the newest WHOLE copy, or — with none —
// a hold that names nothing and says support is informed (NoWholeCopy). The copies seen are recorded
// either way. Returns whether no whole copy exists.
func (m *Manager) HoldAfterFailedUpdateWhole(ctx context.Context, stackName string, at time.Time, undoState string) (bool, error) {
best, found, seen := m.HoldCopies(ctx, stackName)
var seenS []string
for _, p := range seen {
seenS = append(seenS, fmt.Sprintf("tier %d at %s", p.Tier, p.At.UTC().Format(time.RFC3339)))
}
if !found {
if m == nil || m.settings == nil {
return true, fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
}
h := settings.RestoreHold{Stack: stackName, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUpdateFailed,
UndoState: undoState, NoWholeCopy: true, CopiesSeen: seenS}
if err := m.settings.SetRestoreHold(h); err != nil {
return true, fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
}
m.logger.Printf("[ERROR] [backup] %s is HELD STOPPED after a failed update and NO copy on this box brings it back whole (seen: %v; drive files declared: %v; undo: %q) — support must act (R-659)",
stackName, seenS, m.HasDriveFileLegs(stackName), undoState)
return true, nil
}
if err := m.HoldAfterFailedUpdateHolding(stackName, at, best.At, best.Tier, m.UpdateCopyHolds(stackName, best.Tier), undoState); err != nil {
return false, err
}
if h, ok := m.settings.GetRestoreHold(stackName); ok {
h.CopiesSeen = seenS
_ = m.settings.SetRestoreHold(h)
}
return false, nil
}
// FreshWholeCopy answers decision 13's `files_may_change` mark for the automatic update leg (v0.271.0):
// is there a copy on this box, younger than maxAge, that brings the app back WHOLE — the SAME truth
// table the hold uses (WholeOnTier, decisions 25 and 26), so the leg and the hold cannot disagree about
// what "whole" means. The string says why, for the leg's log.
func (m *Manager) FreshWholeCopy(ctx context.Context, stackName string, maxAge time.Duration, now time.Time) (bool, string) {
best, found, seen := m.HoldCopies(ctx, stackName)
if !found {
return false, fmt.Sprintf("no copy on this box brings it back whole (%d copies seen; drive files declared: %v)", len(seen), m.HasDriveFileLegs(stackName))
}
if age := now.Sub(best.At); age > maxAge {
return false, fmt.Sprintf("the newest whole copy (tier %d, %s) is %s old, limit %s", best.Tier, best.At.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge)
}
return true, fmt.Sprintf("tier %d copy from %s", best.Tier, best.At.UTC().Format(time.RFC3339))
}
// HoldNoWholeCopy reports whether the app's hold names no copy (R-659) — the page then offers no
// restore button for it.
func (m *Manager) HoldNoWholeCopy(stackName string) bool {
if m == nil || m.settings == nil {
return false
}
h, ok := m.settings.GetRestoreHold(stackName)
return ok && h.Reason == settings.HoldReasonUpdateFailed && h.NoWholeCopy
}
// UpdateHold returns the stored update hold, for the operator event (R-659).
func (m *Manager) UpdateHold(stackName string) (settings.RestoreHold, bool) {
if m == nil || m.settings == nil {
return settings.RestoreHold{}, false
}
h, ok := m.settings.GetRestoreHold(stackName)
if !ok || h.Reason != settings.HoldReasonUpdateFailed {
return settings.RestoreHold{}, false
}
return h, true
}
// ── Decision 28 (v0.269.0): the box stops an app in a crash loop or an out-of-memory storm ─────────
// UnhealthyRepeatWindow is how soon a second stop counts as a repeat: the sentence then says support is
// informed.
const UnhealthyRepeatWindow = 24 * time.Hour
// HoldUnhealthy records that the box stopped `stack` (kind "crash_loop" or "oom_storm") and returns the
// trip number: 1, or 2 when the previous stop was less than 24 h ago. Same store as every hold, so no
// start path — the boot sweep, the drive gate, the nightly legs — revives it silently.
func (m *Manager) HoldUnhealthy(stack, kind string, at time.Time) (int, error) {
if m == nil || m.settings == nil {
return 0, fmt.Errorf("no settings wired — the unhealthy stop of %s cannot be recorded", stack)
}
trip := 1
if last, ok := m.settings.LastUnhealthyStop(stack); ok && at.Sub(last) < UnhealthyRepeatWindow {
trip = 2
}
h := settings.RestoreHold{Stack: stack, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUnhealthyStop,
UnhealthyKind: kind, Trip: trip}
if err := m.settings.SetRestoreHold(h); err != nil {
return trip, fmt.Errorf("persisting the unhealthy stop of %s: %w", stack, err)
}
if err := m.settings.RecordUnhealthyStop(stack, at); err != nil {
m.logger.Printf("[WARN] [backup] %s: recording the stop time failed: %v", stack, err)
}
m.logger.Printf("[WARN] [backup] %s is STOPPED by the box: %s (trip %d within %s) — Start gives it one more try (decision 28)", stack, kind, trip, UnhealthyRepeatWindow)
return trip, nil
}
// LiftUnhealthyStop is the Start button's half: an unhealthy-stop hold is lifted, any other kind stays.
func (m *Manager) LiftUnhealthyStop(stack string) bool {
if m == nil || m.settings == nil {
return false
}
ok, err := m.settings.ClearUnhealthyStopHold(stack)
if err != nil {
m.logger.Printf("[ERROR] [backup] lifting the unhealthy stop of %s failed: %v", stack, err)
return false
}
return ok
}
// HoldKind names the kind of hold in force ("" when none): update_failed, unhealthy_stop, or "restore".
func (m *Manager) HoldKind(stack string) string {
if m == nil || m.settings == nil {
return ""
}
h, ok := m.settings.GetRestoreHold(stack)
if !ok {
return ""
}
if h.Reason == "" {
return "restore"
}
return h.Reason
}