package backup import ( "context" "errors" "fmt" "path/filepath" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // ── The backup side of the guarded update (update arc slice 4, controller v0.237.0) ───────────── // // 09-update-architecture.md §3 decision 1 (operator ruling 2026-09-02): the safety copy for an update // is a VERIFIED RECENT BACKUP as a PRECONDITION — not a new copy mechanism invented for the update // path. So everything in this file composes machinery that already exists and is proven live: // the Tier-2 unit restore's own predicate (R-102/R-103), the nightly legs (DB dump, volume dump, // unit capture, Tier-2 mirror), the pre-restore safety dump (R-361) and the R-379 hold. // // The stacks package cannot import this one, so the update job reaches all of it through the // stacks.UpdateGuards interface, implemented by an adapter in cmd/controller/main.go. // Tier2RestorePoint is the answer to "could this app be restored from its Tier-2 copy, and from // when?" — the predicate the destructive „Teljes visszaállítás" action is gated on. // // EXTRACTED, NOT DUPLICATED (slice 4). Until v0.237.0 this computation lived inline in the backups // page handler (buildAppBackupRows). The update path needs exactly the same question answered, and // a second copy of a predicate is how this project's two copies of `namespaceRoot` came to differ // (R-203). So there is one function, and the page and the update both call it. type Tier2RestorePoint struct { // Restorable — the copy holds an OPENABLE recovery unit (Tier2Coverage.CanRestoreUnit). Restorable bool // CopyDate — the date the unit restore NAMES: the package's own manifest date, falling back to the // copy date (Tier2Coverage.UnitRestoreDate, R-403). RFC3339 as recorded, "" when unknown. CopyDate string // CopyDateProven — a copy actually SUCCEEDED (LastSuccess is set), never merely an attempt (R-101). CopyDateProven bool // PackagePreserved — the newest run PRESERVED an older package instead of refreshing it (R-403). PackagePreserved bool // CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded. CopyLastSuccess string } // restorePointFromCoverage is the pure half of the predicate. func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint { pkgDate, preserved := cov.UnitRestoreDate() return Tier2RestorePoint{ Restorable: cov.CanRestoreUnit(), CopyDate: pkgDate, CopyDateProven: cov.CopyLastSuccess != "", PackagePreserved: preserved, CopyLastSuccess: cov.CopyLastSuccess, } } // Tier2UnitRestorePoint resolves the app's recorded Tier-2 copy and returns the restore point. The // error is the same refusal Tier2RestoreCoverage raises (no copy, drive gone, pre-v2 layout). func (m *Manager) Tier2UnitRestorePoint(stackName string) (Tier2RestorePoint, error) { cov, err := m.Tier2RestoreCoverage(stackName) if err != nil { return Tier2RestorePoint{}, err } return restorePointFromCoverage(cov), nil } // ProvenCopyTime returns WHEN the data this copy would restore was last proven copied, and false when // there is no proven, restorable copy at all. // // WHY NOT CopyDate, measured rather than assumed. CopyDate is the unit MANIFEST's created_at, and a // capture rewrites the manifest only when the app's DEFINITION changes (compose, app.yaml, controller // version) — a nightly DB dump keeps the same file name, so it does not move it. Measured on demo-hp // 2026-09-13: bookstack's Tier-2 mirror held `bookstack-mariadb.sql` written 2026-09-13T00:30Z while // its manifest still read 2026-09-12T02:15:29Z. Judging "recent" by that date would call a fresh copy // stale — and, worse, a "back up first" run would not move it either on a quiet app, so the update // would be refused forever. // // So the age is the last SUCCESSFUL copy (LastSuccess), which the Tier-2 run records only when it // actually mirrored the unit — EXCEPT when the run preserved an older package (R-403), in which case // the package date is the honest one, because that is what the copy really holds. func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) { if !p.Restorable || !p.CopyDateProven { return time.Time{}, false } src := p.CopyLastSuccess if p.PackagePreserved { src = p.CopyDate } t, err := time.Parse(time.RFC3339, src) if err != nil { return time.Time{}, false } return t, true } // UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an // update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer // sentence is chosen by the caller. // // IsRunning is box-wide, deliberately: the backup/restore single-flight is box-wide, and an update's // "back up first" leg needs that same flag — an update started beside a running backup would either // wait on it invisibly or fail half-way. func (m *Manager) UpdateBusy(stackName string) (bool, string) { if m == nil { return false, "" } if m.IsRunning() { return true, "a backup or restore is running (single-flight held)" } if st := m.RestoreStatus(); st.Running { return true, fmt.Sprintf("restore op %q is running for %q", st.Op, st.Stack) } for _, held := range m.appStop.HeldStacks() { if held == stackName { return true, "an app-data operation (volume dump / export / reconstitute) is holding it" } } return false, "" } // ErrUpdateBackupNoUnit is returned when a "back up first" run completed but the app still has no // openable Tier-2 unit — typically Tier 2 is switched off for the app or has no second target. var ErrUpdateBackupNoUnit = errors.New("a frissítés előtti mentés lefutott, de nem jött létre visszaállítható másolat") // RunAppBackupNow runs THIS app's backup legs now, in the nightly order, and then its Tier-2 copy: // database dump(s) → volume dump (if the app has named volumes) → recovery-unit capture → Tier-2 // mirror. It is the "back up first" of slice 4 Scenario B. // // Composed, not reinvented: every leg is the one runDBDumpsInternal and RunAllTier2 already run, // including the R-181 reserve (admitApp) before the first write and the R-166 app-stop marker inside // DumpAppVolumesSafe. What differs is only the scope — one app instead of all of them — because an // update must not bounce every other app on the box to back up one. func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error { if m.stackProvider == nil { return fmt.Errorf("stack provider not configured") } if m.migrationActive() { return fmt.Errorf("adatáthelyezés folyamatban — a mentés most nem indítható") } if err := m.acquireRunning(); err != nil { return err } m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName) start := time.Now() legErr := func() error { defer m.releaseRunning() defer m.beginAdmissionRun()() drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { return fmt.Errorf("az alkalmazás meghajtója nem határozható meg") } if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) { return fmt.Errorf("az alkalmazás meghajtója nem elérhető (%s)", drivePath) } if !m.admitApp(stackName) { return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath) } nsRoot := m.namespaceRoot(drivePath) discover := m.discoverDBs if discover == nil { discover = func(ctx context.Context) ([]DiscoveredDB, error) { return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames()) } } dbs, err := discover(ctx) if err != nil { return fmt.Errorf("adatbázis-felderítés sikertelen: %w", err) } dumped := 0 for _, db := range dbs { if db.StackName != stackName { continue } res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug()) if res.Error != nil { return fmt.Errorf("adatbázis-mentés sikertelen (%s): %w", db.ContainerName, res.Error) } dumped++ m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size)) } if len(m.stackProvider.GetDockerVolumes(stackName)) > 0 { dump := m.dumpVolumesSafe if dump == nil { dump = m.DumpAppVolumesSafe } if err := dump(stackName); err != nil { return fmt.Errorf("kötetmentés sikertelen: %w", err) } m.logger.Printf("[INFO] [backup] update pre-backup for %s: volume dump OK", stackName) } if err := m.CaptureRecoveryUnit(stackName); err != nil { return fmt.Errorf("a mentési egység rögzítése sikertelen: %w", err) } m.logger.Printf("[INFO] [backup] update pre-backup for %s: recovery unit captured (%d database dump(s))", stackName, dumped) return nil }() if legErr != nil { m.logger.Printf("[ERROR] [backup] update pre-backup for %s FAILED after %s: %v", stackName, time.Since(start).Round(time.Millisecond), legErr) return legErr } runOne := m.perAppTier2 if runOne == nil { runOne = m.RunTier2 } if err := runOne(stackName); err != nil { m.logger.Printf("[ERROR] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v", stackName, err) return fmt.Errorf("a másodlagos másolat elkészítése sikertelen: %w", err) } m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond)) return nil } // WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the // pin: "the state the customer was in a minute ago". It is writeSafetyDump (R-361) unchanged — the // same `pre-restore-` undo naming, the same pruning to three, the same never-the-canonical-name rule // — so it is also picked up by the same exclusions (it never enters a manifest's db_dumps). // // Returns the paths written; an app with no database returns (nil, nil), which is a no-op and never a // failure (measured in writeSafetyDump: `len(mine) == 0` returns an empty set). func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) ([]string, error) { nsRoot := m.AppNamespaceRoot(stackName) if nsRoot == "" { return nil, fmt.Errorf("az alkalmazás mentési helye nem határozható meg") } set, err := m.writeSafetyDump(ctx, stackName, nsRoot) if err != nil { return nil, err } var paths []string for _, f := range set.Files { paths = append(paths, f.Path) } if len(paths) == 0 { m.logger.Printf("[INFO] [backup] update safety dump for %s: the app has no database — nothing to copy (no-op)", stackName) } else { m.logger.Printf("[INFO] [backup] update safety dump for %s: %d file(s) %v", stackName, len(paths), paths) } return paths, nil } // UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app, // the time of the failure, and the PROVEN date of the copy it can be restored from. One named string so // a test asserts it verbatim instead of retyping Hungarian (R-364). const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " + "Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " + "Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon." // holdTimeZone is where the customer-facing hold sentence renders its times. The same zone the web // layer renders the Mentések page's copy dates in (web.getTimezone), so the date in the hold text and // the date on the page it points at are the same string. func holdTimeZone() *time.Location { if loc, err := time.LoadLocation("Europe/Budapest"); err == nil { return loc } return time.UTC } func fmtHoldTime(rfc3339 string) string { t, err := time.Parse(time.RFC3339, rfc3339) if err != nil { return rfc3339 } return t.In(holdTimeZone()).Format("2006-01-02 15:04") } // HoldAfterFailedUpdate records that an app is held stopped because its new version did not come up // healthy. Same storage and same gate as the R-379 hold — every start path that already refuses a // restore hold refuses this one without being touched. // // It returns the error rather than only logging it, unlike holdAppAfterFailedRollback: the update job // has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app // that the next restart button will quietly start again. The caller logs it at ERROR and keeps the // failure on the page. func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time) error { if m == nil || m.settings == nil { return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName) } h := settings.RestoreHold{ Stack: stackName, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUpdateFailed, } if !copyDate.IsZero() { h.CopyDate = copyDate.UTC().Format(time.RFC3339) } if err := m.settings.SetRestoreHold(h); err != nil { return fmt.Errorf("persisting the update hold for %s: %w", stackName, err) } m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: %s)", stackName, h.CopyDate) return nil } // isHeld reports whether an app carries ANY hold. Used by the nightly legs to leave a held app alone. func (m *Manager) isHeld(stackName string) bool { held, _ := m.RestoreHoldFor(stackName) return held } // clearUpdateHoldAfterRestore lifts an UPDATE hold once a person has restored the app successfully. // // "A person clears it by restoring" — slice 4 Part 3. The restore just put the app back on the // definition and data of its recovery unit and started it, which is the exact route back the hold // text names; leaving the hold in place would refuse the next restart of an app that is now fine. // // A RESTORE hold (R-379) is deliberately NOT cleared here: that hold means a previous restore already // left the database in an unknown state, and it stays operator-cleared (`-clear-restore-hold`). func (m *Manager) clearUpdateHoldAfterRestore(stackName string) { if m.settings == nil { return } h, ok := m.settings.GetRestoreHold(stackName) if !ok || h.Reason != settings.HoldReasonUpdateFailed { return } if _, err := m.settings.ClearRestoreHold(stackName); err != nil { m.logger.Printf("[ERROR] [backup] %s was restored, but its update hold could not be cleared: %v", stackName, err) return } m.logger.Printf("[INFO] [backup] %s: restore completed — the update hold (set %s) is CLEARED", stackName, h.At) }