3e813307cc
gates / gates (push) Successful in 13s
Operator ruling 2026-09-13. An app with classified binds walks second drive -> off-site -> own unit (its unit holds no files); volume apps keep 2 -> 1 -> 3. RestoreHold.CopyHolds records what the chosen copy holds and the sentence ends with it; older holds keep their tier-only sentence. Tests on both halves; red-proof: a layout-blind order fails the bind case.
581 lines
26 KiB
Go
581 lines
26 KiB
Go
package backup
|
||
|
||
import (
|
||
"context"
|
||
"errors"
|
||
"fmt"
|
||
"os"
|
||
"path/filepath"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||
)
|
||
|
||
// ── The backup side of the guarded update (update arc slice 4, controller v0.237.0) ─────────────
|
||
//
|
||
// 09-update-architecture.md §3 decision 1 (operator ruling 2026-09-02): the safety copy for an update
|
||
// is a VERIFIED RECENT BACKUP as a PRECONDITION — not a new copy mechanism invented for the update
|
||
// path. So everything in this file composes machinery that already exists and is proven live:
|
||
// the Tier-2 unit restore's own predicate (R-102/R-103), the nightly legs (DB dump, volume dump,
|
||
// unit capture, Tier-2 mirror), the pre-restore safety dump (R-361) and the R-379 hold.
|
||
//
|
||
// The stacks package cannot import this one, so the update job reaches all of it through the
|
||
// stacks.UpdateGuards interface, implemented by an adapter in cmd/controller/main.go.
|
||
|
||
// Tier2RestorePoint is the answer to "could this app be restored from its Tier-2 copy, and from
|
||
// when?" — the predicate the destructive „Teljes visszaállítás" action is gated on.
|
||
//
|
||
// EXTRACTED, NOT DUPLICATED (slice 4). Until v0.237.0 this computation lived inline in the backups
|
||
// page handler (buildAppBackupRows). The update path needs exactly the same question answered, and
|
||
// a second copy of a predicate is how this project's two copies of `namespaceRoot` came to differ
|
||
// (R-203). So there is one function, and the page and the update both call it.
|
||
type Tier2RestorePoint struct {
|
||
// Restorable — the copy holds an OPENABLE recovery unit (Tier2Coverage.CanRestoreUnit).
|
||
Restorable bool
|
||
// CopyDate — the date the unit restore NAMES: the package's own manifest date, falling back to the
|
||
// copy date (Tier2Coverage.UnitRestoreDate, R-403). RFC3339 as recorded, "" when unknown.
|
||
CopyDate string
|
||
// CopyDateProven — a copy actually SUCCEEDED (LastSuccess is set), never merely an attempt (R-101).
|
||
CopyDateProven bool
|
||
// PackagePreserved — the newest run PRESERVED an older package instead of refreshing it (R-403).
|
||
PackagePreserved bool
|
||
// CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded.
|
||
CopyLastSuccess string
|
||
}
|
||
|
||
// restorePointFromCoverage is the pure half of the predicate.
|
||
func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint {
|
||
pkgDate, preserved := cov.UnitRestoreDate()
|
||
return Tier2RestorePoint{
|
||
Restorable: cov.CanRestoreUnit(),
|
||
CopyDate: pkgDate,
|
||
CopyDateProven: cov.CopyLastSuccess != "",
|
||
PackagePreserved: preserved,
|
||
CopyLastSuccess: cov.CopyLastSuccess,
|
||
}
|
||
}
|
||
|
||
// Tier2UnitRestorePoint resolves the app's recorded Tier-2 copy and returns the restore point. The
|
||
// error is the same refusal Tier2RestoreCoverage raises (no copy, drive gone, pre-v2 layout).
|
||
func (m *Manager) Tier2UnitRestorePoint(stackName string) (Tier2RestorePoint, error) {
|
||
cov, err := m.Tier2RestoreCoverage(stackName)
|
||
if err != nil {
|
||
return Tier2RestorePoint{}, err
|
||
}
|
||
return restorePointFromCoverage(cov), nil
|
||
}
|
||
|
||
// ProvenCopyTime returns WHEN the data this copy would restore was last proven copied, and false when
|
||
// there is no proven, restorable copy at all.
|
||
//
|
||
// WHY NOT CopyDate, measured rather than assumed. CopyDate is the unit MANIFEST's created_at, and a
|
||
// capture rewrites the manifest only when the app's DEFINITION changes (compose, app.yaml, controller
|
||
// version) — a nightly DB dump keeps the same file name, so it does not move it. Measured on demo-hp
|
||
// 2026-09-13: bookstack's Tier-2 mirror held `bookstack-mariadb.sql` written 2026-09-13T00:30Z while
|
||
// its manifest still read 2026-09-12T02:15:29Z. Judging "recent" by that date would call a fresh copy
|
||
// stale — and, worse, a "back up first" run would not move it either on a quiet app, so the update
|
||
// would be refused forever.
|
||
//
|
||
// So the age is the last SUCCESSFUL copy (LastSuccess), which the Tier-2 run records only when it
|
||
// actually mirrored the unit — EXCEPT when the run preserved an older package (R-403), in which case
|
||
// the package date is the honest one, because that is what the copy really holds.
|
||
func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
|
||
if !p.Restorable || !p.CopyDateProven {
|
||
return time.Time{}, false
|
||
}
|
||
src := p.CopyLastSuccess
|
||
if p.PackagePreserved {
|
||
src = p.CopyDate
|
||
}
|
||
t, err := time.Parse(time.RFC3339, src)
|
||
if err != nil {
|
||
return time.Time{}, false
|
||
}
|
||
return t, true
|
||
}
|
||
|
||
// ── R-475: any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0) ────
|
||
//
|
||
// Until v0.239.0 the update's precondition was Tier2UnitRestorePoint alone, so an app with no second
|
||
// drive could never be updated — even with a fresh recovery unit on its own drive and an off-site
|
||
// snapshot from last night. The ruling: every backup counts. Tier2UnitRestorePoint itself is NOT
|
||
// changed; the backups page still calls it for the „Teljes visszaállítás" action, which really does
|
||
// restore from the second drive only.
|
||
|
||
// Backup tiers, as the update precondition and the hold sentence name them.
|
||
const (
|
||
UpdateTierLocal = 1 // the app's own recovery unit on its drive — „helyi" on the restore page
|
||
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
|
||
UpdateTierOffsite = 3 // the off-site restic repository
|
||
)
|
||
|
||
// updateTierOrder is the preference order the ruling set: the second drive, then the app's own unit,
|
||
// then off-site. The first tier holding a copy the caller ACCEPTS is chosen.
|
||
var updateTierOrder = []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite}
|
||
|
||
// updateTierOrderBindData (R-479, operator ruling 2026-09-13, v0.241.0) is the order for an app whose
|
||
// DATA lives in bind-mounted files outside its recovery unit: second drive, OFF-SITE, own unit. The own
|
||
// unit then holds the definition and the database dumps but not the files, so a route back that names
|
||
// it would restore settings and not data — measured on demo-hp with gokapi (v0.239.0: „a beállítások
|
||
// visszaálltak … adatot nem"). Off-site carries the mandatory file legs; it comes before the unit.
|
||
var updateTierOrderBindData = []int{UpdateTierSecondDrive, UpdateTierOffsite, UpdateTierLocal}
|
||
|
||
// DataOutsideUnit reports whether the app keeps data in bind-mounted files that the recovery unit does
|
||
// not hold — i.e. the app has classified binds. Nil provider or no binds → false (the unit holds the
|
||
// data: named volumes and database dumps). Pinned by TestR479_.
|
||
func (m *Manager) DataOutsideUnit(stackName string) bool {
|
||
if m == nil || m.stackProvider == nil {
|
||
return false
|
||
}
|
||
binds, has := m.stackProvider.GetStackClassifiedBinds(stackName)
|
||
return has && len(binds) > 0
|
||
}
|
||
|
||
// UpdateTierOrderFor is the tier order the update walks for this app (R-475 / R-479).
|
||
func (m *Manager) UpdateTierOrderFor(stackName string) []int {
|
||
if m.DataOutsideUnit(stackName) {
|
||
return updateTierOrderBindData
|
||
}
|
||
return updateTierOrder
|
||
}
|
||
|
||
// UpdateCopyHolds is the customer phrase for what a copy on `tier` holds for this app — the second half
|
||
// of the R-479 ruling: the hold sentence names WHAT the chosen copy holds, not only where it is.
|
||
func (m *Manager) UpdateCopyHolds(stackName string, tier int) string {
|
||
outside := m.DataOutsideUnit(stackName)
|
||
switch tier {
|
||
case UpdateTierLocal:
|
||
if outside {
|
||
return "csak a beállításokat és az adatbázist tartalmazza, a fájlokat nem"
|
||
}
|
||
return "a beállításokat, az adatbázist és az adatköteteket tartalmazza"
|
||
case UpdateTierSecondDrive:
|
||
if outside {
|
||
return "a beállításokat, az adatbázist és a fájlokat tartalmazza"
|
||
}
|
||
return "a beállításokat, az adatbázist és az adatköteteket tartalmazza"
|
||
case UpdateTierOffsite:
|
||
if outside {
|
||
return "a beállításokat, az adatbázist és a fájlokat tartalmazza"
|
||
}
|
||
return "a beállításokat, az adatbázist és az adatköteteket tartalmazza"
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// UpdateTierLabel is a tier's name in the customer's hold sentence. "" for an unknown tier.
|
||
func UpdateTierLabel(tier int) string {
|
||
switch tier {
|
||
case UpdateTierSecondDrive:
|
||
return "második meghajtó"
|
||
case UpdateTierLocal:
|
||
return "saját meghajtó"
|
||
case UpdateTierOffsite:
|
||
return "távoli mentés"
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// updateOffsiteCheckTimeout bounds the off-site lookup. An update must not stall on an unreachable
|
||
// Storage Box: past this the off-site copy counts as ABSENT (with a WARN), and the update carries on
|
||
// with backing up first. A var only so a test can shorten it.
|
||
var updateOffsiteCheckTimeout = 15 * time.Second
|
||
|
||
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
|
||
type UpdateTierPoint struct {
|
||
Tier int
|
||
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
|
||
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
|
||
At time.Time
|
||
}
|
||
|
||
// UpdateRestorePoints walks the tiers in preference order (2, 1, 3) and returns the FIRST copy that
|
||
// accept admits (nil accepts any), whether one was found, and every copy it looked at on the way.
|
||
//
|
||
// It stops at the first accepted copy, so a box with a fresh second-drive copy never touches the
|
||
// network. The AGE rule is the caller's (stacks applies backup_max_age through accept) — that is what
|
||
// makes "the age rule applies to whichever tier is chosen" one rule, not three (R-475 Scenario M).
|
||
func (m *Manager) UpdateRestorePoints(ctx context.Context, stackName string, accept func(UpdateTierPoint) bool) (UpdateTierPoint, bool, []UpdateTierPoint) {
|
||
var seen []UpdateTierPoint
|
||
for _, tier := range m.UpdateTierOrderFor(stackName) {
|
||
p, ok := m.updateTierPoint(ctx, stackName, tier)
|
||
if !ok {
|
||
continue
|
||
}
|
||
seen = append(seen, p)
|
||
if accept == nil || accept(p) {
|
||
return p, true, seen
|
||
}
|
||
}
|
||
return UpdateTierPoint{}, false, seen
|
||
}
|
||
|
||
func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier int) (UpdateTierPoint, bool) {
|
||
switch tier {
|
||
case UpdateTierSecondDrive:
|
||
get := m.updateTier2PointFn
|
||
if get == nil {
|
||
get = m.Tier2UnitRestorePoint
|
||
}
|
||
rp, err := get(stackName)
|
||
if err != nil {
|
||
if m.isDebug() {
|
||
m.logger.Printf("[DEBUG] [backup] update precondition for %s: no Tier-2 copy (%v)", stackName, err)
|
||
}
|
||
return UpdateTierPoint{}, false
|
||
}
|
||
at, ok := rp.ProvenCopyTime()
|
||
return UpdateTierPoint{Tier: tier, At: at}, ok
|
||
case UpdateTierLocal:
|
||
list := m.updateTier1PointsFn
|
||
if list == nil {
|
||
list = m.ListRestorePoints
|
||
}
|
||
pts, _ := list(stackName)
|
||
for _, rp := range pts {
|
||
if at, err := time.Parse(time.RFC3339, rp.Time); err == nil {
|
||
return UpdateTierPoint{Tier: tier, At: at}, true
|
||
}
|
||
}
|
||
return UpdateTierPoint{}, false
|
||
case UpdateTierOffsite:
|
||
times := m.updateOffsiteTimesFn
|
||
if times == nil {
|
||
if m.settings == nil || !m.OffboxConfigured() {
|
||
return UpdateTierPoint{}, false
|
||
}
|
||
times = m.OffsiteSnapshotTimes
|
||
}
|
||
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
|
||
defer cancel()
|
||
got, err := times(cctx)
|
||
if err != nil {
|
||
if !errors.Is(err, errNoOffsiteTarget) {
|
||
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
|
||
}
|
||
return UpdateTierPoint{}, false
|
||
}
|
||
if at, ok := got[stackName]; ok && !at.IsZero() {
|
||
return UpdateTierPoint{Tier: tier, At: at}, true
|
||
}
|
||
}
|
||
return UpdateTierPoint{}, false
|
||
}
|
||
|
||
// CanBackUpApp reports whether "back up first" can run for this app at all right now — the second
|
||
// half of R-475 Scenario L: an app with no copy anywhere is refused only when this is false too.
|
||
// Cheap and read-only; RunAppBackupNow re-checks everything when it actually runs.
|
||
func (m *Manager) CanBackUpApp(stackName string) (bool, string) {
|
||
if m == nil {
|
||
return false, "backup is not enabled on this box"
|
||
}
|
||
if m.stackProvider == nil {
|
||
return false, "stack provider not configured"
|
||
}
|
||
if m.migrationActive() {
|
||
return false, "a data migration is running"
|
||
}
|
||
drivePath := m.GetAppDrivePath(stackName)
|
||
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
||
return false, "the app's drive cannot be resolved"
|
||
}
|
||
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
|
||
return false, fmt.Sprintf("the app's drive %s is not available", drivePath)
|
||
}
|
||
return true, ""
|
||
}
|
||
|
||
// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an
|
||
// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer
|
||
// sentence is chosen by the caller.
|
||
//
|
||
// IsRunning is box-wide, deliberately: the backup/restore single-flight is box-wide, and an update's
|
||
// "back up first" leg needs that same flag — an update started beside a running backup would either
|
||
// wait on it invisibly or fail half-way.
|
||
func (m *Manager) UpdateBusy(stackName string) (bool, string) {
|
||
if m == nil {
|
||
return false, ""
|
||
}
|
||
if m.IsRunning() {
|
||
return true, "a backup or restore is running (single-flight held)"
|
||
}
|
||
if st := m.RestoreStatus(); st.Running {
|
||
return true, fmt.Sprintf("restore op %q is running for %q", st.Op, st.Stack)
|
||
}
|
||
for _, held := range m.appStop.HeldStacks() {
|
||
if held == stackName {
|
||
return true, "an app-data operation (volume dump / export / reconstitute) is holding it"
|
||
}
|
||
}
|
||
return false, ""
|
||
}
|
||
|
||
// ErrUpdateBackupNoUnit is returned when a "back up first" run completed but the app still has no
|
||
// openable Tier-2 unit — typically Tier 2 is switched off for the app or has no second target.
|
||
var ErrUpdateBackupNoUnit = errors.New("a frissítés előtti mentés lefutott, de nem jött létre visszaállítható másolat")
|
||
|
||
// RunAppBackupNow runs THIS app's backup legs now, in the nightly order, and then its Tier-2 copy:
|
||
// database dump(s) → volume dump (if the app has named volumes) → recovery-unit capture → Tier-2
|
||
// mirror. It is the "back up first" of slice 4 Scenario B.
|
||
//
|
||
// Composed, not reinvented: every leg is the one runDBDumpsInternal and RunAllTier2 already run,
|
||
// including the R-181 reserve (admitApp) before the first write and the R-166 app-stop marker inside
|
||
// DumpAppVolumesSafe. What differs is only the scope — one app instead of all of them — because an
|
||
// update must not bounce every other app on the box to back up one.
|
||
func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
|
||
if m.stackProvider == nil {
|
||
return fmt.Errorf("stack provider not configured")
|
||
}
|
||
if m.migrationActive() {
|
||
return fmt.Errorf("adatáthelyezés folyamatban — a mentés most nem indítható")
|
||
}
|
||
if err := m.acquireRunning(); err != nil {
|
||
return err
|
||
}
|
||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName)
|
||
start := time.Now()
|
||
var nsRoot string
|
||
legErr := func() error {
|
||
defer m.releaseRunning()
|
||
defer m.beginAdmissionRun()()
|
||
|
||
drivePath := m.GetAppDrivePath(stackName)
|
||
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
||
return fmt.Errorf("az alkalmazás meghajtója nem határozható meg")
|
||
}
|
||
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
|
||
return fmt.Errorf("az alkalmazás meghajtója nem elérhető (%s)", drivePath)
|
||
}
|
||
if !m.admitApp(stackName) {
|
||
return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath)
|
||
}
|
||
nsRoot = m.namespaceRoot(drivePath)
|
||
|
||
discover := m.discoverDBs
|
||
if discover == nil {
|
||
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
||
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||
}
|
||
}
|
||
dbs, err := discover(ctx)
|
||
if err != nil {
|
||
return fmt.Errorf("adatbázis-felderítés sikertelen: %w", err)
|
||
}
|
||
dumped := 0
|
||
for _, db := range dbs {
|
||
if db.StackName != stackName {
|
||
continue
|
||
}
|
||
res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug())
|
||
if res.Error != nil {
|
||
return fmt.Errorf("adatbázis-mentés sikertelen (%s): %w", db.ContainerName, res.Error)
|
||
}
|
||
dumped++
|
||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size))
|
||
}
|
||
|
||
if len(m.stackProvider.GetDockerVolumes(stackName)) > 0 {
|
||
dump := m.dumpVolumesSafe
|
||
if dump == nil {
|
||
dump = m.DumpAppVolumesSafe
|
||
}
|
||
if err := dump(stackName); err != nil {
|
||
return fmt.Errorf("kötetmentés sikertelen: %w", err)
|
||
}
|
||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: volume dump OK", stackName)
|
||
}
|
||
|
||
if err := m.CaptureRecoveryUnit(stackName); err != nil {
|
||
return fmt.Errorf("a mentési egység rögzítése sikertelen: %w", err)
|
||
}
|
||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: recovery unit captured (%d database dump(s))", stackName, dumped)
|
||
return nil
|
||
}()
|
||
if legErr != nil {
|
||
m.logger.Printf("[ERROR] [backup] update pre-backup for %s FAILED after %s: %v", stackName, time.Since(start).Round(time.Millisecond), legErr)
|
||
return legErr
|
||
}
|
||
|
||
m.updatePreBackupTail(stackName, nsRoot, time.Now())
|
||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
|
||
return nil
|
||
}
|
||
|
||
// updatePreBackupTail is what "back up first" does after the capture succeeded (R-475).
|
||
//
|
||
// 1. It marks the app's OWN unit as proven current NOW. CaptureRecoveryUnit leaves the manifest alone
|
||
// when nothing changed (the checksum skip), and Tier 1's age is the newest artifact's mtime — so on
|
||
// an app with no database and no named volume a fresh "back up first" would leave Tier 1 as old as
|
||
// its last definition change, and the update would be refused forever. That is the trap
|
||
// ProvenCopyTime documents for Tier 2, one tier down. The capture has just compared the unit with
|
||
// the live definition, so "current as of now" is exactly what it established.
|
||
// 2. It runs the Tier-2 copy, and a Tier-2 failure is a WARN, not a failure: the update may lean on
|
||
// any tier, and the Tier-1 unit it can lean on was just written. Pinned by
|
||
// TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh.
|
||
func (m *Manager) updatePreBackupTail(stackName, nsRoot string, now time.Time) {
|
||
if nsRoot != "" {
|
||
if err := os.Chtimes(RecoveryUnitManifestPath(nsRoot, stackName), now, now); err != nil {
|
||
m.logger.Printf("[WARN] [backup] update pre-backup for %s: could not mark the recovery unit as proven current (%v) — its own-unit copy may read older than it is", stackName, err)
|
||
}
|
||
}
|
||
runOne := m.perAppTier2
|
||
if runOne == nil {
|
||
runOne = m.RunTier2
|
||
}
|
||
if err := runOne(stackName); err != nil {
|
||
m.logger.Printf("[WARN] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v — not fatal: the app's own recovery unit was just captured, and an update may lean on any tier (R-475)", stackName, err)
|
||
}
|
||
}
|
||
|
||
// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the
|
||
// pin: "the state the customer was in a minute ago". It is writeSafetyDump (R-361) unchanged — the
|
||
// same `pre-restore-` undo naming, the same pruning to three, the same never-the-canonical-name rule
|
||
// — so it is also picked up by the same exclusions (it never enters a manifest's db_dumps).
|
||
//
|
||
// Returns the paths written; an app with no database returns (nil, nil), which is a no-op and never a
|
||
// failure (measured in writeSafetyDump: `len(mine) == 0` returns an empty set).
|
||
func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) ([]string, error) {
|
||
nsRoot := m.AppNamespaceRoot(stackName)
|
||
if nsRoot == "" {
|
||
return nil, fmt.Errorf("az alkalmazás mentési helye nem határozható meg")
|
||
}
|
||
set, err := m.writeSafetyDump(ctx, stackName, nsRoot)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
var paths []string
|
||
for _, f := range set.Files {
|
||
paths = append(paths, f.Path)
|
||
}
|
||
if len(paths) == 0 {
|
||
m.logger.Printf("[INFO] [backup] update safety dump for %s: the app has no database — nothing to copy (no-op)", stackName)
|
||
} else {
|
||
m.logger.Printf("[INFO] [backup] update safety dump for %s: %d file(s) %v", stackName, len(paths), paths)
|
||
}
|
||
return paths, nil
|
||
}
|
||
|
||
// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app,
|
||
// the time of the failure, the TIER of the copy it can be restored from (UpdateTierLabel), and that
|
||
// copy's PROVEN date. One named string so a test asserts it verbatim instead of retyping Hungarian
|
||
// (R-364). Since v0.239.0 (R-475) it names the tier: the copy may be on any of three, and each is
|
||
// restored from a different place on the Mentések page.
|
||
const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
|
||
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
|
||
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s — ez a másolat %s."
|
||
|
||
// UpdateHoldTierFmt is the v0.239.0–v0.240.0 sentence, kept for a hold that recorded a tier but not
|
||
// what the copy holds (CopyHolds empty).
|
||
const UpdateHoldTierFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
|
||
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
|
||
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s."
|
||
|
||
// UpdateHoldLegacyFmt is the v0.237.0–v0.238.1 sentence, kept for a hold written before the tier was
|
||
// recorded (CopyTier 0) — every such hold named a Tier-2 copy, but it did not SAY so, and rewriting
|
||
// it now would state a fact the record does not hold.
|
||
const UpdateHoldLegacyFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
|
||
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
|
||
"Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon."
|
||
|
||
// holdTimeZone is where the customer-facing hold sentence renders its times. The same zone the web
|
||
// layer renders the Mentések page's copy dates in (web.getTimezone), so the date in the hold text and
|
||
// the date on the page it points at are the same string.
|
||
func holdTimeZone() *time.Location {
|
||
if loc, err := time.LoadLocation("Europe/Budapest"); err == nil {
|
||
return loc
|
||
}
|
||
return time.UTC
|
||
}
|
||
|
||
func fmtHoldTime(rfc3339 string) string {
|
||
t, err := time.Parse(time.RFC3339, rfc3339)
|
||
if err != nil {
|
||
return rfc3339
|
||
}
|
||
return t.In(holdTimeZone()).Format("2006-01-02 15:04")
|
||
}
|
||
|
||
// HoldAfterFailedUpdate records that an app is held stopped because its new version did not come up
|
||
// healthy. Same storage and same gate as the R-379 hold — every start path that already refuses a
|
||
// restore hold refuses this one without being touched.
|
||
//
|
||
// It returns the error rather than only logging it, unlike holdAppAfterFailedRollback: the update job
|
||
// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app
|
||
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
|
||
// failure on the page.
|
||
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error {
|
||
return m.HoldAfterFailedUpdateHolding(stackName, at, copyDate, copyTier, "")
|
||
}
|
||
|
||
// HoldAfterFailedUpdateHolding is HoldAfterFailedUpdate with the R-479 phrase for what the copy holds;
|
||
// "" records none (the tier-only sentence). The adapter in main.go computes the phrase with
|
||
// UpdateCopyHolds at hold time.
|
||
func (m *Manager) HoldAfterFailedUpdateHolding(stackName string, at time.Time, copyDate time.Time, copyTier int, copyHolds string) error {
|
||
if m == nil || m.settings == nil {
|
||
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
|
||
}
|
||
h := settings.RestoreHold{
|
||
Stack: stackName,
|
||
At: at.UTC().Format(time.RFC3339),
|
||
Reason: settings.HoldReasonUpdateFailed,
|
||
}
|
||
if !copyDate.IsZero() {
|
||
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
|
||
h.CopyTier = copyTier
|
||
h.CopyHolds = copyHolds
|
||
}
|
||
if err := m.settings.SetRestoreHold(h); err != nil {
|
||
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
|
||
}
|
||
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s; holds: %q)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate, h.CopyHolds)
|
||
return nil
|
||
}
|
||
|
||
// isHeld reports whether the nightly legs must leave an app alone: it carries ANY hold, OR a guarded
|
||
// update is moving it right now.
|
||
//
|
||
// THE SECOND HALF WAS FOUND LIVE, v0.238.0 Scenario F on demo-hp 2026-09-13. During the update's
|
||
// 5-minute health wait the app is not yet held, and the periodic capture ran at 10:17:09 and wrote
|
||
// the NEW definition (alpine:3.20, which never started) into the app's PRIMARY unit, 53 s before the
|
||
// hold landed at 10:18:02. The Tier-2 mirror the hold names was intact only because the Tier-2 run is
|
||
// daily — a nightly Tier-2 falling inside a verify window would have mirrored the broken definition
|
||
// over the very copy the customer is told to restore from. An app mid-update has a restore point that
|
||
// must not move, exactly like a held one.
|
||
func (m *Manager) isHeld(stackName string) bool {
|
||
held, _ := m.RestoreHoldFor(stackName)
|
||
if held {
|
||
return true
|
||
}
|
||
return m.updatingCheck != nil && m.updatingCheck(stackName)
|
||
}
|
||
|
||
// SetUpdatingCheck wires the "is a guarded update moving this app" question (stacks.Manager.IsUpdating).
|
||
// INIT-ONLY, in main.go — pinned by TestSlice4_UpdatingCheckIsWiredAtStartup. The backup package cannot
|
||
// import stacks, which is why it is a seam.
|
||
func (m *Manager) SetUpdatingCheck(fn func(stackName string) bool) {
|
||
m.updatingCheck = fn
|
||
}
|
||
|
||
// clearUpdateHoldAfterRestore lifts an UPDATE hold once a person has restored the app successfully.
|
||
//
|
||
// "A person clears it by restoring" — slice 4 Part 3. The restore just put the app back on the
|
||
// definition and data of its recovery unit and started it, which is the exact route back the hold
|
||
// text names; leaving the hold in place would refuse the next restart of an app that is now fine.
|
||
//
|
||
// A RESTORE hold (R-379) is deliberately NOT cleared here: that hold means a previous restore already
|
||
// left the database in an unknown state, and it stays operator-cleared (`-clear-restore-hold`).
|
||
func (m *Manager) clearUpdateHoldAfterRestore(stackName string) {
|
||
if m.settings == nil {
|
||
return
|
||
}
|
||
h, ok := m.settings.GetRestoreHold(stackName)
|
||
if !ok || h.Reason != settings.HoldReasonUpdateFailed {
|
||
return
|
||
}
|
||
if _, err := m.settings.ClearRestoreHold(stackName); err != nil {
|
||
m.logger.Printf("[ERROR] [backup] %s was restored, but its update hold could not be cleared: %v", stackName, err)
|
||
return
|
||
}
|
||
m.logger.Printf("[INFO] [backup] %s: restore completed — the update hold (set %s) is CLEARED", stackName, h.At)
|
||
}
|