controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s

Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1
(own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable
counts as absent with a WARN) and leans on the first FRESH copy; the
backup_max_age rule applies to whichever tier is chosen. No copy anywhere:
back up first. Refused only when nothing exists and no backup can be taken.
RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured
unit proven current. The hold names the tier (második meghajtó / saját
meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text.
A successful off-site restore now lifts an update hold. The backups page
still uses Tier2UnitRestorePoint unchanged.

Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in
felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
2026-09-13 17:16:24 +02:00
parent f946b0d0ca
commit b93c1543da
16 changed files with 1028 additions and 114 deletions
+183 -9
View File
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"time"
@@ -93,6 +94,149 @@ func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
return t, true
}
// ── R-475: any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0) ────
//
// Until v0.239.0 the update's precondition was Tier2UnitRestorePoint alone, so an app with no second
// drive could never be updated — even with a fresh recovery unit on its own drive and an off-site
// snapshot from last night. The ruling: every backup counts. Tier2UnitRestorePoint itself is NOT
// changed; the backups page still calls it for the „Teljes visszaállítás" action, which really does
// restore from the second drive only.
// Backup tiers, as the update precondition and the hold sentence name them.
const (
UpdateTierLocal = 1 // the app's own recovery unit on its drive — „helyi" on the restore page
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
UpdateTierOffsite = 3 // the off-site restic repository
)
// updateTierOrder is the preference order the ruling set: the second drive, then the app's own unit,
// then off-site. The first tier holding a copy the caller ACCEPTS is chosen.
var updateTierOrder = []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite}
// UpdateTierLabel is a tier's name in the customer's hold sentence. "" for an unknown tier.
func UpdateTierLabel(tier int) string {
switch tier {
case UpdateTierSecondDrive:
return "második meghajtó"
case UpdateTierLocal:
return "saját meghajtó"
case UpdateTierOffsite:
return "távoli mentés"
}
return ""
}
// updateOffsiteCheckTimeout bounds the off-site lookup. An update must not stall on an unreachable
// Storage Box: past this the off-site copy counts as ABSENT (with a WARN), and the update carries on
// with backing up first. A var only so a test can shorten it.
var updateOffsiteCheckTimeout = 15 * time.Second
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
type UpdateTierPoint struct {
Tier int
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
At time.Time
}
// UpdateRestorePoints walks the tiers in preference order (2, 1, 3) and returns the FIRST copy that
// accept admits (nil accepts any), whether one was found, and every copy it looked at on the way.
//
// It stops at the first accepted copy, so a box with a fresh second-drive copy never touches the
// network. The AGE rule is the caller's (stacks applies backup_max_age through accept) — that is what
// makes "the age rule applies to whichever tier is chosen" one rule, not three (R-475 Scenario M).
func (m *Manager) UpdateRestorePoints(ctx context.Context, stackName string, accept func(UpdateTierPoint) bool) (UpdateTierPoint, bool, []UpdateTierPoint) {
var seen []UpdateTierPoint
for _, tier := range updateTierOrder {
p, ok := m.updateTierPoint(ctx, stackName, tier)
if !ok {
continue
}
seen = append(seen, p)
if accept == nil || accept(p) {
return p, true, seen
}
}
return UpdateTierPoint{}, false, seen
}
func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier int) (UpdateTierPoint, bool) {
switch tier {
case UpdateTierSecondDrive:
get := m.updateTier2PointFn
if get == nil {
get = m.Tier2UnitRestorePoint
}
rp, err := get(stackName)
if err != nil {
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] update precondition for %s: no Tier-2 copy (%v)", stackName, err)
}
return UpdateTierPoint{}, false
}
at, ok := rp.ProvenCopyTime()
return UpdateTierPoint{Tier: tier, At: at}, ok
case UpdateTierLocal:
list := m.updateTier1PointsFn
if list == nil {
list = m.ListRestorePoints
}
pts, _ := list(stackName)
for _, rp := range pts {
if at, err := time.Parse(time.RFC3339, rp.Time); err == nil {
return UpdateTierPoint{Tier: tier, At: at}, true
}
}
return UpdateTierPoint{}, false
case UpdateTierOffsite:
inv := m.updateOffsiteInvFn
if inv == nil {
if m.settings == nil || !m.OffboxConfigured() {
return UpdateTierPoint{}, false
}
inv = m.OffsiteInventoryList
}
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
defer cancel()
got, err := inv(cctx)
if err != nil {
if !errors.Is(err, errNoOffsiteTarget) {
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
}
return UpdateTierPoint{}, false
}
for _, a := range got.Apps {
if a.App == stackName && !a.LatestAt.IsZero() {
return UpdateTierPoint{Tier: tier, At: a.LatestAt}, true
}
}
}
return UpdateTierPoint{}, false
}
// CanBackUpApp reports whether "back up first" can run for this app at all right now — the second
// half of R-475 Scenario L: an app with no copy anywhere is refused only when this is false too.
// Cheap and read-only; RunAppBackupNow re-checks everything when it actually runs.
func (m *Manager) CanBackUpApp(stackName string) (bool, string) {
if m == nil {
return false, "backup is not enabled on this box"
}
if m.stackProvider == nil {
return false, "stack provider not configured"
}
if m.migrationActive() {
return false, "a data migration is running"
}
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" || !filepath.IsAbs(drivePath) {
return false, "the app's drive cannot be resolved"
}
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
return false, fmt.Sprintf("the app's drive %s is not available", drivePath)
}
return true, ""
}
// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an
// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer
// sentence is chosen by the caller.
@@ -142,6 +286,7 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName)
start := time.Now()
var nsRoot string
legErr := func() error {
defer m.releaseRunning()
defer m.beginAdmissionRun()()
@@ -156,7 +301,7 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
if !m.admitApp(stackName) {
return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath)
}
nsRoot := m.namespaceRoot(drivePath)
nsRoot = m.namespaceRoot(drivePath)
discover := m.discoverDBs
if discover == nil {
@@ -203,16 +348,35 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
return legErr
}
m.updatePreBackupTail(stackName, nsRoot, time.Now())
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
return nil
}
// updatePreBackupTail is what "back up first" does after the capture succeeded (R-475).
//
// 1. It marks the app's OWN unit as proven current NOW. CaptureRecoveryUnit leaves the manifest alone
// when nothing changed (the checksum skip), and Tier 1's age is the newest artifact's mtime — so on
// an app with no database and no named volume a fresh "back up first" would leave Tier 1 as old as
// its last definition change, and the update would be refused forever. That is the trap
// ProvenCopyTime documents for Tier 2, one tier down. The capture has just compared the unit with
// the live definition, so "current as of now" is exactly what it established.
// 2. It runs the Tier-2 copy, and a Tier-2 failure is a WARN, not a failure: the update may lean on
// any tier, and the Tier-1 unit it can lean on was just written. Pinned by
// TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh.
func (m *Manager) updatePreBackupTail(stackName, nsRoot string, now time.Time) {
if nsRoot != "" {
if err := os.Chtimes(RecoveryUnitManifestPath(nsRoot, stackName), now, now); err != nil {
m.logger.Printf("[WARN] [backup] update pre-backup for %s: could not mark the recovery unit as proven current (%v) — its own-unit copy may read older than it is", stackName, err)
}
}
runOne := m.perAppTier2
if runOne == nil {
runOne = m.RunTier2
}
if err := runOne(stackName); err != nil {
m.logger.Printf("[ERROR] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v", stackName, err)
return fmt.Errorf("a másodlagos másolat elkészítése sikertelen: %w", err)
m.logger.Printf("[WARN] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v — not fatal: the app's own recovery unit was just captured, and an update may lean on any tier (R-475)", stackName, err)
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
return nil
}
// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the
@@ -244,9 +408,18 @@ func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) (
}
// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app,
// the time of the failure, and the PROVEN date of the copy it can be restored from. One named string so
// a test asserts it verbatim instead of retyping Hungarian (R-364).
// the time of the failure, the TIER of the copy it can be restored from (UpdateTierLabel), and that
// copy's PROVEN date. One named string so a test asserts it verbatim instead of retyping Hungarian
// (R-364). Since v0.239.0 (R-475) it names the tier: the copy may be on any of three, and each is
// restored from a different place on the Mentések page.
const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s."
// UpdateHoldLegacyFmt is the v0.237.0–v0.238.1 sentence, kept for a hold written before the tier was
// recorded (CopyTier 0) — every such hold named a Tier-2 copy, but it did not SAY so, and rewriting
// it now would state a fact the record does not hold.
const UpdateHoldLegacyFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon."
@@ -276,7 +449,7 @@ func fmtHoldTime(rfc3339 string) string {
// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
// failure on the page.
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time) error {
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error {
if m == nil || m.settings == nil {
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
}
@@ -287,11 +460,12 @@ func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate
}
if !copyDate.IsZero() {
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
h.CopyTier = copyTier
}
if err := m.settings.SetRestoreHold(h); err != nil {
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
}
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: %s)", stackName, h.CopyDate)
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate)
return nil
}