controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s

Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1
(own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable
counts as absent with a WARN) and leans on the first FRESH copy; the
backup_max_age rule applies to whichever tier is chosen. No copy anywhere:
back up first. Refused only when nothing exists and no backup can be taken.
RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured
unit proven current. The hold names the tier (második meghajtó / saját
meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text.
A successful off-site restore now lifts an update hold. The backups page
still uses Tier2UnitRestorePoint unchanged.

Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in
felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
2026-09-13 17:16:24 +02:00
parent f946b0d0ca
commit b93c1543da
16 changed files with 1028 additions and 114 deletions
+98 -35
View File
@@ -6,6 +6,7 @@ import (
"fmt"
"os"
"path/filepath"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
@@ -82,7 +83,7 @@ const (
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani. Kapcsold be a 2. mentést az alkalmazás mentési beállításainál a Mentések oldalon, és várd meg az első sikeres másolatot — utána a frissítés elindítható."
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani, és most új mentés sem készíthető róla. Ellenőrizd a Mentések oldalon, hogy az alkalmazás meghajtója elérhető-e — utána a frissítés elindítható."
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
@@ -107,11 +108,55 @@ const updateSettleWindow = 60 * time.Second
// updatePollEvery is how often the health wait re-reads the stack.
const updatePollEvery = 5 * time.Second
// UpdateRestorePoint is the precondition answer, reduced to what the update needs.
// Backup tiers (R-475), mirroring backup.UpdateTier* — stacks cannot import backup, so
// TestR475_TierConstantsAgree (cmd/controller) pins the two sets equal.
const (
UpdateTierLocal = 1 // the app's own recovery unit
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
UpdateTierOffsite = 3 // off-site
)
// UpdateRestorePoint is one proven, restorable copy the update may lean on. The backup side returns
// only proven, restorable copies (never an attempt clock, never an unopenable unit), so there is no
// "maybe" field here to forget to check.
type UpdateRestorePoint struct {
Restorable bool // an openable recovery unit exists in the Tier-2 copy
Proven bool // a copy actually succeeded (never an attempt clock)
ProvenAt time.Time // when the data in that copy was last proven copied
Tier int // UpdateTierSecondDrive / UpdateTierLocal / UpdateTierOffsite
ProvenAt time.Time // when the data in that copy was last proven written
}
func updateTierName(tier int) string {
switch tier {
case UpdateTierSecondDrive:
return "Tier 2 (second drive)"
case UpdateTierLocal:
return "Tier 1 (own recovery unit)"
case UpdateTierOffsite:
return "Tier 3 (off-site)"
}
return fmt.Sprintf("tier %d", tier)
}
// freshRestorePoint is THE age rule (backup_max_age), applied to whichever tier is being considered
// — R-475 Scenario M: a stale copy on ANY tier is stale.
//
// COMPANION RED-PROOF M (REPORT.md): check the age only for Tier 2 (let any other tier through
// whatever its age). TestR475_M_TheAgeRuleAppliesToTheChosenTier then fails: a 30-hour-old copy of
// the app's own unit carries the update with no backup first.
func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePoint) bool {
return func(p UpdateRestorePoint) bool {
return !p.ProvenAt.IsZero() && now.Sub(p.ProvenAt) <= maxAge
}
}
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
if len(pts) == 0 {
return "none"
}
parts := make([]string, 0, len(pts))
for _, p := range pts {
parts = append(parts, fmt.Sprintf("%s at %s (%s old)", updateTierName(p.Tier), p.ProvenAt.UTC().Format(time.RFC3339), now.Sub(p.ProvenAt).Round(time.Minute)))
}
return strings.Join(parts, "; ")
}
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
@@ -119,10 +164,14 @@ type UpdateRestorePoint struct {
type UpdateGuards interface {
HoldFor(name string) (bool, string)
Busy(name string) (bool, string)
RestorePoint(name string) (UpdateRestorePoint, error)
// RestorePoints walks the tiers in preference order (2, 1, 3) and returns the first copy accept
// admits (nil = any), whether one was found, and every copy looked at (R-475).
RestorePoints(ctx context.Context, name string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint)
// CanBackUp reports whether "back up first" can run for this app now (R-475 Scenario L).
CanBackUp(name string) (bool, string)
BackupNow(ctx context.Context, name string) error
SafetyDump(ctx context.Context, name string) ([]string, error)
HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint) error
}
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
@@ -199,10 +248,19 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
if m.IsMigrating() {
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
}
rp, err := g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven {
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
fmt.Sprintf("no restorable proven Tier-2 unit (restorable=%v proven=%v err=%v)", rp.Restorable, rp.Proven, err))
// R-475: any tier counts, and an app with no copy at all is backed up first by the job. So the only
// refusal left here is Scenario L — no copy on any tier AND no way to make one now. (With a copy
// but no way to back up, the job still applies the age rule and refuses then if the copy is stale.)
// Tier 3 is looked at only on this branch, so an ordinary update never waits on the network here.
//
// COMPANION RED-PROOF (REPORT.md): drop the `!found` condition. TestR475_L then fails — an app
// with a copy but no way to back up is refused too.
if canBackUp, why := g.CanBackUp(name); !canBackUp {
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
}
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
}
if ref := m.updateMemoryRefusal(name, st); ref != nil {
return ref
@@ -383,15 +441,15 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
return
}
rp, err := g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven {
fail(fmt.Sprintf(MsgUpdateNoBackupFmt, name), fmt.Sprintf("precondition vanished: restorable=%v proven=%v err=%v", rp.Restorable, rp.Proven, err))
return
}
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
maxAge := m.backupMaxAge()
if age := start.Sub(rp.ProvenAt); age > maxAge {
m.logger.Printf("[INFO] [stacks] update %s: the proven copy is %s old (limit %s) — backing up first", name, age.Round(time.Minute), maxAge)
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
if ok {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
} else {
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
fail(MsgUpdateJournalFailed, "journal write failed")
return
@@ -400,15 +458,16 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
return
}
rp, err = g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven || m.now().Sub(rp.ProvenAt) > maxAge {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup: restorable=%v proven=%v at=%s err=%v", rp.Restorable, rp.Proven, rp.ProvenAt.Format(time.RFC3339), err))
now := m.now()
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
if !ok {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
return
}
} else {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — proven copy from %s (%s old, limit %s)", name, rp.ProvenAt.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge)
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
}
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
entry.ProvenTier = rp.Tier
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
@@ -472,20 +531,20 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
}
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "journal write failed before up")
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up")
return
}
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
// Containers may already have been recreated on the new image — something may have run.
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "compose up failed: "+err.Error())
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error())
return
}
m.verifyAndConclude(ctx, name, dir, env, rp.ProvenAt, start, &entry)
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
}
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
// known, and a failure holds the app.
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, provenAt, start time.Time, entry *updateJournalEntry) {
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, start time.Time, entry *updateJournalEntry) {
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
}
@@ -493,7 +552,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
waitStart := m.now()
healthy, detail := m.updateHealth(ctx, name, timeout)
if !healthy {
m.failAndHold(ctx, name, dir, env, provenAt, "not healthy: "+detail)
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail)
return
}
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
@@ -506,7 +565,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
}
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, provenAt time.Time, why string) {
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
if _, err := m.updateCompose(dir, env, "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
@@ -514,7 +573,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
msg := MsgUpdateHoldUnsaved
if g := m.guards(); g == nil {
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
} else if err := g.HoldAfterFailedUpdate(name, m.now(), provenAt); err != nil {
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
@@ -619,6 +678,9 @@ type updateJournalEntry struct {
PrevCompose string `json:"prev_compose,omitempty"`
PrevApplied string `json:"prev_applied,omitempty"`
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
// the right copy. 0 in a journal written by v0.238.1 or older.
ProvenTier int `json:"proven_tier,omitempty"`
}
type updateJournal struct {
@@ -787,16 +849,17 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
continue
}
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
rp := UpdateRestorePoint{Tier: e.ProvenTier, ProvenAt: provenAt}
dir := filepath.Dir(st.ComposePath)
go func(name, dir string, e updateJournalEntry, provenAt time.Time) {
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
env := m.stackEnv(dir)
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.failAndHold(ctx, name, dir, env, provenAt, "resumed compose up failed: "+err.Error())
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error())
return
}
m.verifyAndConclude(ctx, name, dir, env, provenAt, e.StartedAt, &e)
}(name, dir, e, provenAt)
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
}(name, dir, e, rp)
}
return len(names)
}