controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s

Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1
(own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable
counts as absent with a WARN) and leans on the first FRESH copy; the
backup_max_age rule applies to whichever tier is chosen. No copy anywhere:
back up first. Refused only when nothing exists and no backup can be taken.
RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured
unit proven current. The hold names the tier (második meghajtó / saját
meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text.
A successful off-site restore now lifts an update hold. The backups page
still uses Tier2UnitRestorePoint unchanged.

Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in
felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
2026-09-13 17:16:24 +02:00
parent f946b0d0ca
commit b93c1543da
16 changed files with 1028 additions and 114 deletions
+98 -35
View File
@@ -6,6 +6,7 @@ import (
"fmt"
"os"
"path/filepath"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
@@ -82,7 +83,7 @@ const (
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani. Kapcsold be a 2. mentést az alkalmazás mentési beállításainál a Mentések oldalon, és várd meg az első sikeres másolatot — utána a frissítés elindítható."
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani, és most új mentés sem készíthető róla. Ellenőrizd a Mentések oldalon, hogy az alkalmazás meghajtója elérhető-e — utána a frissítés elindítható."
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
@@ -107,11 +108,55 @@ const updateSettleWindow = 60 * time.Second
// updatePollEvery is how often the health wait re-reads the stack.
const updatePollEvery = 5 * time.Second
// UpdateRestorePoint is the precondition answer, reduced to what the update needs.
// Backup tiers (R-475), mirroring backup.UpdateTier* — stacks cannot import backup, so
// TestR475_TierConstantsAgree (cmd/controller) pins the two sets equal.
const (
UpdateTierLocal = 1 // the app's own recovery unit
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
UpdateTierOffsite = 3 // off-site
)
// UpdateRestorePoint is one proven, restorable copy the update may lean on. The backup side returns
// only proven, restorable copies (never an attempt clock, never an unopenable unit), so there is no
// "maybe" field here to forget to check.
type UpdateRestorePoint struct {
Restorable bool // an openable recovery unit exists in the Tier-2 copy
Proven bool // a copy actually succeeded (never an attempt clock)
ProvenAt time.Time // when the data in that copy was last proven copied
Tier int // UpdateTierSecondDrive / UpdateTierLocal / UpdateTierOffsite
ProvenAt time.Time // when the data in that copy was last proven written
}
func updateTierName(tier int) string {
switch tier {
case UpdateTierSecondDrive:
return "Tier 2 (second drive)"
case UpdateTierLocal:
return "Tier 1 (own recovery unit)"
case UpdateTierOffsite:
return "Tier 3 (off-site)"
}
return fmt.Sprintf("tier %d", tier)
}
// freshRestorePoint is THE age rule (backup_max_age), applied to whichever tier is being considered
// — R-475 Scenario M: a stale copy on ANY tier is stale.
//
// COMPANION RED-PROOF M (REPORT.md): check the age only for Tier 2 (let any other tier through
// whatever its age). TestR475_M_TheAgeRuleAppliesToTheChosenTier then fails: a 30-hour-old copy of
// the app's own unit carries the update with no backup first.
func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePoint) bool {
return func(p UpdateRestorePoint) bool {
return !p.ProvenAt.IsZero() && now.Sub(p.ProvenAt) <= maxAge
}
}
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
if len(pts) == 0 {
return "none"
}
parts := make([]string, 0, len(pts))
for _, p := range pts {
parts = append(parts, fmt.Sprintf("%s at %s (%s old)", updateTierName(p.Tier), p.ProvenAt.UTC().Format(time.RFC3339), now.Sub(p.ProvenAt).Round(time.Minute)))
}
return strings.Join(parts, "; ")
}
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
@@ -119,10 +164,14 @@ type UpdateRestorePoint struct {
type UpdateGuards interface {
HoldFor(name string) (bool, string)
Busy(name string) (bool, string)
RestorePoint(name string) (UpdateRestorePoint, error)
// RestorePoints walks the tiers in preference order (2, 1, 3) and returns the first copy accept
// admits (nil = any), whether one was found, and every copy looked at (R-475).
RestorePoints(ctx context.Context, name string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint)
// CanBackUp reports whether "back up first" can run for this app now (R-475 Scenario L).
CanBackUp(name string) (bool, string)
BackupNow(ctx context.Context, name string) error
SafetyDump(ctx context.Context, name string) ([]string, error)
HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint) error
}
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
@@ -199,10 +248,19 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
if m.IsMigrating() {
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
}
rp, err := g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven {
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
fmt.Sprintf("no restorable proven Tier-2 unit (restorable=%v proven=%v err=%v)", rp.Restorable, rp.Proven, err))
// R-475: any tier counts, and an app with no copy at all is backed up first by the job. So the only
// refusal left here is Scenario L — no copy on any tier AND no way to make one now. (With a copy
// but no way to back up, the job still applies the age rule and refuses then if the copy is stale.)
// Tier 3 is looked at only on this branch, so an ordinary update never waits on the network here.
//
// COMPANION RED-PROOF (REPORT.md): drop the `!found` condition. TestR475_L then fails — an app
// with a copy but no way to back up is refused too.
if canBackUp, why := g.CanBackUp(name); !canBackUp {
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
}
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
}
if ref := m.updateMemoryRefusal(name, st); ref != nil {
return ref
@@ -383,15 +441,15 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
return
}
rp, err := g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven {
fail(fmt.Sprintf(MsgUpdateNoBackupFmt, name), fmt.Sprintf("precondition vanished: restorable=%v proven=%v err=%v", rp.Restorable, rp.Proven, err))
return
}
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
maxAge := m.backupMaxAge()
if age := start.Sub(rp.ProvenAt); age > maxAge {
m.logger.Printf("[INFO] [stacks] update %s: the proven copy is %s old (limit %s) — backing up first", name, age.Round(time.Minute), maxAge)
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
if ok {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
} else {
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
fail(MsgUpdateJournalFailed, "journal write failed")
return
@@ -400,15 +458,16 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
return
}
rp, err = g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven || m.now().Sub(rp.ProvenAt) > maxAge {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup: restorable=%v proven=%v at=%s err=%v", rp.Restorable, rp.Proven, rp.ProvenAt.Format(time.RFC3339), err))
now := m.now()
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
if !ok {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
return
}
} else {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — proven copy from %s (%s old, limit %s)", name, rp.ProvenAt.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge)
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
}
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
entry.ProvenTier = rp.Tier
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
@@ -472,20 +531,20 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
}
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "journal write failed before up")
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up")
return
}
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
// Containers may already have been recreated on the new image — something may have run.
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "compose up failed: "+err.Error())
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error())
return
}
m.verifyAndConclude(ctx, name, dir, env, rp.ProvenAt, start, &entry)
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
}
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
// known, and a failure holds the app.
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, provenAt, start time.Time, entry *updateJournalEntry) {
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, start time.Time, entry *updateJournalEntry) {
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
}
@@ -493,7 +552,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
waitStart := m.now()
healthy, detail := m.updateHealth(ctx, name, timeout)
if !healthy {
m.failAndHold(ctx, name, dir, env, provenAt, "not healthy: "+detail)
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail)
return
}
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
@@ -506,7 +565,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
}
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, provenAt time.Time, why string) {
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
if _, err := m.updateCompose(dir, env, "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
@@ -514,7 +573,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
msg := MsgUpdateHoldUnsaved
if g := m.guards(); g == nil {
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
} else if err := g.HoldAfterFailedUpdate(name, m.now(), provenAt); err != nil {
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
@@ -619,6 +678,9 @@ type updateJournalEntry struct {
PrevCompose string `json:"prev_compose,omitempty"`
PrevApplied string `json:"prev_applied,omitempty"`
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
// the right copy. 0 in a journal written by v0.238.1 or older.
ProvenTier int `json:"proven_tier,omitempty"`
}
type updateJournal struct {
@@ -787,16 +849,17 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
continue
}
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
rp := UpdateRestorePoint{Tier: e.ProvenTier, ProvenAt: provenAt}
dir := filepath.Dir(st.ComposePath)
go func(name, dir string, e updateJournalEntry, provenAt time.Time) {
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
env := m.stackEnv(dir)
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.failAndHold(ctx, name, dir, env, provenAt, "resumed compose up failed: "+err.Error())
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error())
return
}
m.verifyAndConclude(ctx, name, dir, env, provenAt, e.StartedAt, &e)
}(name, dir, e, provenAt)
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
}(name, dir, e, rp)
}
return len(names)
}
+48 -38
View File
@@ -25,13 +25,15 @@ type fakeGuards struct {
held bool
holdWhy string
busy bool
rp UpdateRestorePoint
rpErr error
rpAfterBackup *UpdateRestorePoint
// points are the copies the backup side holds, in tier order (R-475); pointsAfterBackup replaces
// them when BackupNow succeeds (nil = the backup changed nothing).
points []UpdateRestorePoint
pointsAfterBackup []UpdateRestorePoint
cannotBackUp bool
backupErr error
dumpErr error
holdErr error
holdProvenAt time.Time
holdRP UpdateRestorePoint
pinAtDump string
stackDir string
}
@@ -48,18 +50,29 @@ func (f *fakeGuards) HoldFor(string) (bool, string) {
return f.held, f.holdWhy
}
func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" }
func (f *fakeGuards) RestorePoint(string) (UpdateRestorePoint, error) {
f.note("RestorePoint")
func (f *fakeGuards) RestorePoints(_ context.Context, _ string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint) {
f.note("RestorePoints")
f.mu.Lock()
defer f.mu.Unlock()
return f.rp, f.rpErr
var seen []UpdateRestorePoint
for _, p := range f.points {
seen = append(seen, p)
if accept == nil || accept(p) {
return p, true, seen
}
}
return UpdateRestorePoint{}, false, seen
}
func (f *fakeGuards) CanBackUp(string) (bool, string) {
f.note("CanBackUp")
return !f.cannotBackUp, "fake: the drive is gone"
}
func (f *fakeGuards) BackupNow(context.Context, string) error {
f.note("BackupNow")
f.mu.Lock()
defer f.mu.Unlock()
if f.backupErr == nil && f.rpAfterBackup != nil {
f.rp = *f.rpAfterBackup
if f.backupErr == nil && f.pointsAfterBackup != nil {
f.points = f.pointsAfterBackup
}
return f.backupErr
}
@@ -72,14 +85,14 @@ func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
}
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
}
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, provenAt time.Time) error {
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint) error {
f.note("HoldAfterFailedUpdate")
f.mu.Lock()
defer f.mu.Unlock()
if f.holdErr != nil {
return f.holdErr
}
f.held, f.holdWhy, f.holdProvenAt = true, "HELD-SENTENCE", provenAt
f.held, f.holdWhy, f.holdRP = true, "HELD-SENTENCE", rp
return nil
}
@@ -111,7 +124,7 @@ func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec)
m, dir := newPinManager(t, pinTplOld, pinTplNew,
"deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n")
mustWrite(t, AppliedComposePath(dir), pinTplOld)
g := &fakeGuards{rp: UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Hour)}, stackDir: dir}
g := &fakeGuards{points: []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: slice4T0.Add(-1 * time.Hour)}}, stackDir: dir}
c := &composeRec{fail: map[string]error{}}
m.updateGuards = g
m.updateComposeFn = c.fn
@@ -212,9 +225,8 @@ func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
t.Errorf("compose calls = %q, want %q", got, want)
}
// RestorePoint twice by design: once in the preflight (the refusal), once inside the job (the
// precondition must still hold when the job actually starts).
if got := strings.Join(g.callList(), ","); got != "RestorePoint,RestorePoint,SafetyDump" {
// R-475: the preflight asks only whether a backup could be taken; the job reads the copies once.
if got := strings.Join(g.callList(), ","); got != "CanBackUp,RestorePoints,SafetyDump" {
t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got)
}
}
@@ -223,8 +235,8 @@ func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
m, dir, g, _ := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
g.rpAfterBackup = &UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Minute)}
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: slice4T0.Add(-1 * time.Minute)}}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
@@ -233,7 +245,7 @@ func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
calls := strings.Join(g.callList(), ",")
if !strings.HasPrefix(calls, "RestorePoint,RestorePoint,BackupNow,RestorePoint,SafetyDump") {
if !strings.HasPrefix(calls, "CanBackUp,RestorePoints,BackupNow,RestorePoints,SafetyDump") {
t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
@@ -243,7 +255,7 @@ func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour)
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour)
g.backupErr = errors.New("disk full")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
@@ -265,7 +277,7 @@ func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: rpAfterBackup nil
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: pointsAfterBackup nil
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
@@ -278,21 +290,18 @@ func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
}
}
// ── C: no backup exists that could restore this app ───────────────────────────────────────────────
// ── C / R-475 L: no copy on any tier, and no way to make one ─────────────────────────────────────
// COMPANION RED-PROOF 2 (REPORT.md): make the precondition in UpdatePreflight proceed when
// !rp.Restorable. This test then fails with the update started.
func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
// Until v0.239.0 this refused any app without a restorable Tier-2 unit. R-475: every tier counts and
// an app with nothing is backed up first, so the refusal is now only "nothing anywhere AND no backup
// can be taken". The K half (nothing, but a backup CAN be taken) is TestR475_K in update_tiers_test.go.
func TestSlice4_C_NoCopyAndNoWayToBackUpRefusesBeforeAnythingMoves(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
// A PROVEN, FRESH copy whose unit cannot be opened — the realistic half-copied mirror. Proven and
// fresh on purpose: a fixture that is also unproven would be refused by the proven check alone,
// and a red-proof that drops the restorable check would then pass inertly (observed on the first
// run of red-proof 2, 2026-09-13).
g.rp = UpdateRestorePoint{Restorable: false, Proven: true, ProvenAt: slice4T0.Add(-time.Hour)}
g.points, g.cannotBackUp = nil, true
err := m.StartGuardedUpdate("nextcloud")
var ref *UpdateRefusal
if !errors.As(err, &ref) || ref.Reason != "no_backup" {
t.Fatalf("an app with no restorable copy must be REFUSED (no_backup), got %v", err)
t.Fatalf("no copy anywhere and no way to back up must be REFUSED (no_backup), got %v", err)
}
if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want {
t.Errorf("message = %q", ref.Message)
@@ -304,10 +313,10 @@ func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("a refused update must move nothing")
}
// A copy that exists but was never PROVEN is not a copy (R-101).
g.rp = UpdateRestorePoint{Restorable: true, Proven: false}
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
t.Errorf("an unproven copy must refuse too, got %v", ref)
for _, call := range g.callList() {
if call == "BackupNow" {
t.Error("a refused update must not try to back up")
}
}
}
@@ -412,8 +421,8 @@ func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
if !held {
t.Fatal("an app that did not come up must be HELD")
}
if !g.holdProvenAt.Equal(g.rp.ProvenAt) {
t.Errorf("the hold must name the PROVEN copy date %s, got %s", g.rp.ProvenAt, g.holdProvenAt)
if !g.holdRP.ProvenAt.Equal(g.points[0].ProvenAt) || g.holdRP.Tier != UpdateTierSecondDrive {
t.Errorf("the hold must name the PROVEN copy it leans on (%+v), got %+v", g.points[0], g.holdRP)
}
if st.UpdateError != "HELD-SENTENCE" {
t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError)
@@ -466,6 +475,7 @@ func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry {
StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"},
PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile),
ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339),
ProvenTier: UpdateTierLocal,
}
}
@@ -530,8 +540,8 @@ func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) {
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | down" {
t.Errorf("resumption re-runs `up` then stops the failed app; compose calls = %q", got)
}
if !g.holdProvenAt.Equal(slice4T0.Add(-time.Hour)) {
t.Errorf("the resumed hold must name the journaled proven copy date, got %s", g.holdProvenAt)
if !g.holdRP.ProvenAt.Equal(slice4T0.Add(-time.Hour)) || g.holdRP.Tier != UpdateTierLocal {
t.Errorf("the resumed hold must name the journaled copy (tier %d at %s), got %+v", UpdateTierLocal, slice4T0.Add(-time.Hour), g.holdRP)
}
}
@@ -0,0 +1,175 @@
package stacks
import (
"context"
"errors"
"fmt"
"testing"
"time"
)
// R-475 — any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0).
// Tier order and the Tier-3 timeout are the backup side's (internal/backup/update_tiers_test.go);
// here is the update job's half: which copy it leans on, when it backs up first, when it refuses.
// Every age here is read against slice4T0, the same clock the job reads (R-457).
func hasGuardCall(g *fakeGuards, name string) bool {
for _, c := range g.callList() {
if c == name {
return true
}
}
return false
}
// failedUpdateHold runs an update whose new version never becomes healthy, so the hold records the
// copy the job chose. That is the observable consequence of the choice: the customer is told to
// restore from exactly that copy.
func failedUpdateHold(t *testing.T, points, afterBackup []UpdateRestorePoint) (*fakeGuards, *Stack) {
t.Helper()
m, _, g, _ := newSlice4Manager(t)
g.points, g.pointsAfterBackup = points, afterBackup
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatalf("start: %v", err)
}
return g, waitUpdateDone(t, m, "nextcloud")
}
func TestR475_G_TheSecondDriveIsChosenWhenPresent(t *testing.T) {
fresh := slice4T0.Add(-time.Hour)
g, st := failedUpdateHold(t, []UpdateRestorePoint{
{Tier: UpdateTierSecondDrive, ProvenAt: fresh},
{Tier: UpdateTierLocal, ProvenAt: fresh.Add(30 * time.Minute)},
{Tier: UpdateTierOffsite, ProvenAt: fresh},
}, nil)
if st.UpdatePhase != UpdatePhaseFailed || g.holdRP.Tier != UpdateTierSecondDrive || !g.holdRP.ProvenAt.Equal(fresh) {
t.Errorf("with a fresh second-drive copy that copy is chosen; hold = %+v phase=%q", g.holdRP, st.UpdatePhase)
}
if hasGuardCall(g, "BackupNow") {
t.Error("a fresh copy needs no backup first")
}
}
func TestR475_H_TheOwnUnitAloneCarriesTheUpdate(t *testing.T) {
fresh := slice4T0.Add(-2 * time.Hour)
g, _ := failedUpdateHold(t, []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: fresh}}, nil)
if g.holdRP.Tier != UpdateTierLocal || !g.holdRP.ProvenAt.Equal(fresh) {
t.Errorf("an app with only its own recovery unit must update against it; hold = %+v", g.holdRP)
}
if hasGuardCall(g, "BackupNow") {
t.Error("a fresh own unit needs no backup first — Tier 2 must not be required anywhere")
}
}
func TestR475_I_TheOffsiteCopyAloneCarriesTheUpdate(t *testing.T) {
fresh := slice4T0.Add(-3 * time.Hour)
g, _ := failedUpdateHold(t, []UpdateRestorePoint{{Tier: UpdateTierOffsite, ProvenAt: fresh}}, nil)
if g.holdRP.Tier != UpdateTierOffsite || !g.holdRP.ProvenAt.Equal(fresh) || hasGuardCall(g, "BackupNow") {
t.Errorf("an app with only a fresh off-site copy must update against it; hold = %+v calls=%v", g.holdRP, g.callList())
}
}
func TestR475_K_NoCopyAnywhereIsBackedUpFirstAndTheUpdateProceeds(t *testing.T) {
m, dir, g, _ := newSlice4Manager(t)
g.points = nil
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-time.Minute)}}
if ref := m.UpdatePreflight("nextcloud"); ref != nil {
t.Fatalf("an app with no copy but a working backup must NOT be refused, got %q (%s)", ref.Message, ref.Reason)
}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("the backup made a Tier-1 copy, so the update completes; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
if !hasGuardCall(g, "BackupNow") {
t.Error("with no copy anywhere the job must back up FIRST")
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("pin = %q", got)
}
}
func TestR475_K_NoCopyAndTheBackupFailsMovesNothing(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.points = nil
g.backupErr = errors.New("nincs elég szabad hely")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want {
t.Errorf("UpdateError = %q, want %q", st.UpdateError, want)
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("nothing may move when the backup-first failed")
}
}
// L — nothing anywhere and no way to back up is refused (TestSlice4_C_… pins the refusal itself). The
// other half pins that the refusal is not wider than that: a copy that EXISTS still carries it.
func TestR475_L_AnExistingCopyStillCarriesAnAppThatCannotBeBackedUp(t *testing.T) {
m, _, g, _ := newSlice4Manager(t)
g.cannotBackUp = true
g.points = []UpdateRestorePoint{{Tier: UpdateTierOffsite, ProvenAt: slice4T0.Add(-time.Hour)}}
if ref := m.UpdatePreflight("nextcloud"); ref != nil {
t.Fatalf("a fresh off-site copy must carry the update even when no backup can be taken now, got %q", ref.Message)
}
g.points = nil
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
t.Fatalf("positive control: with no copy the same app must be refused, got %v", ref)
}
}
// M — a stale copy on ANY tier is stale. Each case would pass under a Tier-2-only age check except the
// first, which is the control proving the fixture's clock and limit are real.
func TestR475_M_TheAgeRuleAppliesToTheChosenTier(t *testing.T) {
stale, fresh, justNow := slice4T0.Add(-30*time.Hour), slice4T0.Add(-time.Hour), slice4T0.Add(-time.Minute)
cases := []struct {
name string
points []UpdateRestorePoint
afterBackup []UpdateRestorePoint
wantBackup bool
wantTier int
}{
{"control: a stale second-drive copy is backed up first", []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: stale}},
[]UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: justNow}}, true, UpdateTierSecondDrive},
{"a stale own unit is backed up first", []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: stale}},
[]UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: justNow}}, true, UpdateTierLocal},
{"a stale off-site copy is backed up first", []UpdateRestorePoint{{Tier: UpdateTierOffsite, ProvenAt: stale}},
[]UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: justNow}, {Tier: UpdateTierOffsite, ProvenAt: stale}}, true, UpdateTierLocal},
{"a stale second drive does not block a fresh own unit", []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: stale}, {Tier: UpdateTierLocal, ProvenAt: fresh}},
nil, false, UpdateTierLocal},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
g, st := failedUpdateHold(t, tc.points, tc.afterBackup)
if got := hasGuardCall(g, "BackupNow"); got != tc.wantBackup {
t.Errorf("backed up first = %v, want %v (calls %v)", got, tc.wantBackup, g.callList())
}
if g.holdRP.Tier != tc.wantTier {
t.Errorf("the chosen copy is tier %d, want %d (hold %+v, err %q)", g.holdRP.Tier, tc.wantTier, g.holdRP, st.UpdateError)
}
if age := slice4T0.Sub(g.holdRP.ProvenAt); age > 24*time.Hour {
t.Errorf("the chosen copy is %s old — past the age limit", age)
}
})
}
}
func TestR475_M_TheLimitIsInclusiveOnOneClock(t *testing.T) {
fresh := freshRestorePoint(slice4T0, 24*time.Hour)
for _, tier := range []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite} {
if !fresh(UpdateRestorePoint{Tier: tier, ProvenAt: slice4T0.Add(-24 * time.Hour)}) {
t.Errorf("tier %d: exactly the limit old is still fresh", tier)
}
if fresh(UpdateRestorePoint{Tier: tier, ProvenAt: slice4T0.Add(-24*time.Hour - time.Second)}) {
t.Errorf("tier %d: a second past the limit is stale", tier)
}
if fresh(UpdateRestorePoint{Tier: tier}) {
t.Errorf("tier %d: a copy with no proven time is never fresh", tier)
}
}
}