controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s
gates / gates (push) Successful in 14s
Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1 (own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable counts as absent with a WARN) and leans on the first FRESH copy; the backup_max_age rule applies to whichever tier is chosen. No copy anywhere: back up first. Refused only when nothing exists and no backup can be taken. RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured unit proven current. The hold names the tier (második meghajtó / saját meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text. A successful off-site restore now lifts an update hold. The backups page still uses Tier2UnitRestorePoint unchanged. Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
@@ -25,13 +25,15 @@ type fakeGuards struct {
|
||||
held bool
|
||||
holdWhy string
|
||||
busy bool
|
||||
rp UpdateRestorePoint
|
||||
rpErr error
|
||||
rpAfterBackup *UpdateRestorePoint
|
||||
// points are the copies the backup side holds, in tier order (R-475); pointsAfterBackup replaces
|
||||
// them when BackupNow succeeds (nil = the backup changed nothing).
|
||||
points []UpdateRestorePoint
|
||||
pointsAfterBackup []UpdateRestorePoint
|
||||
cannotBackUp bool
|
||||
backupErr error
|
||||
dumpErr error
|
||||
holdErr error
|
||||
holdProvenAt time.Time
|
||||
holdRP UpdateRestorePoint
|
||||
pinAtDump string
|
||||
stackDir string
|
||||
}
|
||||
@@ -48,18 +50,29 @@ func (f *fakeGuards) HoldFor(string) (bool, string) {
|
||||
return f.held, f.holdWhy
|
||||
}
|
||||
func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" }
|
||||
func (f *fakeGuards) RestorePoint(string) (UpdateRestorePoint, error) {
|
||||
f.note("RestorePoint")
|
||||
func (f *fakeGuards) RestorePoints(_ context.Context, _ string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint) {
|
||||
f.note("RestorePoints")
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
return f.rp, f.rpErr
|
||||
var seen []UpdateRestorePoint
|
||||
for _, p := range f.points {
|
||||
seen = append(seen, p)
|
||||
if accept == nil || accept(p) {
|
||||
return p, true, seen
|
||||
}
|
||||
}
|
||||
return UpdateRestorePoint{}, false, seen
|
||||
}
|
||||
func (f *fakeGuards) CanBackUp(string) (bool, string) {
|
||||
f.note("CanBackUp")
|
||||
return !f.cannotBackUp, "fake: the drive is gone"
|
||||
}
|
||||
func (f *fakeGuards) BackupNow(context.Context, string) error {
|
||||
f.note("BackupNow")
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.backupErr == nil && f.rpAfterBackup != nil {
|
||||
f.rp = *f.rpAfterBackup
|
||||
if f.backupErr == nil && f.pointsAfterBackup != nil {
|
||||
f.points = f.pointsAfterBackup
|
||||
}
|
||||
return f.backupErr
|
||||
}
|
||||
@@ -72,14 +85,14 @@ func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
|
||||
}
|
||||
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
|
||||
}
|
||||
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, provenAt time.Time) error {
|
||||
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint) error {
|
||||
f.note("HoldAfterFailedUpdate")
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.holdErr != nil {
|
||||
return f.holdErr
|
||||
}
|
||||
f.held, f.holdWhy, f.holdProvenAt = true, "HELD-SENTENCE", provenAt
|
||||
f.held, f.holdWhy, f.holdRP = true, "HELD-SENTENCE", rp
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -111,7 +124,7 @@ func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec)
|
||||
m, dir := newPinManager(t, pinTplOld, pinTplNew,
|
||||
"deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n")
|
||||
mustWrite(t, AppliedComposePath(dir), pinTplOld)
|
||||
g := &fakeGuards{rp: UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Hour)}, stackDir: dir}
|
||||
g := &fakeGuards{points: []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: slice4T0.Add(-1 * time.Hour)}}, stackDir: dir}
|
||||
c := &composeRec{fail: map[string]error{}}
|
||||
m.updateGuards = g
|
||||
m.updateComposeFn = c.fn
|
||||
@@ -212,9 +225,8 @@ func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
|
||||
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
|
||||
t.Errorf("compose calls = %q, want %q", got, want)
|
||||
}
|
||||
// RestorePoint twice by design: once in the preflight (the refusal), once inside the job (the
|
||||
// precondition must still hold when the job actually starts).
|
||||
if got := strings.Join(g.callList(), ","); got != "RestorePoint,RestorePoint,SafetyDump" {
|
||||
// R-475: the preflight asks only whether a backup could be taken; the job reads the copies once.
|
||||
if got := strings.Join(g.callList(), ","); got != "CanBackUp,RestorePoints,SafetyDump" {
|
||||
t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got)
|
||||
}
|
||||
}
|
||||
@@ -223,8 +235,8 @@ func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
|
||||
|
||||
func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
|
||||
m, dir, g, _ := newSlice4Manager(t)
|
||||
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
|
||||
g.rpAfterBackup = &UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Minute)}
|
||||
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
|
||||
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: slice4T0.Add(-1 * time.Minute)}}
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -233,7 +245,7 @@ func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
|
||||
t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
calls := strings.Join(g.callList(), ",")
|
||||
if !strings.HasPrefix(calls, "RestorePoint,RestorePoint,BackupNow,RestorePoint,SafetyDump") {
|
||||
if !strings.HasPrefix(calls, "CanBackUp,RestorePoints,BackupNow,RestorePoints,SafetyDump") {
|
||||
t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls)
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||||
@@ -243,7 +255,7 @@ func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
|
||||
|
||||
func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour)
|
||||
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour)
|
||||
g.backupErr = errors.New("disk full")
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -265,7 +277,7 @@ func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
|
||||
|
||||
func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: rpAfterBackup nil
|
||||
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: pointsAfterBackup nil
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -278,21 +290,18 @@ func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// ── C: no backup exists that could restore this app ───────────────────────────────────────────────
|
||||
// ── C / R-475 L: no copy on any tier, and no way to make one ─────────────────────────────────────
|
||||
|
||||
// COMPANION RED-PROOF 2 (REPORT.md): make the precondition in UpdatePreflight proceed when
|
||||
// !rp.Restorable. This test then fails with the update started.
|
||||
func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
|
||||
// Until v0.239.0 this refused any app without a restorable Tier-2 unit. R-475: every tier counts and
|
||||
// an app with nothing is backed up first, so the refusal is now only "nothing anywhere AND no backup
|
||||
// can be taken". The K half (nothing, but a backup CAN be taken) is TestR475_K in update_tiers_test.go.
|
||||
func TestSlice4_C_NoCopyAndNoWayToBackUpRefusesBeforeAnythingMoves(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
// A PROVEN, FRESH copy whose unit cannot be opened — the realistic half-copied mirror. Proven and
|
||||
// fresh on purpose: a fixture that is also unproven would be refused by the proven check alone,
|
||||
// and a red-proof that drops the restorable check would then pass inertly (observed on the first
|
||||
// run of red-proof 2, 2026-09-13).
|
||||
g.rp = UpdateRestorePoint{Restorable: false, Proven: true, ProvenAt: slice4T0.Add(-time.Hour)}
|
||||
g.points, g.cannotBackUp = nil, true
|
||||
err := m.StartGuardedUpdate("nextcloud")
|
||||
var ref *UpdateRefusal
|
||||
if !errors.As(err, &ref) || ref.Reason != "no_backup" {
|
||||
t.Fatalf("an app with no restorable copy must be REFUSED (no_backup), got %v", err)
|
||||
t.Fatalf("no copy anywhere and no way to back up must be REFUSED (no_backup), got %v", err)
|
||||
}
|
||||
if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want {
|
||||
t.Errorf("message = %q", ref.Message)
|
||||
@@ -304,10 +313,10 @@ func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
|
||||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
|
||||
t.Error("a refused update must move nothing")
|
||||
}
|
||||
// A copy that exists but was never PROVEN is not a copy (R-101).
|
||||
g.rp = UpdateRestorePoint{Restorable: true, Proven: false}
|
||||
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
|
||||
t.Errorf("an unproven copy must refuse too, got %v", ref)
|
||||
for _, call := range g.callList() {
|
||||
if call == "BackupNow" {
|
||||
t.Error("a refused update must not try to back up")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -412,8 +421,8 @@ func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
|
||||
if !held {
|
||||
t.Fatal("an app that did not come up must be HELD")
|
||||
}
|
||||
if !g.holdProvenAt.Equal(g.rp.ProvenAt) {
|
||||
t.Errorf("the hold must name the PROVEN copy date %s, got %s", g.rp.ProvenAt, g.holdProvenAt)
|
||||
if !g.holdRP.ProvenAt.Equal(g.points[0].ProvenAt) || g.holdRP.Tier != UpdateTierSecondDrive {
|
||||
t.Errorf("the hold must name the PROVEN copy it leans on (%+v), got %+v", g.points[0], g.holdRP)
|
||||
}
|
||||
if st.UpdateError != "HELD-SENTENCE" {
|
||||
t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError)
|
||||
@@ -466,6 +475,7 @@ func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry {
|
||||
StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"},
|
||||
PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile),
|
||||
ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339),
|
||||
ProvenTier: UpdateTierLocal,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -530,8 +540,8 @@ func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) {
|
||||
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | down" {
|
||||
t.Errorf("resumption re-runs `up` then stops the failed app; compose calls = %q", got)
|
||||
}
|
||||
if !g.holdProvenAt.Equal(slice4T0.Add(-time.Hour)) {
|
||||
t.Errorf("the resumed hold must name the journaled proven copy date, got %s", g.holdProvenAt)
|
||||
if !g.holdRP.ProvenAt.Equal(slice4T0.Add(-time.Hour)) || g.holdRP.Tier != UpdateTierLocal {
|
||||
t.Errorf("the resumed hold must name the journaled copy (tier %d at %s), got %+v", UpdateTierLocal, slice4T0.Add(-time.Hour), g.holdRP)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user