Files
felhom-controller/controller/internal/stacks/update_tiers_test.go
T
admin b93c1543da
gates / gates (push) Successful in 14s
controller v0.239.0: any backup tier lets an app update (R-475)
Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1
(own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable
counts as absent with a WARN) and leans on the first FRESH copy; the
backup_max_age rule applies to whichever tier is chosen. No copy anywhere:
back up first. Refused only when nothing exists and no backup can be taken.
RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured
unit proven current. The hold names the tier (második meghajtó / saját
meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text.
A successful off-site restore now lifts an update hold. The backups page
still uses Tier2UnitRestorePoint unchanged.

Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in
felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
2026-09-13 17:16:24 +02:00

176 lines
7.6 KiB
Go

package stacks
import (
"context"
"errors"
"fmt"
"testing"
"time"
)
// R-475 — any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0).
// Tier order and the Tier-3 timeout are the backup side's (internal/backup/update_tiers_test.go);
// here is the update job's half: which copy it leans on, when it backs up first, when it refuses.
// Every age here is read against slice4T0, the same clock the job reads (R-457).
func hasGuardCall(g *fakeGuards, name string) bool {
for _, c := range g.callList() {
if c == name {
return true
}
}
return false
}
// failedUpdateHold runs an update whose new version never becomes healthy, so the hold records the
// copy the job chose. That is the observable consequence of the choice: the customer is told to
// restore from exactly that copy.
func failedUpdateHold(t *testing.T, points, afterBackup []UpdateRestorePoint) (*fakeGuards, *Stack) {
t.Helper()
m, _, g, _ := newSlice4Manager(t)
g.points, g.pointsAfterBackup = points, afterBackup
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatalf("start: %v", err)
}
return g, waitUpdateDone(t, m, "nextcloud")
}
func TestR475_G_TheSecondDriveIsChosenWhenPresent(t *testing.T) {
fresh := slice4T0.Add(-time.Hour)
g, st := failedUpdateHold(t, []UpdateRestorePoint{
{Tier: UpdateTierSecondDrive, ProvenAt: fresh},
{Tier: UpdateTierLocal, ProvenAt: fresh.Add(30 * time.Minute)},
{Tier: UpdateTierOffsite, ProvenAt: fresh},
}, nil)
if st.UpdatePhase != UpdatePhaseFailed || g.holdRP.Tier != UpdateTierSecondDrive || !g.holdRP.ProvenAt.Equal(fresh) {
t.Errorf("with a fresh second-drive copy that copy is chosen; hold = %+v phase=%q", g.holdRP, st.UpdatePhase)
}
if hasGuardCall(g, "BackupNow") {
t.Error("a fresh copy needs no backup first")
}
}
func TestR475_H_TheOwnUnitAloneCarriesTheUpdate(t *testing.T) {
fresh := slice4T0.Add(-2 * time.Hour)
g, _ := failedUpdateHold(t, []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: fresh}}, nil)
if g.holdRP.Tier != UpdateTierLocal || !g.holdRP.ProvenAt.Equal(fresh) {
t.Errorf("an app with only its own recovery unit must update against it; hold = %+v", g.holdRP)
}
if hasGuardCall(g, "BackupNow") {
t.Error("a fresh own unit needs no backup first — Tier 2 must not be required anywhere")
}
}
func TestR475_I_TheOffsiteCopyAloneCarriesTheUpdate(t *testing.T) {
fresh := slice4T0.Add(-3 * time.Hour)
g, _ := failedUpdateHold(t, []UpdateRestorePoint{{Tier: UpdateTierOffsite, ProvenAt: fresh}}, nil)
if g.holdRP.Tier != UpdateTierOffsite || !g.holdRP.ProvenAt.Equal(fresh) || hasGuardCall(g, "BackupNow") {
t.Errorf("an app with only a fresh off-site copy must update against it; hold = %+v calls=%v", g.holdRP, g.callList())
}
}
func TestR475_K_NoCopyAnywhereIsBackedUpFirstAndTheUpdateProceeds(t *testing.T) {
m, dir, g, _ := newSlice4Manager(t)
g.points = nil
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-time.Minute)}}
if ref := m.UpdatePreflight("nextcloud"); ref != nil {
t.Fatalf("an app with no copy but a working backup must NOT be refused, got %q (%s)", ref.Message, ref.Reason)
}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("the backup made a Tier-1 copy, so the update completes; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
if !hasGuardCall(g, "BackupNow") {
t.Error("with no copy anywhere the job must back up FIRST")
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("pin = %q", got)
}
}
func TestR475_K_NoCopyAndTheBackupFailsMovesNothing(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.points = nil
g.backupErr = errors.New("nincs elég szabad hely")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want {
t.Errorf("UpdateError = %q, want %q", st.UpdateError, want)
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("nothing may move when the backup-first failed")
}
}
// L — nothing anywhere and no way to back up is refused (TestSlice4_C_… pins the refusal itself). The
// other half pins that the refusal is not wider than that: a copy that EXISTS still carries it.
func TestR475_L_AnExistingCopyStillCarriesAnAppThatCannotBeBackedUp(t *testing.T) {
m, _, g, _ := newSlice4Manager(t)
g.cannotBackUp = true
g.points = []UpdateRestorePoint{{Tier: UpdateTierOffsite, ProvenAt: slice4T0.Add(-time.Hour)}}
if ref := m.UpdatePreflight("nextcloud"); ref != nil {
t.Fatalf("a fresh off-site copy must carry the update even when no backup can be taken now, got %q", ref.Message)
}
g.points = nil
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
t.Fatalf("positive control: with no copy the same app must be refused, got %v", ref)
}
}
// M — a stale copy on ANY tier is stale. Each case would pass under a Tier-2-only age check except the
// first, which is the control proving the fixture's clock and limit are real.
func TestR475_M_TheAgeRuleAppliesToTheChosenTier(t *testing.T) {
stale, fresh, justNow := slice4T0.Add(-30*time.Hour), slice4T0.Add(-time.Hour), slice4T0.Add(-time.Minute)
cases := []struct {
name string
points []UpdateRestorePoint
afterBackup []UpdateRestorePoint
wantBackup bool
wantTier int
}{
{"control: a stale second-drive copy is backed up first", []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: stale}},
[]UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: justNow}}, true, UpdateTierSecondDrive},
{"a stale own unit is backed up first", []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: stale}},
[]UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: justNow}}, true, UpdateTierLocal},
{"a stale off-site copy is backed up first", []UpdateRestorePoint{{Tier: UpdateTierOffsite, ProvenAt: stale}},
[]UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: justNow}, {Tier: UpdateTierOffsite, ProvenAt: stale}}, true, UpdateTierLocal},
{"a stale second drive does not block a fresh own unit", []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: stale}, {Tier: UpdateTierLocal, ProvenAt: fresh}},
nil, false, UpdateTierLocal},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
g, st := failedUpdateHold(t, tc.points, tc.afterBackup)
if got := hasGuardCall(g, "BackupNow"); got != tc.wantBackup {
t.Errorf("backed up first = %v, want %v (calls %v)", got, tc.wantBackup, g.callList())
}
if g.holdRP.Tier != tc.wantTier {
t.Errorf("the chosen copy is tier %d, want %d (hold %+v, err %q)", g.holdRP.Tier, tc.wantTier, g.holdRP, st.UpdateError)
}
if age := slice4T0.Sub(g.holdRP.ProvenAt); age > 24*time.Hour {
t.Errorf("the chosen copy is %s old — past the age limit", age)
}
})
}
}
func TestR475_M_TheLimitIsInclusiveOnOneClock(t *testing.T) {
fresh := freshRestorePoint(slice4T0, 24*time.Hour)
for _, tier := range []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite} {
if !fresh(UpdateRestorePoint{Tier: tier, ProvenAt: slice4T0.Add(-24 * time.Hour)}) {
t.Errorf("tier %d: exactly the limit old is still fresh", tier)
}
if fresh(UpdateRestorePoint{Tier: tier, ProvenAt: slice4T0.Add(-24*time.Hour - time.Second)}) {
t.Errorf("tier %d: a second past the limit is stale", tier)
}
if fresh(UpdateRestorePoint{Tier: tier}) {
t.Errorf("tier %d: a copy with no proven time is never fresh", tier)
}
}
}