Files
felhom-controller/controller/internal/stacks/update_test.go
T
admin 0d402f711d
gates / gates (push) Successful in 13s
v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.

The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.

Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.

Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.

Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.

Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-13 11:41:31 +02:00

570 lines
23 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package stacks
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"sync"
"testing"
"time"
)
// Update arc slice 4 — the guarded update. Scenarios A–G of the task, through the real job
// (runGuardedUpdate) with the four process boundaries injected: compose, the health wait, the backup
// side (UpdateGuards) and the clock. Every assertion reads the EFFECT back — the pin in app.yaml, the
// bytes of the live compose file, the journal on disk, Updating/UpdatePhase/UpdateError — never "no error".
var slice4T0 = time.Date(2026, 9, 13, 10, 0, 0, 0, time.UTC)
type fakeGuards struct {
mu sync.Mutex
calls []string
held bool
holdWhy string
busy bool
rp UpdateRestorePoint
rpErr error
rpAfterBackup *UpdateRestorePoint
backupErr error
dumpErr error
holdErr error
holdProvenAt time.Time
pinAtDump string
stackDir string
}
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
func (f *fakeGuards) callList() []string {
f.mu.Lock()
defer f.mu.Unlock()
return append([]string(nil), f.calls...)
}
func (f *fakeGuards) HoldFor(string) (bool, string) {
f.mu.Lock()
defer f.mu.Unlock()
return f.held, f.holdWhy
}
func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" }
func (f *fakeGuards) RestorePoint(string) (UpdateRestorePoint, error) {
f.note("RestorePoint")
f.mu.Lock()
defer f.mu.Unlock()
return f.rp, f.rpErr
}
func (f *fakeGuards) BackupNow(context.Context, string) error {
f.note("BackupNow")
f.mu.Lock()
defer f.mu.Unlock()
if f.backupErr == nil && f.rpAfterBackup != nil {
f.rp = *f.rpAfterBackup
}
return f.backupErr
}
func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
f.note("SafetyDump")
if cfg := LoadAppConfig(f.stackDir); cfg != nil {
f.mu.Lock()
f.pinAtDump = cfg.PinnedImages["web"]
f.mu.Unlock()
}
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
}
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, provenAt time.Time) error {
f.note("HoldAfterFailedUpdate")
f.mu.Lock()
defer f.mu.Unlock()
if f.holdErr != nil {
return f.holdErr
}
f.held, f.holdWhy, f.holdProvenAt = true, "HELD-SENTENCE", provenAt
return nil
}
type composeRec struct {
mu sync.Mutex
calls []string
fail map[string]error // first arg → error
}
func (c *composeRec) fn(_ string, _ []string, args ...string) (string, error) {
c.mu.Lock()
defer c.mu.Unlock()
c.calls = append(c.calls, strings.Join(args, " "))
if err := c.fail[args[0]]; err != nil {
return "", err
}
return "", nil
}
func (c *composeRec) list() []string {
c.mu.Lock()
defer c.mu.Unlock()
return append([]string(nil), c.calls...)
}
// newSlice4Manager: a pinned nextcloud on the OLD version, the catalog offering the NEW one, fresh
// proven copy, and every boundary faked. Returns the manager, its stack dir, the guards and compose.
func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec) {
t.Helper()
m, dir := newPinManager(t, pinTplOld, pinTplNew,
"deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n")
mustWrite(t, AppliedComposePath(dir), pinTplOld)
g := &fakeGuards{rp: UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Hour)}, stackDir: dir}
c := &composeRec{fail: map[string]error{}}
m.updateGuards = g
m.updateComposeFn = c.fn
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return true, "fake healthy" }
m.updateMemoryFn = func(int, int, int, int) (string, string) { return "", "" }
m.updateDiskFreeFn = func() (float64, bool) { return 50, true }
m.updateNowFn = func() time.Time { return slice4T0 } // R-457: the SAME clock the age check reads
m.execFn = func(string, ...string) (string, error) { return "", nil }
return m, dir, g, c
}
func waitUpdateDone(t *testing.T, m *Manager, name string) *Stack {
t.Helper()
deadline := time.Now().Add(5 * time.Second)
for time.Now().Before(deadline) {
if st, ok := m.GetStack(name); ok && !st.Updating {
return st
}
time.Sleep(5 * time.Millisecond)
}
t.Fatal("the update never finished")
return nil
}
func pinOf(t *testing.T, dir string) string { return readPin(t, dir).PinnedImages["web"] }
func fileBody(t *testing.T, p string) string {
t.Helper()
b, err := os.ReadFile(p)
if err != nil {
t.Fatal(err)
}
return string(b)
}
func journalExists(m *Manager) bool {
_, err := os.Stat(m.updateJournalPath())
return err == nil
}
// ── A: the happy path, and the truth ─────────────────────────────────────────────────────────────
// TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth. The health wait BLOCKS until the test releases it;
// while it blocks, the app must read Updating=true / phase=verifying / no error — and the safety dump
// must have run while the pin still named the OLD version.
//
// COMPANION RED-PROOF 1 (REPORT.md): delete the updateHealth call from verifyAndConclude so success is
// declared on the compose exit code. This test then fails at "Updating went false before health was
// known" — which is R-443 exactly.
func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
release := make(chan struct{})
healthCalled := make(chan struct{}, 1)
m.updateHealthFn = func(ctx context.Context, name string, timeout time.Duration) (bool, string) {
healthCalled <- struct{}{}
<-release
return true, "fake healthy"
}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatalf("a fully-qualified update must start: %v", err)
}
select {
case <-healthCalled:
case <-time.After(5 * time.Second):
st, _ := m.GetStack("nextcloud")
t.Fatalf("the health wait was never reached; state: updating=%v phase=%s err=%q", st.Updating, st.UpdatePhase, st.UpdateError)
}
st, _ := m.GetStack("nextcloud")
if !st.Updating {
t.Fatal("Updating went false before health was known — success reported on the compose exit code (R-443)")
}
if st.UpdatePhase != UpdatePhaseVerifying || st.UpdatePhaseLabel != "Működés ellenőrzése…" {
t.Errorf("while waiting for health the phase must be verifying, got %q / %q", st.UpdatePhase, st.UpdatePhaseLabel)
}
if st.UpdateError != "" {
t.Errorf("no error may be shown while verifying, got %q", st.UpdateError)
}
if !journalExists(m) {
t.Error("the journal must exist while the update is in flight (Scenario G depends on it)")
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("by verifying, the pin must have advanced; got %q", got)
}
close(release)
st = waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseDone || st.UpdateError != "" || st.UpdatePhaseLabel != "Frissítve" {
t.Fatalf("after health the update is done: phase=%q label=%q err=%q", st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError)
}
if journalExists(m) {
t.Error("a completed update must clear its journal entry")
}
if _, err := os.Stat(filepath.Join(dir, preUpdateComposeFile)); err == nil {
t.Error("the pre-update copy must be removed after success")
}
if g.pinAtDump != "nextcloud:31.0.14-apache" {
t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump)
}
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
t.Errorf("compose calls = %q, want %q", got, want)
}
// RestorePoint twice by design: once in the preflight (the refusal), once inside the job (the
// precondition must still hold when the job actually starts).
if got := strings.Join(g.callList(), ","); got != "RestorePoint,RestorePoint,SafetyDump" {
t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got)
}
}
// ── B: the proven copy is too old ─────────────────────────────────────────────────────────────────
func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
m, dir, g, _ := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
g.rpAfterBackup = &UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Minute)}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
calls := strings.Join(g.callList(), ",")
if !strings.HasPrefix(calls, "RestorePoint,RestorePoint,BackupNow,RestorePoint,SafetyDump") {
t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("pin = %q", got)
}
}
func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour)
g.backupErr = errors.New("disk full")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want {
t.Errorf("UpdateError = %q, want the backup's own error in the sentence %q", st.UpdateError, want)
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("a failed backup must move nothing; pin = %q", got)
}
if len(c.list()) != 0 {
t.Errorf("a failed backup must reach no compose call; got %v", c.list())
}
if journalExists(m) {
t.Error("the journal must be cleared on a refusal")
}
}
func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: rpAfterBackup nil
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdateError != MsgUpdateBackupNoUnit {
t.Errorf("UpdateError = %q", st.UpdateError)
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("nothing may move when the backup did not produce a fresh restorable copy")
}
}
// ── C: no backup exists that could restore this app ───────────────────────────────────────────────
// COMPANION RED-PROOF 2 (REPORT.md): make the precondition in UpdatePreflight proceed when
// !rp.Restorable. This test then fails with the update started.
func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
// A PROVEN, FRESH copy whose unit cannot be opened — the realistic half-copied mirror. Proven and
// fresh on purpose: a fixture that is also unproven would be refused by the proven check alone,
// and a red-proof that drops the restorable check would then pass inertly (observed on the first
// run of red-proof 2, 2026-09-13).
g.rp = UpdateRestorePoint{Restorable: false, Proven: true, ProvenAt: slice4T0.Add(-time.Hour)}
err := m.StartGuardedUpdate("nextcloud")
var ref *UpdateRefusal
if !errors.As(err, &ref) || ref.Reason != "no_backup" {
t.Fatalf("an app with no restorable copy must be REFUSED (no_backup), got %v", err)
}
if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want {
t.Errorf("message = %q", ref.Message)
}
time.Sleep(50 * time.Millisecond)
if st, _ := m.GetStack("nextcloud"); st.Updating {
t.Error("a refused update must not set Updating")
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("a refused update must move nothing")
}
// A copy that exists but was never PROVEN is not a copy (R-101).
g.rp = UpdateRestorePoint{Restorable: true, Proven: false}
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
t.Errorf("an unproven copy must refuse too, got %v", ref)
}
}
// ── D: the cheap refusals, each one ─────────────────────────────────────────────────────────────────
func TestSlice4_D_CheapRefusals(t *testing.T) {
cases := []struct {
name string
setup func(m *Manager, g *fakeGuards, dir string)
reason string
msg string
}{
{"held", func(m *Manager, g *fakeGuards, _ string) { g.held, g.holdWhy = true, "THE HOLD TEXT" }, "held", "THE HOLD TEXT"},
{"busy", func(m *Manager, g *fakeGuards, _ string) { g.busy = true }, "busy", MsgUpdateBusy},
{"already updating", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Updating = true }, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, "nextcloud")},
{"deploying", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Deploying = true }, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, "nextcloud")},
{"memory", func(m *Manager, _ *fakeGuards, _ string) {
catDir := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates", "nextcloud")
if err := os.WriteFile(filepath.Join(catDir, ".felhom.yml"), []byte("resources:\n mem_request: 900M\n"), 0o644); err != nil {
panic(err)
}
m.updateMemoryFn = func(newReq, _, _, _ int) (string, string) {
return fmt.Sprintf("Nincs elég memória (%d MB)", newReq), ""
}
}, "memory", "Nincs elég memória (900 MB)"},
{"disk", func(m *Manager, _ *fakeGuards, _ string) {
m.updateDiskFreeFn = func() (float64, bool) { return 1.0, true }
}, "disk", fmt.Sprintf(MsgUpdateDiskFmt, 1.0, updateDiskFloorGiB)},
{"guards unwired", func(m *Manager, _ *fakeGuards, _ string) { m.updateGuards = nil }, "guards_unwired", MsgUpdateNoGuards},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
tc.setup(m, g, dir)
ref := m.UpdatePreflight("nextcloud")
if ref == nil || ref.Reason != tc.reason {
t.Fatalf("want refusal %q, got %+v", tc.reason, ref)
}
if ref.Message != tc.msg {
t.Errorf("message = %q, want %q", ref.Message, tc.msg)
}
if err := m.StartGuardedUpdate("nextcloud"); err == nil {
t.Fatal("StartGuardedUpdate must refuse the same")
}
time.Sleep(20 * time.Millisecond)
if len(c.list()) != 0 || pinOf(t, dir) != "nextcloud:31.0.14-apache" {
t.Errorf("a cheap refusal reached the act: compose=%v pin=%s", c.list(), pinOf(t, dir))
}
})
}
}
// ── E: the pull fails ────────────────────────────────────────────────────────────────────────────
func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
c.fail["pull"] = errors.New("exit code 1\nstderr: manifest unknown")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdateError != MsgUpdatePullFailed {
t.Errorf("UpdateError = %q, want the Hungarian sentence and never raw stderr", st.UpdateError)
}
if strings.Contains(st.UpdateError, "manifest unknown") {
t.Error("raw Docker stderr leaked into the customer sentence")
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the pin must be PUT BACK after a failed pull, got %q", got)
}
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
t.Errorf("the live file must be re-rendered to the old version:\n%s", got)
}
if got := fileBody(t, AppliedComposePath(dir)); got != pinTplOld {
t.Errorf("the stored definition must be the old one again:\n%s", got)
}
if got := strings.Join(c.list(), " | "); got != "pull" {
t.Errorf("after a failed pull nothing else runs; compose calls = %q", got)
}
for _, call := range g.callList() {
if call == "HoldAfterFailedUpdate" {
t.Error("a failed PULL ran nothing and must not hold the app")
}
}
}
// ── F: the new version does not come up ─────────────────────────────────────────────────────────
// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test
// then fails: no hold, and the customer is not told the route back.
func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseFailed {
t.Errorf("phase = %q", st.UpdatePhase)
}
held, _ := g.HoldFor("nextcloud")
if !held {
t.Fatal("an app that did not come up must be HELD")
}
if !g.holdProvenAt.Equal(g.rp.ProvenAt) {
t.Errorf("the hold must name the PROVEN copy date %s, got %s", g.rp.ProvenAt, g.holdProvenAt)
}
if st.UpdateError != "HELD-SENTENCE" {
t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError)
}
if st.HoldReason != "HELD-SENTENCE" {
t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("the pin must STAY on the new version (its migration may have run), got %q", got)
}
if got := strings.Join(c.list(), " | "); got != "pull | up -d --remove-orphans | down" {
t.Errorf("the failed app must be stopped; compose calls = %q", got)
}
if journalExists(m) {
t.Error("the journal is cleared once the hold (the durable record) is written")
}
}
func TestSlice4_F_UnsavedHoldSaysSo(t *testing.T) {
m, _, g, _ := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
g.holdErr = errors.New("settings.json read-only")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdateError != MsgUpdateHoldUnsaved {
t.Errorf("an unrecorded hold must be disclosed, got %q", st.UpdateError)
}
}
// ── G: the controller restarts mid-update ────────────────────────────────────────────────────────
func writeTestJournal(t *testing.T, m *Manager, name string, e updateJournalEntry) {
t.Helper()
if err := m.writeUpdateJournal(updateJournal{Updates: map[string]updateJournalEntry{name: e}}); err != nil {
t.Fatal(err)
}
}
// simulateAdvanced puts the stack in the state a crash AFTER the pin moved would leave: pin, live file
// and stored definition all new, the pre-update copy on disk.
func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry {
t.Helper()
mustWrite(t, filepath.Join(dir, preUpdateComposeFile), pinTplOld)
mustWrite(t, filepath.Join(dir, preUpdateAppliedFile), pinTplOld)
if err := m.advancePinToCatalog("nextcloud", dir); err != nil {
t.Fatal(err)
}
return updateJournalEntry{
StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"},
PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile),
ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339),
}
}
func TestSlice4_G_InterruptedBeforeUpIsPutBack(t *testing.T) {
m, dir, _, c := newSlice4Manager(t)
e := simulateAdvanced(t, m, dir)
e.Phase = UpdatePhasePulling
writeTestJournal(t, m, "nextcloud", e)
if resumed := m.RecoverUpdates(); len(resumed) != 0 {
t.Fatalf("an update interrupted before `up` is not resumed, got %v", resumed)
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the pin must be put back, got %q", got)
}
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
t.Errorf("the live file must be the old definition again:\n%s", got)
}
st, _ := m.GetStack("nextcloud")
if st.Updating || st.UpdateError != MsgUpdateInterrupted {
t.Errorf("updating=%v err=%q", st.Updating, st.UpdateError)
}
if journalExists(m) || len(c.list()) != 0 {
t.Error("recovery of a pre-up interruption runs nothing and clears the journal")
}
}
func TestSlice4_G_InterruptedBeforeThePinIsDroppedUntouched(t *testing.T) {
m, dir, _, _ := newSlice4Manager(t)
writeTestJournal(t, m, "nextcloud", updateJournalEntry{Phase: UpdatePhaseSafetyDump, StartedAt: slice4T0})
m.RecoverUpdates()
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || journalExists(m) {
t.Error("an update interrupted before the pin moved must leave the pin and clear the journal")
}
}
func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
e := simulateAdvanced(t, m, dir)
e.Phase = UpdatePhaseVerifying
writeTestJournal(t, m, "nextcloud", e)
guards := m.updateGuards
m.updateGuards = nil // at RecoverUpdates time the backup side is NOT wired yet (main.go order)
resumed := m.RecoverUpdates()
if len(resumed) != 1 || !m.IsUpdating("nextcloud") || !m.UpdatingStacks()["nextcloud"] {
t.Fatalf("an update interrupted after `up` must be marked Updating for the boot sweep; resumed=%v", resumed)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("after `up` the pin is NOT put back — something may have run; got %q", got)
}
m.updateGuards = guards
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "still broken" }
if n := m.ResumeInterruptedUpdates(context.Background()); n != 1 {
t.Fatalf("resumed %d, want 1", n)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); !held || st.UpdatePhase != UpdatePhaseFailed {
t.Errorf("a resumed update that is still unhealthy must end HELD; held=%v phase=%q", held, st.UpdatePhase)
}
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | down" {
t.Errorf("resumption re-runs `up` then stops the failed app; compose calls = %q", got)
}
if !g.holdProvenAt.Equal(slice4T0.Add(-time.Hour)) {
t.Errorf("the resumed hold must name the journaled proven copy date, got %s", g.holdProvenAt)
}
}
// ── the page reads the hold from the ONE store ──────────────────────────────────────────────────
func TestSlice4_GetStacksCarriesTheHoldText(t *testing.T) {
m, _, g, _ := newSlice4Manager(t)
g.held, g.holdWhy = true, "HOLD"
for _, st := range m.GetStacks() {
if st.Name == "nextcloud" && st.HoldReason != "HOLD" {
t.Errorf("GetStacks HoldReason = %q", st.HoldReason)
}
}
g.held = false
if st, _ := m.GetStack("nextcloud"); st.HoldReason != "" {
t.Errorf("a lifted hold must disappear on the next read, got %q", st.HoldReason)
}
}
func TestSlice4_PhaseLabelsAreTheSpecifiedCopy(t *testing.T) {
want := map[string]string{
UpdatePhaseChecking: "Ellenőrzés…",
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
UpdatePhasePulling: "Új verzió letöltése…",
UpdatePhaseStarting: "Indítás az új verzióval…",
UpdatePhaseVerifying: "Működés ellenőrzése…",
UpdatePhaseDone: "Frissítve",
}
for p, l := range want {
if got := UpdatePhaseLabel(p); got != l {
t.Errorf("label(%s) = %q, want %q", p, got, l)
}
}
}