b6810f14ff
gates / gates (push) Successful in 23s
The unit's data files are stamped with the versions that wrote them; the capture keeps the definition the data belongs to; a restore never starts data under another version's definition (unit restores refuse a mismatch; the off-site restore writes the snapshot's definition); every tier's time is its data's; the conversion-copy release needs a dump on the new engine. File-browser sync single-flight + no empty kept folder (R-695); the kept view joins the folder's owning group, language switch resyncs (R-691); a restore-generated login is not shown as the password (R-694). Red-proofs in felhom.eu/documentation/audits/version-travel-2026-09-26/. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
602 lines
25 KiB
Go
602 lines
25 KiB
Go
package stacks
|
||
|
||
import (
|
||
"context"
|
||
"errors"
|
||
"fmt"
|
||
"os"
|
||
"path/filepath"
|
||
"strings"
|
||
"sync"
|
||
"testing"
|
||
"time"
|
||
)
|
||
|
||
// Update arc slice 4 — the guarded update. Scenarios A–G of the task, through the real job
|
||
// (runGuardedUpdate) with the four process boundaries injected: compose, the health wait, the backup
|
||
// side (UpdateGuards) and the clock. Every assertion reads the EFFECT back — the pin in app.yaml, the
|
||
// bytes of the live compose file, the journal on disk, Updating/UpdatePhase/UpdateError — never "no error".
|
||
|
||
var slice4T0 = time.Date(2026, 9, 13, 10, 0, 0, 0, time.UTC)
|
||
|
||
type fakeGuards struct {
|
||
mu sync.Mutex
|
||
calls []string
|
||
held bool
|
||
holdWhy string
|
||
busy bool
|
||
// points are the copies the backup side holds, in tier order (R-475); pointsAfterBackup replaces
|
||
// them when BackupNow succeeds (nil = the backup changed nothing).
|
||
points []UpdateRestorePoint
|
||
pointsAfterBackup []UpdateRestorePoint
|
||
cannotBackUp bool
|
||
backupErr error
|
||
dumpErr error
|
||
holdErr error
|
||
holdRP UpdateRestorePoint
|
||
pinAtDump string
|
||
stackDir string
|
||
undoState string // what the last hold said a failed undo left (v0.263.0)
|
||
// stamps are the app's own unit's recorded database dumps (v0.275.0, DumpStampSource).
|
||
stamps []DataDumpStamp
|
||
}
|
||
|
||
func (f *fakeGuards) DumpStamps(string) []DataDumpStamp {
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
return append([]DataDumpStamp(nil), f.stamps...)
|
||
}
|
||
|
||
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
|
||
func (f *fakeGuards) callList() []string {
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
return append([]string(nil), f.calls...)
|
||
}
|
||
func (f *fakeGuards) HoldFor(string) (bool, string) {
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
return f.held, f.holdWhy
|
||
}
|
||
func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" }
|
||
func (f *fakeGuards) RestorePoints(_ context.Context, _ string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint) {
|
||
f.note("RestorePoints")
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
var seen []UpdateRestorePoint
|
||
for _, p := range f.points {
|
||
seen = append(seen, p)
|
||
if accept == nil || accept(p) {
|
||
return p, true, seen
|
||
}
|
||
}
|
||
return UpdateRestorePoint{}, false, seen
|
||
}
|
||
func (f *fakeGuards) CanBackUp(string) (bool, string) {
|
||
f.note("CanBackUp")
|
||
return !f.cannotBackUp, "fake: the drive is gone"
|
||
}
|
||
func (f *fakeGuards) BackupNow(context.Context, string) error {
|
||
f.note("BackupNow")
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
if f.backupErr == nil && f.pointsAfterBackup != nil {
|
||
f.points = f.pointsAfterBackup
|
||
}
|
||
return f.backupErr
|
||
}
|
||
func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
|
||
f.note("SafetyDump")
|
||
if cfg := LoadAppConfig(f.stackDir); cfg != nil {
|
||
f.mu.Lock()
|
||
f.pinAtDump = cfg.PinnedImages["web"]
|
||
f.mu.Unlock()
|
||
}
|
||
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
|
||
}
|
||
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint, undoState string) error {
|
||
f.note("HoldAfterFailedUpdate")
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
if f.holdErr != nil {
|
||
return f.holdErr
|
||
}
|
||
f.held, f.holdWhy, f.holdRP, f.undoState = true, "HELD-SENTENCE", rp, undoState
|
||
return nil
|
||
}
|
||
|
||
type composeRec struct {
|
||
mu sync.Mutex
|
||
calls []string
|
||
fail map[string]error // first arg → error
|
||
}
|
||
|
||
func (c *composeRec) fn(_ string, _ []string, args ...string) (string, error) {
|
||
c.mu.Lock()
|
||
defer c.mu.Unlock()
|
||
c.calls = append(c.calls, strings.Join(args, " "))
|
||
if err := c.fail[args[0]]; err != nil {
|
||
return "", err
|
||
}
|
||
return "", nil
|
||
}
|
||
func (c *composeRec) list() []string {
|
||
c.mu.Lock()
|
||
defer c.mu.Unlock()
|
||
return append([]string(nil), c.calls...)
|
||
}
|
||
|
||
// newSlice4Manager: a pinned nextcloud on the OLD version, the catalog offering the NEW one, fresh
|
||
// proven copy, and every boundary faked. Returns the manager, its stack dir, the guards and compose.
|
||
func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec) {
|
||
t.Helper()
|
||
m, dir := newPinManager(t, pinTplOld, pinTplNew,
|
||
"deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n")
|
||
mustWrite(t, AppliedComposePath(dir), pinTplOld)
|
||
g := &fakeGuards{points: []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: slice4T0.Add(-1 * time.Hour)}}, stackDir: dir}
|
||
c := &composeRec{fail: map[string]error{}}
|
||
m.updateGuards = g
|
||
m.updateComposeFn = c.fn
|
||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return true, "fake healthy" }
|
||
m.updateMemoryFn = func(int, int, int, int) (error, string) { return nil, "" }
|
||
m.updateDiskFreeFn = func() (float64, bool) { return 50, true }
|
||
m.updateNowFn = func() time.Time { return slice4T0 } // R-457: the SAME clock the age check reads
|
||
m.execFn = func(string, ...string) (string, error) { return "", nil }
|
||
return m, dir, g, c
|
||
}
|
||
|
||
func waitUpdateDone(t *testing.T, m *Manager, name string) *Stack {
|
||
t.Helper()
|
||
deadline := time.Now().Add(5 * time.Second)
|
||
for time.Now().Before(deadline) {
|
||
if st, ok := m.GetStack(name); ok && !st.Updating {
|
||
return st
|
||
}
|
||
time.Sleep(5 * time.Millisecond)
|
||
}
|
||
t.Fatal("the update never finished")
|
||
return nil
|
||
}
|
||
|
||
func pinOf(t *testing.T, dir string) string { return readPin(t, dir).PinnedImages["web"] }
|
||
|
||
func fileBody(t *testing.T, p string) string {
|
||
t.Helper()
|
||
b, err := os.ReadFile(p)
|
||
if err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
return string(b)
|
||
}
|
||
|
||
func journalExists(m *Manager) bool {
|
||
_, err := os.Stat(m.updateJournalPath())
|
||
return err == nil
|
||
}
|
||
|
||
// ── A: the happy path, and the truth ─────────────────────────────────────────────────────────────
|
||
|
||
// TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth. The health wait BLOCKS until the test releases it;
|
||
// while it blocks, the app must read Updating=true / phase=verifying / no error — and the safety dump
|
||
// must have run while the pin still named the OLD version.
|
||
//
|
||
// COMPANION RED-PROOF 1 (REPORT.md): delete the updateHealth call from verifyAndConclude so success is
|
||
// declared on the compose exit code. This test then fails at "Updating went false before health was
|
||
// known" — which is R-443 exactly.
|
||
func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
release := make(chan struct{})
|
||
healthCalled := make(chan struct{}, 1)
|
||
m.updateHealthFn = func(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
||
healthCalled <- struct{}{}
|
||
<-release
|
||
return true, "fake healthy"
|
||
}
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatalf("a fully-qualified update must start: %v", err)
|
||
}
|
||
select {
|
||
case <-healthCalled:
|
||
case <-time.After(5 * time.Second):
|
||
st, _ := m.GetStack("nextcloud")
|
||
t.Fatalf("the health wait was never reached; state: updating=%v phase=%s err=%q", st.Updating, st.UpdatePhase, st.UpdateError)
|
||
}
|
||
st, _ := m.GetStack("nextcloud")
|
||
if !st.Updating {
|
||
t.Fatal("Updating went false before health was known — success reported on the compose exit code (R-443)")
|
||
}
|
||
if st.UpdatePhase != UpdatePhaseVerifying || st.UpdatePhaseLabel != "Működés ellenőrzése…" {
|
||
t.Errorf("while waiting for health the phase must be verifying, got %q / %q", st.UpdatePhase, st.UpdatePhaseLabel)
|
||
}
|
||
if st.UpdateError != "" {
|
||
t.Errorf("no error may be shown while verifying, got %q", st.UpdateError)
|
||
}
|
||
if !journalExists(m) {
|
||
t.Error("the journal must exist while the update is in flight (Scenario G depends on it)")
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||
t.Errorf("by verifying, the pin must have advanced; got %q", got)
|
||
}
|
||
close(release)
|
||
st = waitUpdateDone(t, m, "nextcloud")
|
||
if st.UpdatePhase != UpdatePhaseDone || st.UpdateError != "" || st.UpdatePhaseLabel != "Frissítve" {
|
||
t.Fatalf("after health the update is done: phase=%q label=%q err=%q", st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError)
|
||
}
|
||
if journalExists(m) {
|
||
t.Error("a completed update must clear its journal entry")
|
||
}
|
||
if _, err := os.Stat(filepath.Join(dir, preUpdateComposeFile)); err == nil {
|
||
t.Error("the pre-update copy must be removed after success")
|
||
}
|
||
if g.pinAtDump != "nextcloud:31.0.14-apache" {
|
||
t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump)
|
||
}
|
||
// v0.263.0: `stop` between the pull and `up` is the undo's copy window (the app stops there anyway).
|
||
if got, want := strings.Join(c.list(), " | "), "pull | stop | up -d --remove-orphans"; got != want {
|
||
t.Errorf("compose calls = %q, want %q", got, want)
|
||
}
|
||
// R-475: the preflight asks only whether a backup could be taken; the job reads the copies once.
|
||
if got := strings.Join(g.callList(), ","); got != "CanBackUp,RestorePoints,SafetyDump" {
|
||
t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got)
|
||
}
|
||
}
|
||
|
||
// ── B: the proven copy is too old ─────────────────────────────────────────────────────────────────
|
||
|
||
func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
|
||
m, dir, g, _ := newSlice4Manager(t)
|
||
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
|
||
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierSecondDrive, ProvenAt: slice4T0.Add(-1 * time.Minute)}}
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
st := waitUpdateDone(t, m, "nextcloud")
|
||
if st.UpdatePhase != UpdatePhaseDone {
|
||
t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||
}
|
||
calls := strings.Join(g.callList(), ",")
|
||
if !strings.HasPrefix(calls, "CanBackUp,RestorePoints,BackupNow,RestorePoints,SafetyDump") {
|
||
t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls)
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||
t.Errorf("pin = %q", got)
|
||
}
|
||
}
|
||
|
||
func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour)
|
||
g.backupErr = errors.New("disk full")
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
st := waitUpdateDone(t, m, "nextcloud")
|
||
if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want {
|
||
t.Errorf("UpdateError = %q, want the backup's own error in the sentence %q", st.UpdateError, want)
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||
t.Errorf("a failed backup must move nothing; pin = %q", got)
|
||
}
|
||
if len(c.list()) != 0 {
|
||
t.Errorf("a failed backup must reach no compose call; got %v", c.list())
|
||
}
|
||
if journalExists(m) {
|
||
t.Error("the journal must be cleared on a refusal")
|
||
}
|
||
}
|
||
|
||
func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
g.points[0].ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: pointsAfterBackup nil
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
st := waitUpdateDone(t, m, "nextcloud")
|
||
if st.UpdateError != MsgUpdateBackupNoUnit {
|
||
t.Errorf("UpdateError = %q", st.UpdateError)
|
||
}
|
||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
|
||
t.Error("nothing may move when the backup did not produce a fresh restorable copy")
|
||
}
|
||
}
|
||
|
||
// ── C / R-475 L: no copy on any tier, and no way to make one ─────────────────────────────────────
|
||
|
||
// Until v0.239.0 this refused any app without a restorable Tier-2 unit. R-475: every tier counts and
|
||
// an app with nothing is backed up first, so the refusal is now only "nothing anywhere AND no backup
|
||
// can be taken". The K half (nothing, but a backup CAN be taken) is TestR475_K in update_tiers_test.go.
|
||
func TestSlice4_C_NoCopyAndNoWayToBackUpRefusesBeforeAnythingMoves(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
g.points, g.cannotBackUp = nil, true
|
||
err := m.StartGuardedUpdate("nextcloud")
|
||
var ref *UpdateRefusal
|
||
if !errors.As(err, &ref) || ref.Reason != "no_backup" {
|
||
t.Fatalf("no copy anywhere and no way to back up must be REFUSED (no_backup), got %v", err)
|
||
}
|
||
if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want {
|
||
t.Errorf("message = %q", ref.Message)
|
||
}
|
||
time.Sleep(50 * time.Millisecond)
|
||
if st, _ := m.GetStack("nextcloud"); st.Updating {
|
||
t.Error("a refused update must not set Updating")
|
||
}
|
||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
|
||
t.Error("a refused update must move nothing")
|
||
}
|
||
for _, call := range g.callList() {
|
||
if call == "BackupNow" {
|
||
t.Error("a refused update must not try to back up")
|
||
}
|
||
}
|
||
}
|
||
|
||
// ── D: the cheap refusals, each one ─────────────────────────────────────────────────────────────────
|
||
|
||
func TestSlice4_D_CheapRefusals(t *testing.T) {
|
||
cases := []struct {
|
||
name string
|
||
setup func(m *Manager, g *fakeGuards, dir string)
|
||
reason string
|
||
msg string
|
||
}{
|
||
{"held", func(m *Manager, g *fakeGuards, _ string) { g.held, g.holdWhy = true, "THE HOLD TEXT" }, "held", "THE HOLD TEXT"},
|
||
{"busy", func(m *Manager, g *fakeGuards, _ string) { g.busy = true }, "busy", MsgUpdateBusy},
|
||
{"already updating", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Updating = true }, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, "nextcloud")},
|
||
{"deploying", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Deploying = true }, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, "nextcloud")},
|
||
{"memory", func(m *Manager, _ *fakeGuards, _ string) {
|
||
catDir := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates", "nextcloud")
|
||
if err := os.WriteFile(filepath.Join(catDir, ".felhom.yml"), []byte("resources:\n mem_request: 900M\n"), 0o644); err != nil {
|
||
panic(err)
|
||
}
|
||
m.updateMemoryFn = func(newReq, _, _, _ int) (error, string) {
|
||
return fmt.Errorf("Nincs elég memória (%d MB)", newReq), ""
|
||
}
|
||
}, "memory", "Nincs elég memória (900 MB)"},
|
||
{"disk", func(m *Manager, _ *fakeGuards, _ string) {
|
||
m.updateDiskFreeFn = func() (float64, bool) { return 1.0, true }
|
||
}, "disk", fmt.Sprintf(MsgUpdateDiskFmt, 1.0, updateDiskFloorGiB)},
|
||
{"guards unwired", func(m *Manager, _ *fakeGuards, _ string) { m.updateGuards = nil }, "guards_unwired", MsgUpdateNoGuards},
|
||
}
|
||
for _, tc := range cases {
|
||
t.Run(tc.name, func(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
tc.setup(m, g, dir)
|
||
ref := m.UpdatePreflight("nextcloud")
|
||
if ref == nil || ref.Reason != tc.reason {
|
||
t.Fatalf("want refusal %q, got %+v", tc.reason, ref)
|
||
}
|
||
if ref.Message != tc.msg {
|
||
t.Errorf("message = %q, want %q", ref.Message, tc.msg)
|
||
}
|
||
if err := m.StartGuardedUpdate("nextcloud"); err == nil {
|
||
t.Fatal("StartGuardedUpdate must refuse the same")
|
||
}
|
||
time.Sleep(20 * time.Millisecond)
|
||
if len(c.list()) != 0 || pinOf(t, dir) != "nextcloud:31.0.14-apache" {
|
||
t.Errorf("a cheap refusal reached the act: compose=%v pin=%s", c.list(), pinOf(t, dir))
|
||
}
|
||
})
|
||
}
|
||
}
|
||
|
||
// ── E: the pull fails ────────────────────────────────────────────────────────────────────────────
|
||
|
||
func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
c.fail["pull"] = errors.New("exit code 1\nstderr: manifest unknown")
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
st := waitUpdateDone(t, m, "nextcloud")
|
||
if st.UpdateError != MsgUpdatePullFailed {
|
||
t.Errorf("UpdateError = %q, want the Hungarian sentence and never raw stderr", st.UpdateError)
|
||
}
|
||
if strings.Contains(st.UpdateError, "manifest unknown") {
|
||
t.Error("raw Docker stderr leaked into the customer sentence")
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||
t.Errorf("the pin must be PUT BACK after a failed pull, got %q", got)
|
||
}
|
||
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
|
||
t.Errorf("the live file must be re-rendered to the old version:\n%s", got)
|
||
}
|
||
if got := fileBody(t, AppliedComposePath(dir)); got != pinTplOld {
|
||
t.Errorf("the stored definition must be the old one again:\n%s", got)
|
||
}
|
||
if got := strings.Join(c.list(), " | "); got != "pull" {
|
||
t.Errorf("after a failed pull nothing else runs; compose calls = %q", got)
|
||
}
|
||
for _, call := range g.callList() {
|
||
if call == "HoldAfterFailedUpdate" {
|
||
t.Error("a failed PULL ran nothing and must not hold the app")
|
||
}
|
||
}
|
||
}
|
||
|
||
// ── F: the new version does not come up ─────────────────────────────────────────────────────────
|
||
|
||
// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test
|
||
// then fails: no hold, and the customer is not told the route back.
|
||
//
|
||
// v0.263.0: a failed health check is UNDONE first (undo.go). Here the old version fails too (the same
|
||
// fake health answers the undo), so this is now "the undo failed as well → HOLD, saying so". The pin is
|
||
// BACK on the old version because the undo put the old definition back before it checked it; the case
|
||
// where no undo can be attempted — the pin stays new — is TestSlice4_G_InterruptedAfterUpResumesTheHealthWait.
|
||
func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
st := waitUpdateDone(t, m, "nextcloud")
|
||
if st.UpdatePhase != UpdatePhaseFailed {
|
||
t.Errorf("phase = %q", st.UpdatePhase)
|
||
}
|
||
held, _ := g.HoldFor("nextcloud")
|
||
if !held {
|
||
t.Fatal("an app that did not come up must be HELD")
|
||
}
|
||
if !g.holdRP.ProvenAt.Equal(g.points[0].ProvenAt) || g.holdRP.Tier != UpdateTierSecondDrive {
|
||
t.Errorf("the hold must name the PROVEN copy it leans on (%+v), got %+v", g.points[0], g.holdRP)
|
||
}
|
||
if st.UpdateError != "HELD-SENTENCE" {
|
||
t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError)
|
||
}
|
||
if st.HoldReason != "HELD-SENTENCE" {
|
||
t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason)
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||
t.Errorf("the undo put the old definition back before its check failed; the pin must say so, got %q", got)
|
||
}
|
||
if g.undoState != UndoStateNotStarted {
|
||
t.Errorf("the hold must say the undo was tried and the old version did not start, got undo state %q", g.undoState)
|
||
}
|
||
// The `logs` call before each `down` is R-621: the hold keeps the app's own log BEFORE the `down`
|
||
// destroys it — once for the new version, once for the old one the undo tried. The order is the
|
||
// assertion — a capture after the `down` would read empty.
|
||
if got := strings.Join(c.list(), " | "); got != "pull | stop | up -d --remove-orphans | logs --no-color --tail 400 | down | up -d --remove-orphans | logs --no-color --tail 400 | down" {
|
||
t.Errorf("the failed app must be stopped, its log kept FIRST, the undo tried and its log kept too; compose calls = %q", got)
|
||
}
|
||
if journalExists(m) {
|
||
t.Error("the journal is cleared once the hold (the durable record) is written")
|
||
}
|
||
}
|
||
|
||
func TestSlice4_F_UnsavedHoldSaysSo(t *testing.T) {
|
||
m, _, g, _ := newSlice4Manager(t)
|
||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
|
||
g.holdErr = errors.New("settings.json read-only")
|
||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdateError != MsgUpdateHoldUnsaved {
|
||
t.Errorf("an unrecorded hold must be disclosed, got %q", st.UpdateError)
|
||
}
|
||
}
|
||
|
||
// ── G: the controller restarts mid-update ────────────────────────────────────────────────────────
|
||
|
||
func writeTestJournal(t *testing.T, m *Manager, name string, e updateJournalEntry) {
|
||
t.Helper()
|
||
if err := m.writeUpdateJournal(updateJournal{Updates: map[string]updateJournalEntry{name: e}}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
}
|
||
|
||
// simulateAdvanced puts the stack in the state a crash AFTER the pin moved would leave: pin, live file
|
||
// and stored definition all new, the pre-update copy on disk.
|
||
func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry {
|
||
t.Helper()
|
||
mustWrite(t, filepath.Join(dir, preUpdateComposeFile), pinTplOld)
|
||
mustWrite(t, filepath.Join(dir, preUpdateAppliedFile), pinTplOld)
|
||
if err := m.advancePinToCatalog("nextcloud", dir); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
return updateJournalEntry{
|
||
StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"},
|
||
PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile),
|
||
ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339),
|
||
ProvenTier: UpdateTierLocal,
|
||
}
|
||
}
|
||
|
||
func TestSlice4_G_InterruptedBeforeUpIsPutBack(t *testing.T) {
|
||
m, dir, _, c := newSlice4Manager(t)
|
||
e := simulateAdvanced(t, m, dir)
|
||
e.Phase = UpdatePhasePulling
|
||
writeTestJournal(t, m, "nextcloud", e)
|
||
|
||
if resumed := m.RecoverUpdates(); len(resumed) != 0 {
|
||
t.Fatalf("an update interrupted before `up` is not resumed, got %v", resumed)
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||
t.Errorf("the pin must be put back, got %q", got)
|
||
}
|
||
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
|
||
t.Errorf("the live file must be the old definition again:\n%s", got)
|
||
}
|
||
st, _ := m.GetStack("nextcloud")
|
||
if st.Updating || st.UpdateError != MsgUpdateInterrupted {
|
||
t.Errorf("updating=%v err=%q", st.Updating, st.UpdateError)
|
||
}
|
||
if journalExists(m) || len(c.list()) != 0 {
|
||
t.Error("recovery of a pre-up interruption runs nothing and clears the journal")
|
||
}
|
||
}
|
||
|
||
func TestSlice4_G_InterruptedBeforeThePinIsDroppedUntouched(t *testing.T) {
|
||
m, dir, _, _ := newSlice4Manager(t)
|
||
writeTestJournal(t, m, "nextcloud", updateJournalEntry{Phase: UpdatePhaseSafetyDump, StartedAt: slice4T0})
|
||
m.RecoverUpdates()
|
||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || journalExists(m) {
|
||
t.Error("an update interrupted before the pin moved must leave the pin and clear the journal")
|
||
}
|
||
}
|
||
|
||
func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) {
|
||
m, dir, g, c := newSlice4Manager(t)
|
||
e := simulateAdvanced(t, m, dir)
|
||
e.Phase = UpdatePhaseVerifying
|
||
writeTestJournal(t, m, "nextcloud", e)
|
||
guards := m.updateGuards
|
||
m.updateGuards = nil // at RecoverUpdates time the backup side is NOT wired yet (main.go order)
|
||
|
||
resumed := m.RecoverUpdates()
|
||
if len(resumed) != 1 || !m.IsUpdating("nextcloud") || !m.UpdatingStacks()["nextcloud"] {
|
||
t.Fatalf("an update interrupted after `up` must be marked Updating for the boot sweep; resumed=%v", resumed)
|
||
}
|
||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||
t.Errorf("after `up` the pin is NOT put back — something may have run; got %q", got)
|
||
}
|
||
|
||
m.updateGuards = guards
|
||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "still broken" }
|
||
if n := m.ResumeInterruptedUpdates(context.Background()); n != 1 {
|
||
t.Fatalf("resumed %d, want 1", n)
|
||
}
|
||
st := waitUpdateDone(t, m, "nextcloud")
|
||
if held, _ := g.HoldFor("nextcloud"); !held || st.UpdatePhase != UpdatePhaseFailed {
|
||
t.Errorf("a resumed update that is still unhealthy must end HELD; held=%v phase=%q", held, st.UpdatePhase)
|
||
}
|
||
// Same R-621 capture on the resumed path — a hold reached by resumption keeps its evidence too.
|
||
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | logs --no-color --tail 400 | down" {
|
||
t.Errorf("resumption re-runs `up`, keeps the log, then stops the failed app; compose calls = %q", got)
|
||
}
|
||
if !g.holdRP.ProvenAt.Equal(slice4T0.Add(-time.Hour)) || g.holdRP.Tier != UpdateTierLocal {
|
||
t.Errorf("the resumed hold must name the journaled copy (tier %d at %s), got %+v", UpdateTierLocal, slice4T0.Add(-time.Hour), g.holdRP)
|
||
}
|
||
}
|
||
|
||
// ── the page reads the hold from the ONE store ──────────────────────────────────────────────────
|
||
|
||
func TestSlice4_GetStacksCarriesTheHoldText(t *testing.T) {
|
||
m, _, g, _ := newSlice4Manager(t)
|
||
g.held, g.holdWhy = true, "HOLD"
|
||
for _, st := range m.GetStacks() {
|
||
if st.Name == "nextcloud" && st.HoldReason != "HOLD" {
|
||
t.Errorf("GetStacks HoldReason = %q", st.HoldReason)
|
||
}
|
||
}
|
||
g.held = false
|
||
if st, _ := m.GetStack("nextcloud"); st.HoldReason != "" {
|
||
t.Errorf("a lifted hold must disappear on the next read, got %q", st.HoldReason)
|
||
}
|
||
}
|
||
|
||
func TestSlice4_PhaseLabelsAreTheSpecifiedCopy(t *testing.T) {
|
||
want := map[string]string{
|
||
UpdatePhaseChecking: "Ellenőrzés…",
|
||
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
|
||
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
|
||
UpdatePhasePulling: "Új verzió letöltése…",
|
||
UpdatePhaseStarting: "Indítás az új verzióval…",
|
||
UpdatePhaseVerifying: "Működés ellenőrzése…",
|
||
UpdatePhaseDone: "Frissítve",
|
||
}
|
||
for p, l := range want {
|
||
if got := UpdatePhaseLabel(p); got != l {
|
||
t.Errorf("label(%s) = %q, want %q", p, got, l)
|
||
}
|
||
}
|
||
}
|