v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s

POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.

The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.

Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.

Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.

Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.

Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-13 11:41:31 +02:00
parent 1552716722
commit 0d402f711d
25 changed files with 2709 additions and 123 deletions
+66 -46
View File
@@ -4,6 +4,7 @@ import (
"crypto/rand"
"encoding/base64"
"encoding/hex"
"errors"
"fmt"
"log"
"math/big"
@@ -236,52 +237,12 @@ func (m *Manager) DeployStack(req DeployRequest) (string, error) {
}
// --- Memory validation ---
var deployWarning string
reservedMB := m.cfg.System.ReservedMemoryMB
totalMB, usedMB, memErr := system.GetMemoryMB()
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
// either never fire (host total) or always fire (host used > guest cap).
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
totalMB = gt
memErr = nil
}
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
usedMB = committedReqMB
}
if memErr != nil {
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
} else {
usableMB := totalMB - reservedMB
newReqMB := ParseMemoryMB(meta.Resources.MemRequest)
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
// Hard block: committed + new request exceeds usable memory
if newReqMB > 0 && usedMB+newReqMB > usableMB {
clearDeploying()
return "", fmt.Errorf(
"Nincs elég memória az alkalmazás telepítéséhez. "+
"Szükséges: %d MB, Elérhető: %d MB "+
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
newReqMB,
usableMB-usedMB,
totalMB,
usedMB,
reservedMB,
)
}
// Soft warning: limits exceed total (overcommit)
_, currentLimitMB := m.CommittedMemory()
newLimitMB := ParseMemoryMB(meta.Resources.MemLimit)
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
deployWarning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
"Normál használat mellett ez nem okoz problémát."
}
// Slice 4: the block moved into memoryVerdict so the guarded update applies the SAME check with
// the SAME wording. Behaviour here is unchanged — same inputs, same log line, same refusal text.
refusal, deployWarning := m.memoryVerdict(ParseMemoryMB(meta.Resources.MemRequest), ParseMemoryMB(meta.Resources.MemLimit), 0, 0)
if refusal != "" {
clearDeploying()
return "", errors.New(refusal)
}
// Debug: log received values (redact passwords/secrets)
@@ -1197,3 +1158,62 @@ func randomAlphanumeric(length int) (string, error) {
}
return string(result), nil
}
// memoryVerdict is the deploy's memory check, extracted so the guarded update uses it unchanged
// (slice 4). releasedReqMB/releasedLimitMB are what the act FREES before it takes the new amount — an
// update replaces the app's own current request, so counting both would refuse an update that fits.
// A deploy releases nothing and passes 0, 0.
//
// Returns the refusal (the deploy's own Hungarian wording, "" = admitted) and the soft overcommit
// warning. An unreadable memory reading admits with a WARN, exactly as the deploy always has.
func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string) {
reservedMB := m.cfg.System.ReservedMemoryMB
totalMB, usedMB, memErr := system.GetMemoryMB()
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
// either never fire (host total) or always fire (host used > guest cap).
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
totalMB = gt
memErr = nil
}
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
usedMB = committedReqMB
}
if memErr != nil {
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
return "", ""
}
usedMB -= releasedReqMB
if usedMB < 0 {
usedMB = 0
}
usableMB := totalMB - reservedMB
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
// Hard block: committed + new request exceeds usable memory
if newReqMB > 0 && usedMB+newReqMB > usableMB {
return fmt.Sprintf(
"Nincs elég memória az alkalmazás telepítéséhez. "+
"Szükséges: %d MB, Elérhető: %d MB "+
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
newReqMB,
usableMB-usedMB,
totalMB,
usedMB,
reservedMB,
), ""
}
// Soft warning: limits exceed total (overcommit)
_, currentLimitMB := m.CommittedMemory()
currentLimitMB -= releasedLimitMB
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
warning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
"Normál használat mellett ez nem okoz problémát."
}
return "", warning
}
+4 -2
View File
@@ -407,7 +407,9 @@ func TestGroupE_RestartStackReachesTheRecorder(t *testing.T) {
func TestGroupE_EveryBringUpPathCallsTheRecorder(t *testing.T) {
callers := map[string]bool{} // enclosing func name -> calls recordInstalledImages
fset := token.NewFileSet()
for _, src := range []string{"manager.go", "deploy.go"} {
// update.go since v0.237.0: the guarded update replaced UpdateStack, and it records in
// verifyAndConclude — only after the app's health is known (slice 4).
for _, src := range []string{"manager.go", "deploy.go", "update.go"} {
f, err := parser.ParseFile(fset, src, nil, 0)
if err != nil {
t.Fatal(err)
@@ -433,7 +435,7 @@ func TestGroupE_EveryBringUpPathCallsTheRecorder(t *testing.T) {
}
}
}
for _, want := range []string{"StartStack", "RestartStack", "UpdateStack", "runComposeDeploy"} {
for _, want := range []string{"StartStack", "RestartStack", "verifyAndConclude", "runComposeDeploy"} {
if !callers[want] {
t.Errorf("%s does not call recordInstalledImages — a bring-up path that records nothing leaves a stale record standing", want)
}
+44 -61
View File
@@ -2,6 +2,7 @@ package stacks
import (
"bytes"
"context"
"fmt"
"log"
"os"
@@ -130,17 +131,28 @@ type HealthCheckDetail struct {
// Stack represents a docker compose stack on disk.
type Stack struct {
Name string `json:"name"`
Meta Metadata `json:"meta"`
ComposePath string `json:"compose_path"`
State ContainerState `json:"state"`
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
Protected bool `json:"protected"`
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
Containers []ContainerInfo `json:"containers"`
AppConfig *AppConfig `json:"app_config,omitempty"`
Deploying bool `json:"deploying"` // compose up in progress
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
Name string `json:"name"`
Meta Metadata `json:"meta"`
ComposePath string `json:"compose_path"`
State ContainerState `json:"state"`
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
Protected bool `json:"protected"`
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
Containers []ContainerInfo `json:"containers"`
AppConfig *AppConfig `json:"app_config,omitempty"`
Deploying bool `json:"deploying"` // compose up in progress
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
// Updating / UpdatePhase / UpdatePhaseLabel / UpdateError (update arc slice 4, v0.237.0) are the
// guarded update's in-memory progress, the same shape as Deploying/DeployError: the API answers
// 202 at once and the page polls GET /api/stacks/{name}. See update.go.
Updating bool `json:"updating"`
UpdatePhase string `json:"update_phase,omitempty"`
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
UpdateError string `json:"update_error,omitempty"`
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
HoldReason string `json:"hold_reason,omitempty"`
HealthProbe *HealthProbeResult `json:"health_probe,omitempty"` // controller-side probe result
LastUpdated time.Time `json:"last_updated"`
// RestartingSince (C9-F2) is when this stack was FIRST observed in StateRestarting during the
@@ -198,6 +210,16 @@ type Manager struct {
restartPolicyCache map[string]string
// execFn replaces execCommand's process boundary in tests; nil in production.
execFn func(name string, args ...string) (string, error)
// --- guarded update (slice 4, update.go) ---
updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED
updateComposeFn func(dir string, env []string, args ...string) (string, error)
updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string)
updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string)
updateDiskFreeFn func() (freeGiB float64, ok bool)
updateNowFn func() time.Time
updateJournalMu sync.Mutex
updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist
// inspectRestartPolicyFn is the docker-inspect seam for the above; nil in production
// (dockerRestartPolicy). Tests inject a scripted lookup and never touch docker.
inspectRestartPolicyFn func(containerName string) (string, error)
@@ -960,12 +982,15 @@ func aggregateState(containers []ContainerInfo, policyOf restartPolicyLookup) Co
func (m *Manager) GetStacks() []Stack {
m.mu.RLock()
defer m.mu.RUnlock()
result := make([]Stack, 0, len(m.stacks))
for _, s := range m.stacks {
result = append(result, deepCopyStack(s))
}
g := m.updateGuards
m.mu.RUnlock()
for i := range result {
fillHoldReason(g, &result[i])
}
// Sort alphabetically by display name for consistent UI ordering
sort.Slice(result, func(i, j int) bool {
@@ -984,6 +1009,7 @@ func (m *Manager) GetStack(name string) (*Stack, bool) {
return nil, false
}
cp := deepCopyStack(s)
fillHoldReason(m.updateGuards, &cp)
return &cp, true
}
@@ -1224,54 +1250,11 @@ func (m *Manager) RestartStack(name string) error {
return m.RefreshStatus()
}
func (m *Manager) UpdateStack(name string) error {
stack, ok := m.GetStack(name)
if !ok {
return fmt.Errorf("stack %q not found", name)
}
m.logger.Printf("[INFO] [stacks] Updating stack: %s", name)
start := time.Now()
dir := filepath.Dir(stack.ComposePath)
// v0.235.0 — ADVANCE THE PIN FIRST, AND RE-RENDER BEFORE THE PULL.
//
// This is the ONE act entitled to move a version; the freeze exists so that nothing else can.
// The ordering is load-bearing, not stylistic: `compose pull` and `up -d` act on the file on
// disk, so the catalog's current definition has to BE that file before either runs. Setting the
// pin afterwards would pull the frozen version and change nothing, while reporting success — and
// a button that lies is worse than a button that refuses.
//
// A FAILED PIN WRITE REFUSES THE UPDATE, deliberately the opposite of recordInstalledImages.
// That field is an observation and a failed write is a bookkeeping gap; this one is INTENT, and
// an update whose intent could not be recorded leaves the box running a version it has no record
// of choosing — the exact ambiguity R-166 closed for desired_state, one field over.
if err := m.advancePinToCatalog(name, dir); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s update refused: %v", name, err)
return fmt.Errorf("updating stack %s: %w", name, err)
}
env := m.stackEnv(dir)
if m.isDebug() {
m.checkLocalImages(name, dir)
}
if _, err := m.composeExecCustomEnv(dir, env, "pull"); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s update (pull) failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
return fmt.Errorf("pulling images for %s: %w", name, err)
}
if _, err := m.composeExecCustomEnv(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s update (up) failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
return fmt.Errorf("recreating %s: %w", name, err)
}
m.logger.Printf("[INFO] [stacks] Stack %s updated successfully (took %.1fs)", name, time.Since(start).Seconds())
m.recordInstalledImages(name, dir, env)
m.logPostStartStatus(name, dir, env)
return m.RefreshStatus()
}
// UpdateStack was REMOVED in v0.237.0 (update arc slice 4). It advanced the pin, pulled, ran `up -d`
// and reported success on the compose exit code — no copy first, no refusals, and HTTP 200 over a
// crash loop (R-443). Its only caller was the API, which now runs StartGuardedUpdate (update.go).
// Deleting it rather than leaving it is deliberate: an unguarded update path that still compiles is
// one caller away from being the next R-439.
func (m *Manager) GetLogs(name string, lines int) (string, error) {
stack, ok := m.GetStack(name)
+802
View File
@@ -0,0 +1,802 @@
package stacks
import (
"context"
"encoding/json"
"fmt"
"os"
"path/filepath"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
)
// ── The guarded update (update arc slice 4, controller v0.237.0) ─────────────────────────────────
//
// WHAT IT REPLACED. `UpdateStack` advanced the pin, pulled, ran `up -d` and returned — no copy first,
// no check of memory, disk, a running backup or a held app, and a success the moment `up` returned.
// SPIKE-app-update-2026-09-01 §4 measured that as HTTP 200 over an app that was already crash-looping
// (R-443), and R-439 is that a held app could be updated at all.
//
// THE SEQUENCE, and the order is the point:
//
// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling
// → starting → verifying → done | failed
//
// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02,
// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The
// same predicate that permits the destructive „Teljes visszaállítás" permits the update — the
// route back IS that restore, so an update without it has no route back.
// 2. The safety dump is taken BEFORE the pin moves: it is "the state the customer was in a minute ago",
// and a minute later the migration may have run.
// 3. The pin moves BEFORE the pull (v0.235.0's reason: pull and up act on the file on disk).
// 4. A PULL failure puts the pin BACK — nothing ran, so reverting is safe and honest (Scenario E).
// 5. A HEALTH failure leaves the pin where it is — the new version's migration may have run, and a
// pin claiming the old version would be a record of something untrue (Scenario F). The app is
// HELD STOPPED and the customer is told which backup it can be restored from.
//
// WHAT IT DELIBERATELY DOES NOT DO: put the old version back by itself. SPIKE-upgrade-test-2026-09-06
// measured that whether the old image starts on migrated data depends on the app (PrivateBin yes,
// Docmost and Nextcloud no) and cannot be predicted. The route back is the restore.
//
// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10),
// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at
// the next startup (Scenario G) — the AppStopGuard pattern.
// Update phases, as recorded in the journal and served as Stack.UpdatePhase.
const (
UpdatePhaseChecking = "checking"
UpdatePhaseBackingUp = "backing-up"
UpdatePhaseSafetyDump = "safety-dump"
UpdatePhasePinning = "pinning"
UpdatePhasePulling = "pulling"
UpdatePhaseStarting = "starting"
UpdatePhaseVerifying = "verifying"
UpdatePhaseDone = "done"
UpdatePhaseFailed = "failed"
)
// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the
// specification — it is instantaneous and is the first step of fetching the new version, so it shares
// the pull's label rather than inventing a sentence nobody would read.
var updatePhaseLabels = map[string]string{
UpdatePhaseChecking: "Ellenőrzés…",
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
UpdatePhasePinning: "Új verzió letöltése…",
UpdatePhasePulling: "Új verzió letöltése…",
UpdatePhaseStarting: "Indítás az új verzióval…",
UpdatePhaseVerifying: "Működés ellenőrzése…",
UpdatePhaseDone: "Frissítve",
UpdatePhaseFailed: "A frissítés nem sikerült",
}
// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one.
func UpdatePhaseLabel(phase string) string { return updatePhaseLabels[phase] }
// Customer sentences. Named so tests compare against the constant, never a retyped literal (R-364).
const (
MsgUpdateNoGuards = "A frissítés nem indítható: a frissítés előtti biztonsági ellenőrzés nem érhető el ezen a szerveren."
MsgUpdateNotDeployed = "Az alkalmazás nincs telepítve, ezért nem frissíthető."
MsgUpdateDeployingFmt = "A(z) %s telepítése még folyamatban van — a frissítés utána indítható."
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani. Kapcsold be a 2. mentést az alkalmazás mentési beállításainál a Mentések oldalon, és várd meg az első sikeres másolatot — utána a frissítés elindítható."
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
MsgUpdateDumpFailFmt = "A frissítés nem indult el, mert az adatbázis pillanatkép nem készült el: %v. Az alkalmazás változatlanul fut tovább."
MsgUpdatePinFailed = "A frissítés nem indult el: az új verzió leírása nem olvasható be. Az alkalmazás változatlanul fut tovább."
MsgUpdateJournalFailed = "A frissítés nem indult el: a frissítés naplója nem menthető. Az alkalmazás változatlanul fut tovább."
MsgUpdatePullFailed = "Az új verzió letöltése nem sikerült, ezért a frissítés elmaradt. Az alkalmazás a korábbi verzióval fut tovább."
MsgUpdateInterrupted = "A frissítés megszakadt, mert a vezérlő újraindult, mielőtt az új verzió elindult volna. Az alkalmazás a korábbi verzióval fut tovább."
MsgUpdateHoldUnsaved = "A frissítés nem sikerült, az alkalmazás le lett állítva, de a leállítás rögzítése nem sikerült. Ne indítsd újra — vedd fel velünk a kapcsolatot."
)
// updateDiskFloorGiB is the free space the Docker data root must have before a pull. A FIXED FLOOR,
// stated as such: the new image set's size is not known without a registry query (the catalog
// records tags, not sizes, and §8.1 of 09 already declines registry calls on the customer box), so
// the rule is "not less than 2 GB", not "enough for these images".
const updateDiskFloorGiB = 2.0
// updateSettleWindow is the rule for an app with no .felhom.yml health check: every container
// running, none restarting, for this long.
const updateSettleWindow = 60 * time.Second
// updatePollEvery is how often the health wait re-reads the stack.
const updatePollEvery = 5 * time.Second
// UpdateRestorePoint is the precondition answer, reduced to what the update needs.
type UpdateRestorePoint struct {
Restorable bool // an openable recovery unit exists in the Tier-2 copy
Proven bool // a copy actually succeeded (never an attempt clock)
ProvenAt time.Time // when the data in that copy was last proven copied
}
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
// backup, so cmd/controller/main.go wires an adapter (TestSlice4_UpdateGuardsAreWiredAtStartup).
type UpdateGuards interface {
HoldFor(name string) (bool, string)
Busy(name string) (bool, string)
RestorePoint(name string) (UpdateRestorePoint, error)
BackupNow(ctx context.Context, name string) error
SafetyDump(ctx context.Context, name string) ([]string, error)
HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error
}
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
// an update that cannot see the backup cannot promise a route back.
func (m *Manager) SetUpdateGuards(g UpdateGuards) {
m.mu.Lock()
m.updateGuards = g
m.mu.Unlock()
}
func (m *Manager) guards() UpdateGuards {
m.mu.RLock()
defer m.mu.RUnlock()
return m.updateGuards
}
func fillHoldReason(g UpdateGuards, st *Stack) {
if g == nil || st == nil || !st.Deployed {
return
}
if held, why := g.HoldFor(st.Name); held {
st.HoldReason = why
}
}
// UpdateRefusal is a refusal taken before anything moved. Reason is a stable key for logs and tests;
// Message is the customer sentence.
type UpdateRefusal struct {
Reason string
Message string
}
func (r *UpdateRefusal) Error() string { return r.Message }
func (m *Manager) now() time.Time {
if m.updateNowFn != nil {
return m.updateNowFn()
}
return time.Now()
}
func (m *Manager) refuseUpdate(name, reason, msg, detail string) *UpdateRefusal {
m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail)
return &UpdateRefusal{Reason: reason, Message: msg}
}
// UpdatePreflight runs every CHEAP refusal (slice 4 Part 1 + the precondition's existence), in order,
// and returns the first. Nothing is moved and nothing is recorded by it. The router calls it before
// recording the customer's intent, so an update that was never going to happen records nothing.
func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
st, ok := m.GetStack(name)
if !ok {
return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack")
}
if !st.Deployed {
return m.refuseUpdate(name, "not_deployed", MsgUpdateNotDeployed, "not deployed")
}
g := m.guards()
if g == nil {
return m.refuseUpdate(name, "guards_unwired", MsgUpdateNoGuards, "no UpdateGuards wired — fail closed")
}
if st.Deploying {
return m.refuseUpdate(name, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, name), "a deploy is in progress")
}
if st.Updating {
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "an update is already in progress")
}
if held, why := g.HoldFor(name); held {
return m.refuseUpdate(name, "held", why, "the app is held")
}
if busy, why := g.Busy(name); busy {
return m.refuseUpdate(name, "busy", MsgUpdateBusy, why)
}
if m.IsMigrating() {
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
}
rp, err := g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven {
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
fmt.Sprintf("no restorable proven Tier-2 unit (restorable=%v proven=%v err=%v)", rp.Restorable, rp.Proven, err))
}
if ref := m.updateMemoryRefusal(name, st); ref != nil {
return ref
}
free, known := m.updateDiskFree()
switch {
case !known:
m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB)
case free < updateDiskFloorGiB:
return m.refuseUpdate(name, "disk", fmt.Sprintf(MsgUpdateDiskFmt, free, updateDiskFloorGiB),
fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB))
}
return nil
}
// updateMemoryRefusal applies the deploy's memory check to the NEW template's request, releasing the
// app's CURRENT request first (an update replaces it). An unknown new request proceeds with a WARN.
func (m *Manager) updateMemoryRefusal(name string, st *Stack) *UpdateRefusal {
catPath := m.CatalogTemplatePath(name, ".felhom.yml")
if _, err := os.Stat(catPath); err != nil {
m.logger.Printf("[WARN] [stacks] update %s: the new template's memory request is unknown (%v) — proceeding without the memory check", name, err)
return nil
}
newMeta := LoadMetadata(filepath.Dir(catPath))
newReq, newLim := ParseMemoryMB(newMeta.Resources.MemRequest), ParseMemoryMB(newMeta.Resources.MemLimit)
if newReq == 0 {
m.logger.Printf("[WARN] [stacks] update %s: the new template declares no memory request — proceeding without the memory check", name)
return nil
}
oldReq, oldLim := ParseMemoryMB(st.Meta.Resources.MemRequest), ParseMemoryMB(st.Meta.Resources.MemLimit)
verdict := m.updateMemoryFn
if verdict == nil {
verdict = m.memoryVerdict
}
if refusal, _ := verdict(newReq, newLim, oldReq, oldLim); refusal != "" {
return m.refuseUpdate(name, "memory", refusal, fmt.Sprintf("new_req=%dMB replacing %dMB does not fit", newReq, oldReq))
}
return nil
}
func (m *Manager) updateDiskFree() (float64, bool) {
if m.updateDiskFreeFn != nil {
return m.updateDiskFreeFn()
}
du := system.GetDiskUsage(system.DockerVolumePath)
if du == nil {
return 0, false
}
return du.AvailGB, true
}
// StartGuardedUpdate re-checks the cheap refusals, claims the Updating flag atomically and launches the
// job. It returns as soon as the job has STARTED — the result arrives on GET /api/stacks/{name}.
func (m *Manager) StartGuardedUpdate(name string) error {
if ref := m.UpdatePreflight(name); ref != nil {
return ref
}
m.mu.Lock()
s, ok := m.stacks[name]
if !ok {
m.mu.Unlock()
return &UpdateRefusal{Reason: "not_found", Message: fmt.Sprintf("stack %q not found", name)}
}
// A second press between the preflight and here is the race this lock closes.
if s.Updating || s.Deploying {
m.mu.Unlock()
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
}
s.Updating, s.UpdateError = true, ""
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
m.mu.Unlock()
m.logger.Printf("[INFO] [stacks] update %s: accepted — guarded update started", name)
go m.runGuardedUpdate(context.Background(), name)
return nil
}
// IsUpdating reports whether a guarded update is in progress for the app.
func (m *Manager) IsUpdating(name string) bool {
m.mu.RLock()
defer m.mu.RUnlock()
s, ok := m.stacks[name]
return ok && s.Updating
}
// UpdatingStacks is the set of apps an update is currently moving — for the dead-app alarm, which
// must not count an app the update itself is recreating (R-330's class, a third mechanism).
func (m *Manager) UpdatingStacks() map[string]bool {
m.mu.RLock()
defer m.mu.RUnlock()
var out map[string]bool
for name, s := range m.stacks {
if s.Updating {
if out == nil {
out = map[string]bool{}
}
out[name] = true
}
}
return out
}
func (m *Manager) setUpdatePhase(name, phase string) {
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
}
m.mu.Unlock()
}
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure.
func (m *Manager) finishUpdate(name, phase, msg string) {
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Updating = false
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
s.UpdateError = msg
}
m.mu.Unlock()
}
func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) {
if m.updateComposeFn != nil {
return m.updateComposeFn(dir, env, args...)
}
return m.composeExecCustomEnv(dir, env, args...)
}
func (m *Manager) updateHealth(ctx context.Context, name string, timeout time.Duration) (bool, string) {
if m.updateHealthFn != nil {
return m.updateHealthFn(ctx, name, timeout)
}
return m.waitUpdateHealthy(ctx, name, timeout)
}
func (m *Manager) healthTimeout() time.Duration {
if m.cfg == nil {
return 5 * time.Minute
}
return m.cfg.Update.HealthTimeoutDuration()
}
func (m *Manager) backupMaxAge() time.Duration {
if m.cfg == nil {
return 24 * time.Hour
}
return m.cfg.Update.BackupMaxAgeDuration()
}
// pre-update copies, kept in the stack dir so they travel with it. Neither name is one the syncer
// copies (it copies exactly docker-compose.yml and .felhom.yml).
const (
preUpdateComposeFile = "pre-update-compose.yml"
preUpdateAppliedFile = "pre-update-applied.yml"
)
func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
start := m.now()
st, ok := m.GetStack(name)
if !ok {
m.finishUpdate(name, UpdatePhaseFailed, fmt.Sprintf("stack %q not found", name))
return
}
dir := filepath.Dir(st.ComposePath)
g := m.guards()
entry := updateJournalEntry{StartedAt: start}
fail := func(msg, detail string) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail)
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, msg)
}
if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) {
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateJournalFailed)
return
}
if g == nil {
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
return
}
rp, err := g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven {
fail(fmt.Sprintf(MsgUpdateNoBackupFmt, name), fmt.Sprintf("precondition vanished: restorable=%v proven=%v err=%v", rp.Restorable, rp.Proven, err))
return
}
maxAge := m.backupMaxAge()
if age := start.Sub(rp.ProvenAt); age > maxAge {
m.logger.Printf("[INFO] [stacks] update %s: the proven copy is %s old (limit %s) — backing up first", name, age.Round(time.Minute), maxAge)
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
fail(MsgUpdateJournalFailed, "journal write failed")
return
}
if err := g.BackupNow(ctx, name); err != nil {
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
return
}
rp, err = g.RestorePoint(name)
if err != nil || !rp.Restorable || !rp.Proven || m.now().Sub(rp.ProvenAt) > maxAge {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup: restorable=%v proven=%v at=%s err=%v", rp.Restorable, rp.Proven, rp.ProvenAt.Format(time.RFC3339), err))
return
}
} else {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — proven copy from %s (%s old, limit %s)", name, rp.ProvenAt.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge)
}
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
fail(MsgUpdateJournalFailed, "journal write failed")
return
}
paths, err := g.SafetyDump(ctx, name)
if err != nil {
fail(fmt.Sprintf(MsgUpdateDumpFailFmt, err), "safety dump: "+err.Error())
return
}
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
// PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash
// at any later instant can put it back (Scenario G).
prevLive, err := os.ReadFile(st.ComposePath)
if err != nil {
fail(MsgUpdatePinFailed, "reading the live compose file: "+err.Error())
return
}
if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil {
fail(MsgUpdateJournalFailed, "saving the pre-update compose copy: "+err.Error())
return
}
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
if applied, aerr := LoadAppliedDefinition(dir); aerr == nil {
if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil {
entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile)
}
}
if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 {
entry.PrevPin = map[string]string{}
for k, v := range cfg.PinnedImages {
entry.PrevPin[k] = v
}
}
if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) {
m.removePreUpdateCopies(dir)
fail(MsgUpdateJournalFailed, "journal write failed")
return
}
if err := m.advancePinToCatalog(name, dir); err != nil {
m.pinBack(name, dir, entry)
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
return
}
env := m.stackEnv(dir)
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
m.pinBack(name, dir, entry)
fail(MsgUpdateJournalFailed, "journal write failed")
return
}
if _, err := m.updateCompose(dir, env, "pull"); err != nil {
// Scenario E: NOTHING RAN. The containers are the old ones and still running, so the honest
// state is the old pin and the old file — put both back.
m.pinBack(name, dir, entry)
m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err)
fail(MsgUpdatePullFailed, "pull failed: "+err.Error())
return
}
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "journal write failed before up")
return
}
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
// Containers may already have been recreated on the new image — something may have run.
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "compose up failed: "+err.Error())
return
}
m.verifyAndConclude(ctx, name, dir, env, rp.ProvenAt, start, &entry)
}
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
// known, and a failure holds the app.
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, provenAt, start time.Time, entry *updateJournalEntry) {
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
}
timeout := m.healthTimeout()
waitStart := m.now()
healthy, detail := m.updateHealth(ctx, name, timeout)
if !healthy {
m.failAndHold(ctx, name, dir, env, provenAt, "not healthy: "+detail)
return
}
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
m.recordInstalledImages(name, dir, env)
_ = m.RefreshStatus()
m.clearJournal(name)
m.removePreUpdateCopies(dir)
m.finishUpdate(name, UpdatePhaseDone, "")
m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second))
}
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, provenAt time.Time, why string) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
if _, err := m.updateCompose(dir, env, "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
}
msg := MsgUpdateHoldUnsaved
if g := m.guards(); g == nil {
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
} else if err := g.HoldAfterFailedUpdate(name, m.now(), provenAt); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
}
_ = m.RefreshStatus()
m.clearJournal(name)
m.removePreUpdateCopies(dir)
m.finishUpdate(name, UpdatePhaseFailed, msg)
}
// pinBack restores the pin, the stored definition and the live file from the journaled copies.
func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
prevLive, lerr := os.ReadFile(entry.PrevCompose)
if lerr != nil {
m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr)
}
if len(entry.PrevPin) > 0 {
applied := prevLive
if entry.PrevApplied != "" {
if b, err := os.ReadFile(entry.PrevApplied); err == nil {
applied = b
}
}
if err := m.SetPin(name, dir, entry.PrevPin, applied); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: putting the pin back failed: %v", name, err)
}
}
if lerr == nil {
if err := os.WriteFile(ComposePathIn(dir), prevLive, 0o644); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err)
}
}
m.removePreUpdateCopies(dir)
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
}
func (m *Manager) removePreUpdateCopies(dir string) {
_ = os.Remove(filepath.Join(dir, preUpdateComposeFile))
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
}
// waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the
// existing probe, or — for an app with none — every container running and none restarting for
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
deadline := m.now().Add(timeout)
var runningSince time.Time
last := "no observation yet"
for {
_ = m.RefreshStatus()
st, ok := m.GetStack(name)
switch {
case !ok:
last = "stack vanished"
runningSince = time.Time{}
case st.State == StateRunning:
if hc := st.Meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
if c := findProbeContainer(name, st.Containers); c != "" {
res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.HealthProbe = res
}
m.mu.Unlock()
if res.Healthy {
return true, "the app's health check passed"
}
last = "health check failing"
} else {
last = "no probe container"
}
} else {
if runningSince.IsZero() {
runningSince = m.now()
}
if m.now().Sub(runningSince) >= updateSettleWindow {
return true, fmt.Sprintf("all containers running, none restarting, for %s (no health check declared)", updateSettleWindow)
}
last = "running, settling"
}
default:
runningSince = time.Time{}
last = "state " + string(st.State)
}
if !m.now().Before(deadline) {
return false, fmt.Sprintf("not healthy within %s (last: %s)", timeout, last)
}
select {
case <-ctx.Done():
return false, "cancelled: " + ctx.Err().Error()
case <-time.After(updatePollEvery):
}
}
}
// ── the journal ──────────────────────────────────────────────────────────────────────────────────
type updateJournalEntry struct {
Phase string `json:"phase"`
StartedAt time.Time `json:"started_at"`
PrevPin map[string]string `json:"prev_pin,omitempty"`
PrevCompose string `json:"prev_compose,omitempty"`
PrevApplied string `json:"prev_applied,omitempty"`
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
}
type updateJournal struct {
Updates map[string]updateJournalEntry `json:"updates"`
}
func (m *Manager) updateJournalPath() string {
return filepath.Join(m.cfg.Paths.DataDir, "update-journal.json")
}
func (m *Manager) readUpdateJournal() updateJournal {
j := updateJournal{Updates: map[string]updateJournalEntry{}}
data, err := os.ReadFile(m.updateJournalPath())
if err != nil {
return j
}
if err := json.Unmarshal(data, &j); err != nil {
m.logger.Printf("[WARN] [stacks] update journal at %s is corrupt (%v) — quarantining", m.updateJournalPath(), err)
_ = os.Rename(m.updateJournalPath(), fmt.Sprintf("%s.corrupt-%d", m.updateJournalPath(), time.Now().Unix()))
return updateJournal{Updates: map[string]updateJournalEntry{}}
}
if j.Updates == nil {
j.Updates = map[string]updateJournalEntry{}
}
return j
}
// writeUpdateJournal is atomic and fsynced (the AppStopGuard shape): the point is surviving a power cut.
func (m *Manager) writeUpdateJournal(j updateJournal) error {
p := m.updateJournalPath()
if len(j.Updates) == 0 {
if err := os.Remove(p); err != nil && !os.IsNotExist(err) {
return err
}
return nil
}
data, err := json.MarshalIndent(j, "", " ")
if err != nil {
return err
}
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
return err
}
tmp := p + ".tmp"
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
if err != nil {
return err
}
if _, err := f.Write(data); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Sync(); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Close(); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, p)
}
// enterUpdatePhase journals the phase BEFORE it starts and mirrors it for the UI. False means the
// journal could not be written — the caller must not perform a mutation it could not record.
func (m *Manager) enterUpdatePhase(name string, entry *updateJournalEntry, phase string) bool {
entry.Phase = phase
m.updateJournalMu.Lock()
j := m.readUpdateJournal()
j.Updates[name] = *entry
err := m.writeUpdateJournal(j)
m.updateJournalMu.Unlock()
m.setUpdatePhase(name, phase)
if err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: journal write for phase %s failed: %v", name, phase, err)
return false
}
m.logger.Printf("[INFO] [stacks] update %s: phase %s", name, phase)
return true
}
func (m *Manager) clearJournal(name string) {
m.updateJournalMu.Lock()
defer m.updateJournalMu.Unlock()
j := m.readUpdateJournal()
delete(j.Updates, name)
if err := m.writeUpdateJournal(j); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: clearing the journal entry failed: %v", name, err)
}
}
// RecoverUpdates reads the journal at startup (Scenario G). Call it BEFORE the boot reconciler.
//
// - interrupted BEFORE the pin moved (checking, backing-up, safety-dump): nothing moved — the entry is
// dropped and the app carries the "interrupted" sentence;
// - interrupted while pinning or pulling: nothing RAN — the pin and definition are put back (as E);
// - interrupted while starting or verifying: something may have run — the app is marked Updating
// (so the boot sweep and the dead-app alarm leave it alone) and queued for ResumeInterruptedUpdates,
// which re-runs `up -d` and the health wait, ending in A or F.
//
// It needs no backup wiring, because nothing here holds an app — that is left to the resumed job.
func (m *Manager) RecoverUpdates() []string {
m.updateJournalMu.Lock()
j := m.readUpdateJournal()
m.updateJournalMu.Unlock()
if len(j.Updates) == 0 {
return nil
}
var resumed []string
for name, e := range j.Updates {
st, ok := m.GetStack(name)
if !ok {
m.logger.Printf("[WARN] [stacks] update recovery: %s is in the journal (phase %s) but no longer exists — dropping the entry", name, e.Phase)
m.clearJournal(name)
continue
}
dir := filepath.Dir(st.ComposePath)
switch e.Phase {
case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump:
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
case UpdatePhasePinning, UpdatePhasePulling:
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.pinBack(name, dir, e)
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
case UpdatePhaseStarting, UpdatePhaseVerifying:
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Updating, s.UpdateError = true, ""
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
}
m.updateResume = append(m.updateResume, name)
m.mu.Unlock()
resumed = append(resumed, name)
default:
m.logger.Printf("[WARN] [stacks] update recovery: %s has unknown phase %q — dropping the entry", name, e.Phase)
m.clearJournal(name)
}
}
return resumed
}
// ResumeInterruptedUpdates continues the updates RecoverUpdates queued, once the backup side is wired
// (a resumed update that fails must be able to HOLD). Returns how many were resumed.
func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
m.mu.Lock()
names := m.updateResume
m.updateResume = nil
m.mu.Unlock()
for _, name := range names {
st, ok := m.GetStack(name)
if !ok {
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
continue
}
m.updateJournalMu.Lock()
e, ok := m.readUpdateJournal().Updates[name]
m.updateJournalMu.Unlock()
if !ok {
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
continue
}
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
dir := filepath.Dir(st.ComposePath)
go func(name, dir string, e updateJournalEntry, provenAt time.Time) {
env := m.stackEnv(dir)
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.failAndHold(ctx, name, dir, env, provenAt, "resumed compose up failed: "+err.Error())
return
}
m.verifyAndConclude(ctx, name, dir, env, provenAt, e.StartedAt, &e)
}(name, dir, e, provenAt)
}
return len(names)
}
+569
View File
@@ -0,0 +1,569 @@
package stacks
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"sync"
"testing"
"time"
)
// Update arc slice 4 — the guarded update. Scenarios A–G of the task, through the real job
// (runGuardedUpdate) with the four process boundaries injected: compose, the health wait, the backup
// side (UpdateGuards) and the clock. Every assertion reads the EFFECT back — the pin in app.yaml, the
// bytes of the live compose file, the journal on disk, Updating/UpdatePhase/UpdateError — never "no error".
var slice4T0 = time.Date(2026, 9, 13, 10, 0, 0, 0, time.UTC)
type fakeGuards struct {
mu sync.Mutex
calls []string
held bool
holdWhy string
busy bool
rp UpdateRestorePoint
rpErr error
rpAfterBackup *UpdateRestorePoint
backupErr error
dumpErr error
holdErr error
holdProvenAt time.Time
pinAtDump string
stackDir string
}
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
func (f *fakeGuards) callList() []string {
f.mu.Lock()
defer f.mu.Unlock()
return append([]string(nil), f.calls...)
}
func (f *fakeGuards) HoldFor(string) (bool, string) {
f.mu.Lock()
defer f.mu.Unlock()
return f.held, f.holdWhy
}
func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" }
func (f *fakeGuards) RestorePoint(string) (UpdateRestorePoint, error) {
f.note("RestorePoint")
f.mu.Lock()
defer f.mu.Unlock()
return f.rp, f.rpErr
}
func (f *fakeGuards) BackupNow(context.Context, string) error {
f.note("BackupNow")
f.mu.Lock()
defer f.mu.Unlock()
if f.backupErr == nil && f.rpAfterBackup != nil {
f.rp = *f.rpAfterBackup
}
return f.backupErr
}
func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
f.note("SafetyDump")
if cfg := LoadAppConfig(f.stackDir); cfg != nil {
f.mu.Lock()
f.pinAtDump = cfg.PinnedImages["web"]
f.mu.Unlock()
}
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
}
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, provenAt time.Time) error {
f.note("HoldAfterFailedUpdate")
f.mu.Lock()
defer f.mu.Unlock()
if f.holdErr != nil {
return f.holdErr
}
f.held, f.holdWhy, f.holdProvenAt = true, "HELD-SENTENCE", provenAt
return nil
}
type composeRec struct {
mu sync.Mutex
calls []string
fail map[string]error // first arg → error
}
func (c *composeRec) fn(_ string, _ []string, args ...string) (string, error) {
c.mu.Lock()
defer c.mu.Unlock()
c.calls = append(c.calls, strings.Join(args, " "))
if err := c.fail[args[0]]; err != nil {
return "", err
}
return "", nil
}
func (c *composeRec) list() []string {
c.mu.Lock()
defer c.mu.Unlock()
return append([]string(nil), c.calls...)
}
// newSlice4Manager: a pinned nextcloud on the OLD version, the catalog offering the NEW one, fresh
// proven copy, and every boundary faked. Returns the manager, its stack dir, the guards and compose.
func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec) {
t.Helper()
m, dir := newPinManager(t, pinTplOld, pinTplNew,
"deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n")
mustWrite(t, AppliedComposePath(dir), pinTplOld)
g := &fakeGuards{rp: UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Hour)}, stackDir: dir}
c := &composeRec{fail: map[string]error{}}
m.updateGuards = g
m.updateComposeFn = c.fn
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return true, "fake healthy" }
m.updateMemoryFn = func(int, int, int, int) (string, string) { return "", "" }
m.updateDiskFreeFn = func() (float64, bool) { return 50, true }
m.updateNowFn = func() time.Time { return slice4T0 } // R-457: the SAME clock the age check reads
m.execFn = func(string, ...string) (string, error) { return "", nil }
return m, dir, g, c
}
func waitUpdateDone(t *testing.T, m *Manager, name string) *Stack {
t.Helper()
deadline := time.Now().Add(5 * time.Second)
for time.Now().Before(deadline) {
if st, ok := m.GetStack(name); ok && !st.Updating {
return st
}
time.Sleep(5 * time.Millisecond)
}
t.Fatal("the update never finished")
return nil
}
func pinOf(t *testing.T, dir string) string { return readPin(t, dir).PinnedImages["web"] }
func fileBody(t *testing.T, p string) string {
t.Helper()
b, err := os.ReadFile(p)
if err != nil {
t.Fatal(err)
}
return string(b)
}
func journalExists(m *Manager) bool {
_, err := os.Stat(m.updateJournalPath())
return err == nil
}
// ── A: the happy path, and the truth ─────────────────────────────────────────────────────────────
// TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth. The health wait BLOCKS until the test releases it;
// while it blocks, the app must read Updating=true / phase=verifying / no error — and the safety dump
// must have run while the pin still named the OLD version.
//
// COMPANION RED-PROOF 1 (REPORT.md): delete the updateHealth call from verifyAndConclude so success is
// declared on the compose exit code. This test then fails at "Updating went false before health was
// known" — which is R-443 exactly.
func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
release := make(chan struct{})
healthCalled := make(chan struct{}, 1)
m.updateHealthFn = func(ctx context.Context, name string, timeout time.Duration) (bool, string) {
healthCalled <- struct{}{}
<-release
return true, "fake healthy"
}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatalf("a fully-qualified update must start: %v", err)
}
select {
case <-healthCalled:
case <-time.After(5 * time.Second):
st, _ := m.GetStack("nextcloud")
t.Fatalf("the health wait was never reached; state: updating=%v phase=%s err=%q", st.Updating, st.UpdatePhase, st.UpdateError)
}
st, _ := m.GetStack("nextcloud")
if !st.Updating {
t.Fatal("Updating went false before health was known — success reported on the compose exit code (R-443)")
}
if st.UpdatePhase != UpdatePhaseVerifying || st.UpdatePhaseLabel != "Működés ellenőrzése…" {
t.Errorf("while waiting for health the phase must be verifying, got %q / %q", st.UpdatePhase, st.UpdatePhaseLabel)
}
if st.UpdateError != "" {
t.Errorf("no error may be shown while verifying, got %q", st.UpdateError)
}
if !journalExists(m) {
t.Error("the journal must exist while the update is in flight (Scenario G depends on it)")
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("by verifying, the pin must have advanced; got %q", got)
}
close(release)
st = waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseDone || st.UpdateError != "" || st.UpdatePhaseLabel != "Frissítve" {
t.Fatalf("after health the update is done: phase=%q label=%q err=%q", st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError)
}
if journalExists(m) {
t.Error("a completed update must clear its journal entry")
}
if _, err := os.Stat(filepath.Join(dir, preUpdateComposeFile)); err == nil {
t.Error("the pre-update copy must be removed after success")
}
if g.pinAtDump != "nextcloud:31.0.14-apache" {
t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump)
}
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
t.Errorf("compose calls = %q, want %q", got, want)
}
// RestorePoint twice by design: once in the preflight (the refusal), once inside the job (the
// precondition must still hold when the job actually starts).
if got := strings.Join(g.callList(), ","); got != "RestorePoint,RestorePoint,SafetyDump" {
t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got)
}
}
// ── B: the proven copy is too old ─────────────────────────────────────────────────────────────────
func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
m, dir, g, _ := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
g.rpAfterBackup = &UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Minute)}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
calls := strings.Join(g.callList(), ",")
if !strings.HasPrefix(calls, "RestorePoint,RestorePoint,BackupNow,RestorePoint,SafetyDump") {
t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("pin = %q", got)
}
}
func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour)
g.backupErr = errors.New("disk full")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want {
t.Errorf("UpdateError = %q, want the backup's own error in the sentence %q", st.UpdateError, want)
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("a failed backup must move nothing; pin = %q", got)
}
if len(c.list()) != 0 {
t.Errorf("a failed backup must reach no compose call; got %v", c.list())
}
if journalExists(m) {
t.Error("the journal must be cleared on a refusal")
}
}
func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: rpAfterBackup nil
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdateError != MsgUpdateBackupNoUnit {
t.Errorf("UpdateError = %q", st.UpdateError)
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("nothing may move when the backup did not produce a fresh restorable copy")
}
}
// ── C: no backup exists that could restore this app ───────────────────────────────────────────────
// COMPANION RED-PROOF 2 (REPORT.md): make the precondition in UpdatePreflight proceed when
// !rp.Restorable. This test then fails with the update started.
func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
// A PROVEN, FRESH copy whose unit cannot be opened — the realistic half-copied mirror. Proven and
// fresh on purpose: a fixture that is also unproven would be refused by the proven check alone,
// and a red-proof that drops the restorable check would then pass inertly (observed on the first
// run of red-proof 2, 2026-09-13).
g.rp = UpdateRestorePoint{Restorable: false, Proven: true, ProvenAt: slice4T0.Add(-time.Hour)}
err := m.StartGuardedUpdate("nextcloud")
var ref *UpdateRefusal
if !errors.As(err, &ref) || ref.Reason != "no_backup" {
t.Fatalf("an app with no restorable copy must be REFUSED (no_backup), got %v", err)
}
if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want {
t.Errorf("message = %q", ref.Message)
}
time.Sleep(50 * time.Millisecond)
if st, _ := m.GetStack("nextcloud"); st.Updating {
t.Error("a refused update must not set Updating")
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
t.Error("a refused update must move nothing")
}
// A copy that exists but was never PROVEN is not a copy (R-101).
g.rp = UpdateRestorePoint{Restorable: true, Proven: false}
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
t.Errorf("an unproven copy must refuse too, got %v", ref)
}
}
// ── D: the cheap refusals, each one ─────────────────────────────────────────────────────────────────
func TestSlice4_D_CheapRefusals(t *testing.T) {
cases := []struct {
name string
setup func(m *Manager, g *fakeGuards, dir string)
reason string
msg string
}{
{"held", func(m *Manager, g *fakeGuards, _ string) { g.held, g.holdWhy = true, "THE HOLD TEXT" }, "held", "THE HOLD TEXT"},
{"busy", func(m *Manager, g *fakeGuards, _ string) { g.busy = true }, "busy", MsgUpdateBusy},
{"already updating", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Updating = true }, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, "nextcloud")},
{"deploying", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Deploying = true }, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, "nextcloud")},
{"memory", func(m *Manager, _ *fakeGuards, _ string) {
catDir := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates", "nextcloud")
if err := os.WriteFile(filepath.Join(catDir, ".felhom.yml"), []byte("resources:\n mem_request: 900M\n"), 0o644); err != nil {
panic(err)
}
m.updateMemoryFn = func(newReq, _, _, _ int) (string, string) {
return fmt.Sprintf("Nincs elég memória (%d MB)", newReq), ""
}
}, "memory", "Nincs elég memória (900 MB)"},
{"disk", func(m *Manager, _ *fakeGuards, _ string) {
m.updateDiskFreeFn = func() (float64, bool) { return 1.0, true }
}, "disk", fmt.Sprintf(MsgUpdateDiskFmt, 1.0, updateDiskFloorGiB)},
{"guards unwired", func(m *Manager, _ *fakeGuards, _ string) { m.updateGuards = nil }, "guards_unwired", MsgUpdateNoGuards},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
tc.setup(m, g, dir)
ref := m.UpdatePreflight("nextcloud")
if ref == nil || ref.Reason != tc.reason {
t.Fatalf("want refusal %q, got %+v", tc.reason, ref)
}
if ref.Message != tc.msg {
t.Errorf("message = %q, want %q", ref.Message, tc.msg)
}
if err := m.StartGuardedUpdate("nextcloud"); err == nil {
t.Fatal("StartGuardedUpdate must refuse the same")
}
time.Sleep(20 * time.Millisecond)
if len(c.list()) != 0 || pinOf(t, dir) != "nextcloud:31.0.14-apache" {
t.Errorf("a cheap refusal reached the act: compose=%v pin=%s", c.list(), pinOf(t, dir))
}
})
}
}
// ── E: the pull fails ────────────────────────────────────────────────────────────────────────────
func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
c.fail["pull"] = errors.New("exit code 1\nstderr: manifest unknown")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdateError != MsgUpdatePullFailed {
t.Errorf("UpdateError = %q, want the Hungarian sentence and never raw stderr", st.UpdateError)
}
if strings.Contains(st.UpdateError, "manifest unknown") {
t.Error("raw Docker stderr leaked into the customer sentence")
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the pin must be PUT BACK after a failed pull, got %q", got)
}
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
t.Errorf("the live file must be re-rendered to the old version:\n%s", got)
}
if got := fileBody(t, AppliedComposePath(dir)); got != pinTplOld {
t.Errorf("the stored definition must be the old one again:\n%s", got)
}
if got := strings.Join(c.list(), " | "); got != "pull" {
t.Errorf("after a failed pull nothing else runs; compose calls = %q", got)
}
for _, call := range g.callList() {
if call == "HoldAfterFailedUpdate" {
t.Error("a failed PULL ran nothing and must not hold the app")
}
}
}
// ── F: the new version does not come up ─────────────────────────────────────────────────────────
// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test
// then fails: no hold, and the customer is not told the route back.
func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseFailed {
t.Errorf("phase = %q", st.UpdatePhase)
}
held, _ := g.HoldFor("nextcloud")
if !held {
t.Fatal("an app that did not come up must be HELD")
}
if !g.holdProvenAt.Equal(g.rp.ProvenAt) {
t.Errorf("the hold must name the PROVEN copy date %s, got %s", g.rp.ProvenAt, g.holdProvenAt)
}
if st.UpdateError != "HELD-SENTENCE" {
t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError)
}
if st.HoldReason != "HELD-SENTENCE" {
t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("the pin must STAY on the new version (its migration may have run), got %q", got)
}
if got := strings.Join(c.list(), " | "); got != "pull | up -d --remove-orphans | down" {
t.Errorf("the failed app must be stopped; compose calls = %q", got)
}
if journalExists(m) {
t.Error("the journal is cleared once the hold (the durable record) is written")
}
}
func TestSlice4_F_UnsavedHoldSaysSo(t *testing.T) {
m, _, g, _ := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
g.holdErr = errors.New("settings.json read-only")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdateError != MsgUpdateHoldUnsaved {
t.Errorf("an unrecorded hold must be disclosed, got %q", st.UpdateError)
}
}
// ── G: the controller restarts mid-update ────────────────────────────────────────────────────────
func writeTestJournal(t *testing.T, m *Manager, name string, e updateJournalEntry) {
t.Helper()
if err := m.writeUpdateJournal(updateJournal{Updates: map[string]updateJournalEntry{name: e}}); err != nil {
t.Fatal(err)
}
}
// simulateAdvanced puts the stack in the state a crash AFTER the pin moved would leave: pin, live file
// and stored definition all new, the pre-update copy on disk.
func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry {
t.Helper()
mustWrite(t, filepath.Join(dir, preUpdateComposeFile), pinTplOld)
mustWrite(t, filepath.Join(dir, preUpdateAppliedFile), pinTplOld)
if err := m.advancePinToCatalog("nextcloud", dir); err != nil {
t.Fatal(err)
}
return updateJournalEntry{
StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"},
PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile),
ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339),
}
}
func TestSlice4_G_InterruptedBeforeUpIsPutBack(t *testing.T) {
m, dir, _, c := newSlice4Manager(t)
e := simulateAdvanced(t, m, dir)
e.Phase = UpdatePhasePulling
writeTestJournal(t, m, "nextcloud", e)
if resumed := m.RecoverUpdates(); len(resumed) != 0 {
t.Fatalf("an update interrupted before `up` is not resumed, got %v", resumed)
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the pin must be put back, got %q", got)
}
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
t.Errorf("the live file must be the old definition again:\n%s", got)
}
st, _ := m.GetStack("nextcloud")
if st.Updating || st.UpdateError != MsgUpdateInterrupted {
t.Errorf("updating=%v err=%q", st.Updating, st.UpdateError)
}
if journalExists(m) || len(c.list()) != 0 {
t.Error("recovery of a pre-up interruption runs nothing and clears the journal")
}
}
func TestSlice4_G_InterruptedBeforeThePinIsDroppedUntouched(t *testing.T) {
m, dir, _, _ := newSlice4Manager(t)
writeTestJournal(t, m, "nextcloud", updateJournalEntry{Phase: UpdatePhaseSafetyDump, StartedAt: slice4T0})
m.RecoverUpdates()
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || journalExists(m) {
t.Error("an update interrupted before the pin moved must leave the pin and clear the journal")
}
}
func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
e := simulateAdvanced(t, m, dir)
e.Phase = UpdatePhaseVerifying
writeTestJournal(t, m, "nextcloud", e)
guards := m.updateGuards
m.updateGuards = nil // at RecoverUpdates time the backup side is NOT wired yet (main.go order)
resumed := m.RecoverUpdates()
if len(resumed) != 1 || !m.IsUpdating("nextcloud") || !m.UpdatingStacks()["nextcloud"] {
t.Fatalf("an update interrupted after `up` must be marked Updating for the boot sweep; resumed=%v", resumed)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("after `up` the pin is NOT put back — something may have run; got %q", got)
}
m.updateGuards = guards
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "still broken" }
if n := m.ResumeInterruptedUpdates(context.Background()); n != 1 {
t.Fatalf("resumed %d, want 1", n)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); !held || st.UpdatePhase != UpdatePhaseFailed {
t.Errorf("a resumed update that is still unhealthy must end HELD; held=%v phase=%q", held, st.UpdatePhase)
}
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | down" {
t.Errorf("resumption re-runs `up` then stops the failed app; compose calls = %q", got)
}
if !g.holdProvenAt.Equal(slice4T0.Add(-time.Hour)) {
t.Errorf("the resumed hold must name the journaled proven copy date, got %s", g.holdProvenAt)
}
}
// ── the page reads the hold from the ONE store ──────────────────────────────────────────────────
func TestSlice4_GetStacksCarriesTheHoldText(t *testing.T) {
m, _, g, _ := newSlice4Manager(t)
g.held, g.holdWhy = true, "HOLD"
for _, st := range m.GetStacks() {
if st.Name == "nextcloud" && st.HoldReason != "HOLD" {
t.Errorf("GetStacks HoldReason = %q", st.HoldReason)
}
}
g.held = false
if st, _ := m.GetStack("nextcloud"); st.HoldReason != "" {
t.Errorf("a lifted hold must disappear on the next read, got %q", st.HoldReason)
}
}
func TestSlice4_PhaseLabelsAreTheSpecifiedCopy(t *testing.T) {
want := map[string]string{
UpdatePhaseChecking: "Ellenőrzés…",
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
UpdatePhasePulling: "Új verzió letöltése…",
UpdatePhaseStarting: "Indítás az új verzióval…",
UpdatePhaseVerifying: "Működés ellenőrzése…",
UpdatePhaseDone: "Frissítve",
}
for p, l := range want {
if got := UpdatePhaseLabel(p); got != l {
t.Errorf("label(%s) = %q, want %q", p, got, l)
}
}
}