v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.
The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.
Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.
Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.
Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -4,6 +4,7 @@ import (
|
||||
"crypto/rand"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"math/big"
|
||||
@@ -236,52 +237,12 @@ func (m *Manager) DeployStack(req DeployRequest) (string, error) {
|
||||
}
|
||||
|
||||
// --- Memory validation ---
|
||||
var deployWarning string
|
||||
reservedMB := m.cfg.System.ReservedMemoryMB
|
||||
totalMB, usedMB, memErr := system.GetMemoryMB()
|
||||
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
|
||||
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
|
||||
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
|
||||
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
|
||||
// either never fire (host total) or always fire (host used > guest cap).
|
||||
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
|
||||
totalMB = gt
|
||||
memErr = nil
|
||||
}
|
||||
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
|
||||
usedMB = committedReqMB
|
||||
}
|
||||
if memErr != nil {
|
||||
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
|
||||
} else {
|
||||
usableMB := totalMB - reservedMB
|
||||
newReqMB := ParseMemoryMB(meta.Resources.MemRequest)
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
|
||||
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
|
||||
|
||||
// Hard block: committed + new request exceeds usable memory
|
||||
if newReqMB > 0 && usedMB+newReqMB > usableMB {
|
||||
clearDeploying()
|
||||
return "", fmt.Errorf(
|
||||
"Nincs elég memória az alkalmazás telepítéséhez. "+
|
||||
"Szükséges: %d MB, Elérhető: %d MB "+
|
||||
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
|
||||
newReqMB,
|
||||
usableMB-usedMB,
|
||||
totalMB,
|
||||
usedMB,
|
||||
reservedMB,
|
||||
)
|
||||
}
|
||||
|
||||
// Soft warning: limits exceed total (overcommit)
|
||||
_, currentLimitMB := m.CommittedMemory()
|
||||
newLimitMB := ParseMemoryMB(meta.Resources.MemLimit)
|
||||
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
|
||||
deployWarning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
|
||||
"Normál használat mellett ez nem okoz problémát."
|
||||
}
|
||||
// Slice 4: the block moved into memoryVerdict so the guarded update applies the SAME check with
|
||||
// the SAME wording. Behaviour here is unchanged — same inputs, same log line, same refusal text.
|
||||
refusal, deployWarning := m.memoryVerdict(ParseMemoryMB(meta.Resources.MemRequest), ParseMemoryMB(meta.Resources.MemLimit), 0, 0)
|
||||
if refusal != "" {
|
||||
clearDeploying()
|
||||
return "", errors.New(refusal)
|
||||
}
|
||||
|
||||
// Debug: log received values (redact passwords/secrets)
|
||||
@@ -1197,3 +1158,62 @@ func randomAlphanumeric(length int) (string, error) {
|
||||
}
|
||||
return string(result), nil
|
||||
}
|
||||
|
||||
// memoryVerdict is the deploy's memory check, extracted so the guarded update uses it unchanged
|
||||
// (slice 4). releasedReqMB/releasedLimitMB are what the act FREES before it takes the new amount — an
|
||||
// update replaces the app's own current request, so counting both would refuse an update that fits.
|
||||
// A deploy releases nothing and passes 0, 0.
|
||||
//
|
||||
// Returns the refusal (the deploy's own Hungarian wording, "" = admitted) and the soft overcommit
|
||||
// warning. An unreadable memory reading admits with a WARN, exactly as the deploy always has.
|
||||
func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string) {
|
||||
reservedMB := m.cfg.System.ReservedMemoryMB
|
||||
totalMB, usedMB, memErr := system.GetMemoryMB()
|
||||
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
|
||||
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
|
||||
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
|
||||
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
|
||||
// either never fire (host total) or always fire (host used > guest cap).
|
||||
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
|
||||
totalMB = gt
|
||||
memErr = nil
|
||||
}
|
||||
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
|
||||
usedMB = committedReqMB
|
||||
}
|
||||
if memErr != nil {
|
||||
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
|
||||
return "", ""
|
||||
}
|
||||
usedMB -= releasedReqMB
|
||||
if usedMB < 0 {
|
||||
usedMB = 0
|
||||
}
|
||||
usableMB := totalMB - reservedMB
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
|
||||
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
|
||||
|
||||
// Hard block: committed + new request exceeds usable memory
|
||||
if newReqMB > 0 && usedMB+newReqMB > usableMB {
|
||||
return fmt.Sprintf(
|
||||
"Nincs elég memória az alkalmazás telepítéséhez. "+
|
||||
"Szükséges: %d MB, Elérhető: %d MB "+
|
||||
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
|
||||
newReqMB,
|
||||
usableMB-usedMB,
|
||||
totalMB,
|
||||
usedMB,
|
||||
reservedMB,
|
||||
), ""
|
||||
}
|
||||
|
||||
// Soft warning: limits exceed total (overcommit)
|
||||
_, currentLimitMB := m.CommittedMemory()
|
||||
currentLimitMB -= releasedLimitMB
|
||||
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
|
||||
warning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
|
||||
"Normál használat mellett ez nem okoz problémát."
|
||||
}
|
||||
return "", warning
|
||||
}
|
||||
|
||||
@@ -407,7 +407,9 @@ func TestGroupE_RestartStackReachesTheRecorder(t *testing.T) {
|
||||
func TestGroupE_EveryBringUpPathCallsTheRecorder(t *testing.T) {
|
||||
callers := map[string]bool{} // enclosing func name -> calls recordInstalledImages
|
||||
fset := token.NewFileSet()
|
||||
for _, src := range []string{"manager.go", "deploy.go"} {
|
||||
// update.go since v0.237.0: the guarded update replaced UpdateStack, and it records in
|
||||
// verifyAndConclude — only after the app's health is known (slice 4).
|
||||
for _, src := range []string{"manager.go", "deploy.go", "update.go"} {
|
||||
f, err := parser.ParseFile(fset, src, nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -433,7 +435,7 @@ func TestGroupE_EveryBringUpPathCallsTheRecorder(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, want := range []string{"StartStack", "RestartStack", "UpdateStack", "runComposeDeploy"} {
|
||||
for _, want := range []string{"StartStack", "RestartStack", "verifyAndConclude", "runComposeDeploy"} {
|
||||
if !callers[want] {
|
||||
t.Errorf("%s does not call recordInstalledImages — a bring-up path that records nothing leaves a stale record standing", want)
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package stacks
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
@@ -130,17 +131,28 @@ type HealthCheckDetail struct {
|
||||
|
||||
// Stack represents a docker compose stack on disk.
|
||||
type Stack struct {
|
||||
Name string `json:"name"`
|
||||
Meta Metadata `json:"meta"`
|
||||
ComposePath string `json:"compose_path"`
|
||||
State ContainerState `json:"state"`
|
||||
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
|
||||
Protected bool `json:"protected"`
|
||||
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
|
||||
Containers []ContainerInfo `json:"containers"`
|
||||
AppConfig *AppConfig `json:"app_config,omitempty"`
|
||||
Deploying bool `json:"deploying"` // compose up in progress
|
||||
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
|
||||
Name string `json:"name"`
|
||||
Meta Metadata `json:"meta"`
|
||||
ComposePath string `json:"compose_path"`
|
||||
State ContainerState `json:"state"`
|
||||
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
|
||||
Protected bool `json:"protected"`
|
||||
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
|
||||
Containers []ContainerInfo `json:"containers"`
|
||||
AppConfig *AppConfig `json:"app_config,omitempty"`
|
||||
Deploying bool `json:"deploying"` // compose up in progress
|
||||
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
|
||||
// Updating / UpdatePhase / UpdatePhaseLabel / UpdateError (update arc slice 4, v0.237.0) are the
|
||||
// guarded update's in-memory progress, the same shape as Deploying/DeployError: the API answers
|
||||
// 202 at once and the page polls GET /api/stacks/{name}. See update.go.
|
||||
Updating bool `json:"updating"`
|
||||
UpdatePhase string `json:"update_phase,omitempty"`
|
||||
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
|
||||
UpdateError string `json:"update_error,omitempty"`
|
||||
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
|
||||
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
|
||||
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
|
||||
HoldReason string `json:"hold_reason,omitempty"`
|
||||
HealthProbe *HealthProbeResult `json:"health_probe,omitempty"` // controller-side probe result
|
||||
LastUpdated time.Time `json:"last_updated"`
|
||||
// RestartingSince (C9-F2) is when this stack was FIRST observed in StateRestarting during the
|
||||
@@ -198,6 +210,16 @@ type Manager struct {
|
||||
restartPolicyCache map[string]string
|
||||
// execFn replaces execCommand's process boundary in tests; nil in production.
|
||||
execFn func(name string, args ...string) (string, error)
|
||||
|
||||
// --- guarded update (slice 4, update.go) ---
|
||||
updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED
|
||||
updateComposeFn func(dir string, env []string, args ...string) (string, error)
|
||||
updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string)
|
||||
updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string)
|
||||
updateDiskFreeFn func() (freeGiB float64, ok bool)
|
||||
updateNowFn func() time.Time
|
||||
updateJournalMu sync.Mutex
|
||||
updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist
|
||||
// inspectRestartPolicyFn is the docker-inspect seam for the above; nil in production
|
||||
// (dockerRestartPolicy). Tests inject a scripted lookup and never touch docker.
|
||||
inspectRestartPolicyFn func(containerName string) (string, error)
|
||||
@@ -960,12 +982,15 @@ func aggregateState(containers []ContainerInfo, policyOf restartPolicyLookup) Co
|
||||
|
||||
func (m *Manager) GetStacks() []Stack {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
|
||||
result := make([]Stack, 0, len(m.stacks))
|
||||
for _, s := range m.stacks {
|
||||
result = append(result, deepCopyStack(s))
|
||||
}
|
||||
g := m.updateGuards
|
||||
m.mu.RUnlock()
|
||||
for i := range result {
|
||||
fillHoldReason(g, &result[i])
|
||||
}
|
||||
|
||||
// Sort alphabetically by display name for consistent UI ordering
|
||||
sort.Slice(result, func(i, j int) bool {
|
||||
@@ -984,6 +1009,7 @@ func (m *Manager) GetStack(name string) (*Stack, bool) {
|
||||
return nil, false
|
||||
}
|
||||
cp := deepCopyStack(s)
|
||||
fillHoldReason(m.updateGuards, &cp)
|
||||
return &cp, true
|
||||
}
|
||||
|
||||
@@ -1224,54 +1250,11 @@ func (m *Manager) RestartStack(name string) error {
|
||||
return m.RefreshStatus()
|
||||
}
|
||||
|
||||
func (m *Manager) UpdateStack(name string) error {
|
||||
stack, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
return fmt.Errorf("stack %q not found", name)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Updating stack: %s", name)
|
||||
start := time.Now()
|
||||
dir := filepath.Dir(stack.ComposePath)
|
||||
|
||||
// v0.235.0 — ADVANCE THE PIN FIRST, AND RE-RENDER BEFORE THE PULL.
|
||||
//
|
||||
// This is the ONE act entitled to move a version; the freeze exists so that nothing else can.
|
||||
// The ordering is load-bearing, not stylistic: `compose pull` and `up -d` act on the file on
|
||||
// disk, so the catalog's current definition has to BE that file before either runs. Setting the
|
||||
// pin afterwards would pull the frozen version and change nothing, while reporting success — and
|
||||
// a button that lies is worse than a button that refuses.
|
||||
//
|
||||
// A FAILED PIN WRITE REFUSES THE UPDATE, deliberately the opposite of recordInstalledImages.
|
||||
// That field is an observation and a failed write is a bookkeeping gap; this one is INTENT, and
|
||||
// an update whose intent could not be recorded leaves the box running a version it has no record
|
||||
// of choosing — the exact ambiguity R-166 closed for desired_state, one field over.
|
||||
if err := m.advancePinToCatalog(name, dir); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] Stack %s update refused: %v", name, err)
|
||||
return fmt.Errorf("updating stack %s: %w", name, err)
|
||||
}
|
||||
|
||||
env := m.stackEnv(dir)
|
||||
|
||||
if m.isDebug() {
|
||||
m.checkLocalImages(name, dir)
|
||||
}
|
||||
|
||||
if _, err := m.composeExecCustomEnv(dir, env, "pull"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] Stack %s update (pull) failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
|
||||
return fmt.Errorf("pulling images for %s: %w", name, err)
|
||||
}
|
||||
|
||||
if _, err := m.composeExecCustomEnv(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] Stack %s update (up) failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
|
||||
return fmt.Errorf("recreating %s: %w", name, err)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Stack %s updated successfully (took %.1fs)", name, time.Since(start).Seconds())
|
||||
m.recordInstalledImages(name, dir, env)
|
||||
m.logPostStartStatus(name, dir, env)
|
||||
return m.RefreshStatus()
|
||||
}
|
||||
// UpdateStack was REMOVED in v0.237.0 (update arc slice 4). It advanced the pin, pulled, ran `up -d`
|
||||
// and reported success on the compose exit code — no copy first, no refusals, and HTTP 200 over a
|
||||
// crash loop (R-443). Its only caller was the API, which now runs StartGuardedUpdate (update.go).
|
||||
// Deleting it rather than leaving it is deliberate: an unguarded update path that still compiles is
|
||||
// one caller away from being the next R-439.
|
||||
|
||||
func (m *Manager) GetLogs(name string, lines int) (string, error) {
|
||||
stack, ok := m.GetStack(name)
|
||||
|
||||
@@ -0,0 +1,802 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||||
)
|
||||
|
||||
// ── The guarded update (update arc slice 4, controller v0.237.0) ─────────────────────────────────
|
||||
//
|
||||
// WHAT IT REPLACED. `UpdateStack` advanced the pin, pulled, ran `up -d` and returned — no copy first,
|
||||
// no check of memory, disk, a running backup or a held app, and a success the moment `up` returned.
|
||||
// SPIKE-app-update-2026-09-01 §4 measured that as HTTP 200 over an app that was already crash-looping
|
||||
// (R-443), and R-439 is that a held app could be updated at all.
|
||||
//
|
||||
// THE SEQUENCE, and the order is the point:
|
||||
//
|
||||
// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling
|
||||
// → starting → verifying → done | failed
|
||||
//
|
||||
// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02,
|
||||
// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The
|
||||
// same predicate that permits the destructive „Teljes visszaállítás" permits the update — the
|
||||
// route back IS that restore, so an update without it has no route back.
|
||||
// 2. The safety dump is taken BEFORE the pin moves: it is "the state the customer was in a minute ago",
|
||||
// and a minute later the migration may have run.
|
||||
// 3. The pin moves BEFORE the pull (v0.235.0's reason: pull and up act on the file on disk).
|
||||
// 4. A PULL failure puts the pin BACK — nothing ran, so reverting is safe and honest (Scenario E).
|
||||
// 5. A HEALTH failure leaves the pin where it is — the new version's migration may have run, and a
|
||||
// pin claiming the old version would be a record of something untrue (Scenario F). The app is
|
||||
// HELD STOPPED and the customer is told which backup it can be restored from.
|
||||
//
|
||||
// WHAT IT DELIBERATELY DOES NOT DO: put the old version back by itself. SPIKE-upgrade-test-2026-09-06
|
||||
// measured that whether the old image starts on migrated data depends on the app (PrivateBin yes,
|
||||
// Docmost and Nextcloud no) and cannot be predicted. The route back is the restore.
|
||||
//
|
||||
// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10),
|
||||
// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at
|
||||
// the next startup (Scenario G) — the AppStopGuard pattern.
|
||||
|
||||
// Update phases, as recorded in the journal and served as Stack.UpdatePhase.
|
||||
const (
|
||||
UpdatePhaseChecking = "checking"
|
||||
UpdatePhaseBackingUp = "backing-up"
|
||||
UpdatePhaseSafetyDump = "safety-dump"
|
||||
UpdatePhasePinning = "pinning"
|
||||
UpdatePhasePulling = "pulling"
|
||||
UpdatePhaseStarting = "starting"
|
||||
UpdatePhaseVerifying = "verifying"
|
||||
UpdatePhaseDone = "done"
|
||||
UpdatePhaseFailed = "failed"
|
||||
)
|
||||
|
||||
// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the
|
||||
// specification — it is instantaneous and is the first step of fetching the new version, so it shares
|
||||
// the pull's label rather than inventing a sentence nobody would read.
|
||||
var updatePhaseLabels = map[string]string{
|
||||
UpdatePhaseChecking: "Ellenőrzés…",
|
||||
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
|
||||
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
|
||||
UpdatePhasePinning: "Új verzió letöltése…",
|
||||
UpdatePhasePulling: "Új verzió letöltése…",
|
||||
UpdatePhaseStarting: "Indítás az új verzióval…",
|
||||
UpdatePhaseVerifying: "Működés ellenőrzése…",
|
||||
UpdatePhaseDone: "Frissítve",
|
||||
UpdatePhaseFailed: "A frissítés nem sikerült",
|
||||
}
|
||||
|
||||
// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one.
|
||||
func UpdatePhaseLabel(phase string) string { return updatePhaseLabels[phase] }
|
||||
|
||||
// Customer sentences. Named so tests compare against the constant, never a retyped literal (R-364).
|
||||
const (
|
||||
MsgUpdateNoGuards = "A frissítés nem indítható: a frissítés előtti biztonsági ellenőrzés nem érhető el ezen a szerveren."
|
||||
MsgUpdateNotDeployed = "Az alkalmazás nincs telepítve, ezért nem frissíthető."
|
||||
MsgUpdateDeployingFmt = "A(z) %s telepítése még folyamatban van — a frissítés utána indítható."
|
||||
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
|
||||
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
|
||||
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
|
||||
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani. Kapcsold be a 2. mentést az alkalmazás mentési beállításainál a Mentések oldalon, és várd meg az első sikeres másolatot — utána a frissítés elindítható."
|
||||
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
|
||||
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
|
||||
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
|
||||
MsgUpdateDumpFailFmt = "A frissítés nem indult el, mert az adatbázis pillanatkép nem készült el: %v. Az alkalmazás változatlanul fut tovább."
|
||||
MsgUpdatePinFailed = "A frissítés nem indult el: az új verzió leírása nem olvasható be. Az alkalmazás változatlanul fut tovább."
|
||||
MsgUpdateJournalFailed = "A frissítés nem indult el: a frissítés naplója nem menthető. Az alkalmazás változatlanul fut tovább."
|
||||
MsgUpdatePullFailed = "Az új verzió letöltése nem sikerült, ezért a frissítés elmaradt. Az alkalmazás a korábbi verzióval fut tovább."
|
||||
MsgUpdateInterrupted = "A frissítés megszakadt, mert a vezérlő újraindult, mielőtt az új verzió elindult volna. Az alkalmazás a korábbi verzióval fut tovább."
|
||||
MsgUpdateHoldUnsaved = "A frissítés nem sikerült, az alkalmazás le lett állítva, de a leállítás rögzítése nem sikerült. Ne indítsd újra — vedd fel velünk a kapcsolatot."
|
||||
)
|
||||
|
||||
// updateDiskFloorGiB is the free space the Docker data root must have before a pull. A FIXED FLOOR,
|
||||
// stated as such: the new image set's size is not known without a registry query (the catalog
|
||||
// records tags, not sizes, and §8.1 of 09 already declines registry calls on the customer box), so
|
||||
// the rule is "not less than 2 GB", not "enough for these images".
|
||||
const updateDiskFloorGiB = 2.0
|
||||
|
||||
// updateSettleWindow is the rule for an app with no .felhom.yml health check: every container
|
||||
// running, none restarting, for this long.
|
||||
const updateSettleWindow = 60 * time.Second
|
||||
|
||||
// updatePollEvery is how often the health wait re-reads the stack.
|
||||
const updatePollEvery = 5 * time.Second
|
||||
|
||||
// UpdateRestorePoint is the precondition answer, reduced to what the update needs.
|
||||
type UpdateRestorePoint struct {
|
||||
Restorable bool // an openable recovery unit exists in the Tier-2 copy
|
||||
Proven bool // a copy actually succeeded (never an attempt clock)
|
||||
ProvenAt time.Time // when the data in that copy was last proven copied
|
||||
}
|
||||
|
||||
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
|
||||
// backup, so cmd/controller/main.go wires an adapter (TestSlice4_UpdateGuardsAreWiredAtStartup).
|
||||
type UpdateGuards interface {
|
||||
HoldFor(name string) (bool, string)
|
||||
Busy(name string) (bool, string)
|
||||
RestorePoint(name string) (UpdateRestorePoint, error)
|
||||
BackupNow(ctx context.Context, name string) error
|
||||
SafetyDump(ctx context.Context, name string) ([]string, error)
|
||||
HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error
|
||||
}
|
||||
|
||||
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
|
||||
// an update that cannot see the backup cannot promise a route back.
|
||||
func (m *Manager) SetUpdateGuards(g UpdateGuards) {
|
||||
m.mu.Lock()
|
||||
m.updateGuards = g
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
func (m *Manager) guards() UpdateGuards {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
return m.updateGuards
|
||||
}
|
||||
|
||||
func fillHoldReason(g UpdateGuards, st *Stack) {
|
||||
if g == nil || st == nil || !st.Deployed {
|
||||
return
|
||||
}
|
||||
if held, why := g.HoldFor(st.Name); held {
|
||||
st.HoldReason = why
|
||||
}
|
||||
}
|
||||
|
||||
// UpdateRefusal is a refusal taken before anything moved. Reason is a stable key for logs and tests;
|
||||
// Message is the customer sentence.
|
||||
type UpdateRefusal struct {
|
||||
Reason string
|
||||
Message string
|
||||
}
|
||||
|
||||
func (r *UpdateRefusal) Error() string { return r.Message }
|
||||
|
||||
func (m *Manager) now() time.Time {
|
||||
if m.updateNowFn != nil {
|
||||
return m.updateNowFn()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
func (m *Manager) refuseUpdate(name, reason, msg, detail string) *UpdateRefusal {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail)
|
||||
return &UpdateRefusal{Reason: reason, Message: msg}
|
||||
}
|
||||
|
||||
// UpdatePreflight runs every CHEAP refusal (slice 4 Part 1 + the precondition's existence), in order,
|
||||
// and returns the first. Nothing is moved and nothing is recorded by it. The router calls it before
|
||||
// recording the customer's intent, so an update that was never going to happen records nothing.
|
||||
func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack")
|
||||
}
|
||||
if !st.Deployed {
|
||||
return m.refuseUpdate(name, "not_deployed", MsgUpdateNotDeployed, "not deployed")
|
||||
}
|
||||
g := m.guards()
|
||||
if g == nil {
|
||||
return m.refuseUpdate(name, "guards_unwired", MsgUpdateNoGuards, "no UpdateGuards wired — fail closed")
|
||||
}
|
||||
if st.Deploying {
|
||||
return m.refuseUpdate(name, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, name), "a deploy is in progress")
|
||||
}
|
||||
if st.Updating {
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "an update is already in progress")
|
||||
}
|
||||
if held, why := g.HoldFor(name); held {
|
||||
return m.refuseUpdate(name, "held", why, "the app is held")
|
||||
}
|
||||
if busy, why := g.Busy(name); busy {
|
||||
return m.refuseUpdate(name, "busy", MsgUpdateBusy, why)
|
||||
}
|
||||
if m.IsMigrating() {
|
||||
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
|
||||
}
|
||||
rp, err := g.RestorePoint(name)
|
||||
if err != nil || !rp.Restorable || !rp.Proven {
|
||||
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
|
||||
fmt.Sprintf("no restorable proven Tier-2 unit (restorable=%v proven=%v err=%v)", rp.Restorable, rp.Proven, err))
|
||||
}
|
||||
if ref := m.updateMemoryRefusal(name, st); ref != nil {
|
||||
return ref
|
||||
}
|
||||
free, known := m.updateDiskFree()
|
||||
switch {
|
||||
case !known:
|
||||
m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB)
|
||||
case free < updateDiskFloorGiB:
|
||||
return m.refuseUpdate(name, "disk", fmt.Sprintf(MsgUpdateDiskFmt, free, updateDiskFloorGiB),
|
||||
fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// updateMemoryRefusal applies the deploy's memory check to the NEW template's request, releasing the
|
||||
// app's CURRENT request first (an update replaces it). An unknown new request proceeds with a WARN.
|
||||
func (m *Manager) updateMemoryRefusal(name string, st *Stack) *UpdateRefusal {
|
||||
catPath := m.CatalogTemplatePath(name, ".felhom.yml")
|
||||
if _, err := os.Stat(catPath); err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] update %s: the new template's memory request is unknown (%v) — proceeding without the memory check", name, err)
|
||||
return nil
|
||||
}
|
||||
newMeta := LoadMetadata(filepath.Dir(catPath))
|
||||
newReq, newLim := ParseMemoryMB(newMeta.Resources.MemRequest), ParseMemoryMB(newMeta.Resources.MemLimit)
|
||||
if newReq == 0 {
|
||||
m.logger.Printf("[WARN] [stacks] update %s: the new template declares no memory request — proceeding without the memory check", name)
|
||||
return nil
|
||||
}
|
||||
oldReq, oldLim := ParseMemoryMB(st.Meta.Resources.MemRequest), ParseMemoryMB(st.Meta.Resources.MemLimit)
|
||||
verdict := m.updateMemoryFn
|
||||
if verdict == nil {
|
||||
verdict = m.memoryVerdict
|
||||
}
|
||||
if refusal, _ := verdict(newReq, newLim, oldReq, oldLim); refusal != "" {
|
||||
return m.refuseUpdate(name, "memory", refusal, fmt.Sprintf("new_req=%dMB replacing %dMB does not fit", newReq, oldReq))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *Manager) updateDiskFree() (float64, bool) {
|
||||
if m.updateDiskFreeFn != nil {
|
||||
return m.updateDiskFreeFn()
|
||||
}
|
||||
du := system.GetDiskUsage(system.DockerVolumePath)
|
||||
if du == nil {
|
||||
return 0, false
|
||||
}
|
||||
return du.AvailGB, true
|
||||
}
|
||||
|
||||
// StartGuardedUpdate re-checks the cheap refusals, claims the Updating flag atomically and launches the
|
||||
// job. It returns as soon as the job has STARTED — the result arrives on GET /api/stacks/{name}.
|
||||
func (m *Manager) StartGuardedUpdate(name string) error {
|
||||
if ref := m.UpdatePreflight(name); ref != nil {
|
||||
return ref
|
||||
}
|
||||
m.mu.Lock()
|
||||
s, ok := m.stacks[name]
|
||||
if !ok {
|
||||
m.mu.Unlock()
|
||||
return &UpdateRefusal{Reason: "not_found", Message: fmt.Sprintf("stack %q not found", name)}
|
||||
}
|
||||
// A second press between the preflight and here is the race this lock closes.
|
||||
if s.Updating || s.Deploying {
|
||||
m.mu.Unlock()
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
|
||||
}
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
||||
m.mu.Unlock()
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] update %s: accepted — guarded update started", name)
|
||||
go m.runGuardedUpdate(context.Background(), name)
|
||||
return nil
|
||||
}
|
||||
|
||||
// IsUpdating reports whether a guarded update is in progress for the app.
|
||||
func (m *Manager) IsUpdating(name string) bool {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
s, ok := m.stacks[name]
|
||||
return ok && s.Updating
|
||||
}
|
||||
|
||||
// UpdatingStacks is the set of apps an update is currently moving — for the dead-app alarm, which
|
||||
// must not count an app the update itself is recreating (R-330's class, a third mechanism).
|
||||
func (m *Manager) UpdatingStacks() map[string]bool {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
var out map[string]bool
|
||||
for name, s := range m.stacks {
|
||||
if s.Updating {
|
||||
if out == nil {
|
||||
out = map[string]bool{}
|
||||
}
|
||||
out[name] = true
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (m *Manager) setUpdatePhase(name, phase string) {
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
||||
}
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure.
|
||||
func (m *Manager) finishUpdate(name, phase, msg string) {
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating = false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
||||
s.UpdateError = msg
|
||||
}
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) {
|
||||
if m.updateComposeFn != nil {
|
||||
return m.updateComposeFn(dir, env, args...)
|
||||
}
|
||||
return m.composeExecCustomEnv(dir, env, args...)
|
||||
}
|
||||
|
||||
func (m *Manager) updateHealth(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
||||
if m.updateHealthFn != nil {
|
||||
return m.updateHealthFn(ctx, name, timeout)
|
||||
}
|
||||
return m.waitUpdateHealthy(ctx, name, timeout)
|
||||
}
|
||||
|
||||
func (m *Manager) healthTimeout() time.Duration {
|
||||
if m.cfg == nil {
|
||||
return 5 * time.Minute
|
||||
}
|
||||
return m.cfg.Update.HealthTimeoutDuration()
|
||||
}
|
||||
|
||||
func (m *Manager) backupMaxAge() time.Duration {
|
||||
if m.cfg == nil {
|
||||
return 24 * time.Hour
|
||||
}
|
||||
return m.cfg.Update.BackupMaxAgeDuration()
|
||||
}
|
||||
|
||||
// pre-update copies, kept in the stack dir so they travel with it. Neither name is one the syncer
|
||||
// copies (it copies exactly docker-compose.yml and .felhom.yml).
|
||||
const (
|
||||
preUpdateComposeFile = "pre-update-compose.yml"
|
||||
preUpdateAppliedFile = "pre-update-applied.yml"
|
||||
)
|
||||
|
||||
func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
start := m.now()
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, fmt.Sprintf("stack %q not found", name))
|
||||
return
|
||||
}
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
g := m.guards()
|
||||
entry := updateJournalEntry{StartedAt: start}
|
||||
fail := func(msg, detail string) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail)
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
||||
}
|
||||
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateJournalFailed)
|
||||
return
|
||||
}
|
||||
if g == nil {
|
||||
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
|
||||
return
|
||||
}
|
||||
rp, err := g.RestorePoint(name)
|
||||
if err != nil || !rp.Restorable || !rp.Proven {
|
||||
fail(fmt.Sprintf(MsgUpdateNoBackupFmt, name), fmt.Sprintf("precondition vanished: restorable=%v proven=%v err=%v", rp.Restorable, rp.Proven, err))
|
||||
return
|
||||
}
|
||||
|
||||
maxAge := m.backupMaxAge()
|
||||
if age := start.Sub(rp.ProvenAt); age > maxAge {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: the proven copy is %s old (limit %s) — backing up first", name, age.Round(time.Minute), maxAge)
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
}
|
||||
if err := g.BackupNow(ctx, name); err != nil {
|
||||
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
|
||||
return
|
||||
}
|
||||
rp, err = g.RestorePoint(name)
|
||||
if err != nil || !rp.Restorable || !rp.Proven || m.now().Sub(rp.ProvenAt) > maxAge {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup: restorable=%v proven=%v at=%s err=%v", rp.Restorable, rp.Proven, rp.ProvenAt.Format(time.RFC3339), err))
|
||||
return
|
||||
}
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met — proven copy from %s (%s old, limit %s)", name, rp.ProvenAt.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge)
|
||||
}
|
||||
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
|
||||
|
||||
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
}
|
||||
paths, err := g.SafetyDump(ctx, name)
|
||||
if err != nil {
|
||||
fail(fmt.Sprintf(MsgUpdateDumpFailFmt, err), "safety dump: "+err.Error())
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
|
||||
|
||||
// PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash
|
||||
// at any later instant can put it back (Scenario G).
|
||||
prevLive, err := os.ReadFile(st.ComposePath)
|
||||
if err != nil {
|
||||
fail(MsgUpdatePinFailed, "reading the live compose file: "+err.Error())
|
||||
return
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil {
|
||||
fail(MsgUpdateJournalFailed, "saving the pre-update compose copy: "+err.Error())
|
||||
return
|
||||
}
|
||||
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
|
||||
if applied, aerr := LoadAppliedDefinition(dir); aerr == nil {
|
||||
if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil {
|
||||
entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile)
|
||||
}
|
||||
}
|
||||
if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 {
|
||||
entry.PrevPin = map[string]string{}
|
||||
for k, v := range cfg.PinnedImages {
|
||||
entry.PrevPin[k] = v
|
||||
}
|
||||
}
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) {
|
||||
m.removePreUpdateCopies(dir)
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
}
|
||||
if err := m.advancePinToCatalog(name, dir); err != nil {
|
||||
m.pinBack(name, dir, entry)
|
||||
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
env := m.stackEnv(dir)
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
|
||||
m.pinBack(name, dir, entry)
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
}
|
||||
if _, err := m.updateCompose(dir, env, "pull"); err != nil {
|
||||
// Scenario E: NOTHING RAN. The containers are the old ones and still running, so the honest
|
||||
// state is the old pin and the old file — put both back.
|
||||
m.pinBack(name, dir, entry)
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err)
|
||||
fail(MsgUpdatePullFailed, "pull failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
|
||||
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "journal write failed before up")
|
||||
return
|
||||
}
|
||||
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
// Containers may already have been recreated on the new image — something may have run.
|
||||
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "compose up failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
m.verifyAndConclude(ctx, name, dir, env, rp.ProvenAt, start, &entry)
|
||||
}
|
||||
|
||||
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
|
||||
// known, and a failure holds the app.
|
||||
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, provenAt, start time.Time, entry *updateJournalEntry) {
|
||||
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
|
||||
}
|
||||
timeout := m.healthTimeout()
|
||||
waitStart := m.now()
|
||||
healthy, detail := m.updateHealth(ctx, name, timeout)
|
||||
if !healthy {
|
||||
m.failAndHold(ctx, name, dir, env, provenAt, "not healthy: "+detail)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
|
||||
m.recordInstalledImages(name, dir, env)
|
||||
_ = m.RefreshStatus()
|
||||
m.clearJournal(name)
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.finishUpdate(name, UpdatePhaseDone, "")
|
||||
m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second))
|
||||
}
|
||||
|
||||
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
|
||||
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, provenAt time.Time, why string) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
|
||||
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
||||
}
|
||||
msg := MsgUpdateHoldUnsaved
|
||||
if g := m.guards(); g == nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
|
||||
} else if err := g.HoldAfterFailedUpdate(name, m.now(), provenAt); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
}
|
||||
_ = m.RefreshStatus()
|
||||
m.clearJournal(name)
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
||||
}
|
||||
|
||||
// pinBack restores the pin, the stored definition and the live file from the journaled copies.
|
||||
func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
|
||||
prevLive, lerr := os.ReadFile(entry.PrevCompose)
|
||||
if lerr != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr)
|
||||
}
|
||||
if len(entry.PrevPin) > 0 {
|
||||
applied := prevLive
|
||||
if entry.PrevApplied != "" {
|
||||
if b, err := os.ReadFile(entry.PrevApplied); err == nil {
|
||||
applied = b
|
||||
}
|
||||
}
|
||||
if err := m.SetPin(name, dir, entry.PrevPin, applied); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: putting the pin back failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
if lerr == nil {
|
||||
if err := os.WriteFile(ComposePathIn(dir), prevLive, 0o644); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
|
||||
}
|
||||
|
||||
func (m *Manager) removePreUpdateCopies(dir string) {
|
||||
_ = os.Remove(filepath.Join(dir, preUpdateComposeFile))
|
||||
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
|
||||
}
|
||||
|
||||
// waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the
|
||||
// existing probe, or — for an app with none — every container running and none restarting for
|
||||
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
|
||||
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
||||
deadline := m.now().Add(timeout)
|
||||
var runningSince time.Time
|
||||
last := "no observation yet"
|
||||
for {
|
||||
_ = m.RefreshStatus()
|
||||
st, ok := m.GetStack(name)
|
||||
switch {
|
||||
case !ok:
|
||||
last = "stack vanished"
|
||||
runningSince = time.Time{}
|
||||
case st.State == StateRunning:
|
||||
if hc := st.Meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
||||
if c := findProbeContainer(name, st.Containers); c != "" {
|
||||
res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.HealthProbe = res
|
||||
}
|
||||
m.mu.Unlock()
|
||||
if res.Healthy {
|
||||
return true, "the app's health check passed"
|
||||
}
|
||||
last = "health check failing"
|
||||
} else {
|
||||
last = "no probe container"
|
||||
}
|
||||
} else {
|
||||
if runningSince.IsZero() {
|
||||
runningSince = m.now()
|
||||
}
|
||||
if m.now().Sub(runningSince) >= updateSettleWindow {
|
||||
return true, fmt.Sprintf("all containers running, none restarting, for %s (no health check declared)", updateSettleWindow)
|
||||
}
|
||||
last = "running, settling"
|
||||
}
|
||||
default:
|
||||
runningSince = time.Time{}
|
||||
last = "state " + string(st.State)
|
||||
}
|
||||
if !m.now().Before(deadline) {
|
||||
return false, fmt.Sprintf("not healthy within %s (last: %s)", timeout, last)
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return false, "cancelled: " + ctx.Err().Error()
|
||||
case <-time.After(updatePollEvery):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── the journal ──────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
type updateJournalEntry struct {
|
||||
Phase string `json:"phase"`
|
||||
StartedAt time.Time `json:"started_at"`
|
||||
PrevPin map[string]string `json:"prev_pin,omitempty"`
|
||||
PrevCompose string `json:"prev_compose,omitempty"`
|
||||
PrevApplied string `json:"prev_applied,omitempty"`
|
||||
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
|
||||
}
|
||||
|
||||
type updateJournal struct {
|
||||
Updates map[string]updateJournalEntry `json:"updates"`
|
||||
}
|
||||
|
||||
func (m *Manager) updateJournalPath() string {
|
||||
return filepath.Join(m.cfg.Paths.DataDir, "update-journal.json")
|
||||
}
|
||||
|
||||
func (m *Manager) readUpdateJournal() updateJournal {
|
||||
j := updateJournal{Updates: map[string]updateJournalEntry{}}
|
||||
data, err := os.ReadFile(m.updateJournalPath())
|
||||
if err != nil {
|
||||
return j
|
||||
}
|
||||
if err := json.Unmarshal(data, &j); err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] update journal at %s is corrupt (%v) — quarantining", m.updateJournalPath(), err)
|
||||
_ = os.Rename(m.updateJournalPath(), fmt.Sprintf("%s.corrupt-%d", m.updateJournalPath(), time.Now().Unix()))
|
||||
return updateJournal{Updates: map[string]updateJournalEntry{}}
|
||||
}
|
||||
if j.Updates == nil {
|
||||
j.Updates = map[string]updateJournalEntry{}
|
||||
}
|
||||
return j
|
||||
}
|
||||
|
||||
// writeUpdateJournal is atomic and fsynced (the AppStopGuard shape): the point is surviving a power cut.
|
||||
func (m *Manager) writeUpdateJournal(j updateJournal) error {
|
||||
p := m.updateJournalPath()
|
||||
if len(j.Updates) == 0 {
|
||||
if err := os.Remove(p); err != nil && !os.IsNotExist(err) {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
data, err := json.MarshalIndent(j, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
tmp := p + ".tmp"
|
||||
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := f.Write(data); err != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := f.Sync(); err != nil {
|
||||
f.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp, p)
|
||||
}
|
||||
|
||||
// enterUpdatePhase journals the phase BEFORE it starts and mirrors it for the UI. False means the
|
||||
// journal could not be written — the caller must not perform a mutation it could not record.
|
||||
func (m *Manager) enterUpdatePhase(name string, entry *updateJournalEntry, phase string) bool {
|
||||
entry.Phase = phase
|
||||
m.updateJournalMu.Lock()
|
||||
j := m.readUpdateJournal()
|
||||
j.Updates[name] = *entry
|
||||
err := m.writeUpdateJournal(j)
|
||||
m.updateJournalMu.Unlock()
|
||||
m.setUpdatePhase(name, phase)
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: journal write for phase %s failed: %v", name, phase, err)
|
||||
return false
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: phase %s", name, phase)
|
||||
return true
|
||||
}
|
||||
|
||||
func (m *Manager) clearJournal(name string) {
|
||||
m.updateJournalMu.Lock()
|
||||
defer m.updateJournalMu.Unlock()
|
||||
j := m.readUpdateJournal()
|
||||
delete(j.Updates, name)
|
||||
if err := m.writeUpdateJournal(j); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: clearing the journal entry failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
|
||||
// RecoverUpdates reads the journal at startup (Scenario G). Call it BEFORE the boot reconciler.
|
||||
//
|
||||
// - interrupted BEFORE the pin moved (checking, backing-up, safety-dump): nothing moved — the entry is
|
||||
// dropped and the app carries the "interrupted" sentence;
|
||||
// - interrupted while pinning or pulling: nothing RAN — the pin and definition are put back (as E);
|
||||
// - interrupted while starting or verifying: something may have run — the app is marked Updating
|
||||
// (so the boot sweep and the dead-app alarm leave it alone) and queued for ResumeInterruptedUpdates,
|
||||
// which re-runs `up -d` and the health wait, ending in A or F.
|
||||
//
|
||||
// It needs no backup wiring, because nothing here holds an app — that is left to the resumed job.
|
||||
func (m *Manager) RecoverUpdates() []string {
|
||||
m.updateJournalMu.Lock()
|
||||
j := m.readUpdateJournal()
|
||||
m.updateJournalMu.Unlock()
|
||||
if len(j.Updates) == 0 {
|
||||
return nil
|
||||
}
|
||||
var resumed []string
|
||||
for name, e := range j.Updates {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s is in the journal (phase %s) but no longer exists — dropping the entry", name, e.Phase)
|
||||
m.clearJournal(name)
|
||||
continue
|
||||
}
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
switch e.Phase {
|
||||
case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
case UpdatePhasePinning, UpdatePhasePulling:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.pinBack(name, dir, e)
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
case UpdatePhaseStarting, UpdatePhaseVerifying:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
|
||||
}
|
||||
m.updateResume = append(m.updateResume, name)
|
||||
m.mu.Unlock()
|
||||
resumed = append(resumed, name)
|
||||
default:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s has unknown phase %q — dropping the entry", name, e.Phase)
|
||||
m.clearJournal(name)
|
||||
}
|
||||
}
|
||||
return resumed
|
||||
}
|
||||
|
||||
// ResumeInterruptedUpdates continues the updates RecoverUpdates queued, once the backup side is wired
|
||||
// (a resumed update that fails must be able to HOLD). Returns how many were resumed.
|
||||
func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
||||
m.mu.Lock()
|
||||
names := m.updateResume
|
||||
m.updateResume = nil
|
||||
m.mu.Unlock()
|
||||
for _, name := range names {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
continue
|
||||
}
|
||||
m.updateJournalMu.Lock()
|
||||
e, ok := m.readUpdateJournal().Updates[name]
|
||||
m.updateJournalMu.Unlock()
|
||||
if !ok {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
continue
|
||||
}
|
||||
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
go func(name, dir string, e updateJournalEntry, provenAt time.Time) {
|
||||
env := m.stackEnv(dir)
|
||||
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
|
||||
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
m.failAndHold(ctx, name, dir, env, provenAt, "resumed compose up failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
m.verifyAndConclude(ctx, name, dir, env, provenAt, e.StartedAt, &e)
|
||||
}(name, dir, e, provenAt)
|
||||
}
|
||||
return len(names)
|
||||
}
|
||||
@@ -0,0 +1,569 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Update arc slice 4 — the guarded update. Scenarios A–G of the task, through the real job
|
||||
// (runGuardedUpdate) with the four process boundaries injected: compose, the health wait, the backup
|
||||
// side (UpdateGuards) and the clock. Every assertion reads the EFFECT back — the pin in app.yaml, the
|
||||
// bytes of the live compose file, the journal on disk, Updating/UpdatePhase/UpdateError — never "no error".
|
||||
|
||||
var slice4T0 = time.Date(2026, 9, 13, 10, 0, 0, 0, time.UTC)
|
||||
|
||||
type fakeGuards struct {
|
||||
mu sync.Mutex
|
||||
calls []string
|
||||
held bool
|
||||
holdWhy string
|
||||
busy bool
|
||||
rp UpdateRestorePoint
|
||||
rpErr error
|
||||
rpAfterBackup *UpdateRestorePoint
|
||||
backupErr error
|
||||
dumpErr error
|
||||
holdErr error
|
||||
holdProvenAt time.Time
|
||||
pinAtDump string
|
||||
stackDir string
|
||||
}
|
||||
|
||||
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
|
||||
func (f *fakeGuards) callList() []string {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
return append([]string(nil), f.calls...)
|
||||
}
|
||||
func (f *fakeGuards) HoldFor(string) (bool, string) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
return f.held, f.holdWhy
|
||||
}
|
||||
func (f *fakeGuards) Busy(string) (bool, string) { return f.busy, "fake busy" }
|
||||
func (f *fakeGuards) RestorePoint(string) (UpdateRestorePoint, error) {
|
||||
f.note("RestorePoint")
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
return f.rp, f.rpErr
|
||||
}
|
||||
func (f *fakeGuards) BackupNow(context.Context, string) error {
|
||||
f.note("BackupNow")
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.backupErr == nil && f.rpAfterBackup != nil {
|
||||
f.rp = *f.rpAfterBackup
|
||||
}
|
||||
return f.backupErr
|
||||
}
|
||||
func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
|
||||
f.note("SafetyDump")
|
||||
if cfg := LoadAppConfig(f.stackDir); cfg != nil {
|
||||
f.mu.Lock()
|
||||
f.pinAtDump = cfg.PinnedImages["web"]
|
||||
f.mu.Unlock()
|
||||
}
|
||||
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
|
||||
}
|
||||
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, provenAt time.Time) error {
|
||||
f.note("HoldAfterFailedUpdate")
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.holdErr != nil {
|
||||
return f.holdErr
|
||||
}
|
||||
f.held, f.holdWhy, f.holdProvenAt = true, "HELD-SENTENCE", provenAt
|
||||
return nil
|
||||
}
|
||||
|
||||
type composeRec struct {
|
||||
mu sync.Mutex
|
||||
calls []string
|
||||
fail map[string]error // first arg → error
|
||||
}
|
||||
|
||||
func (c *composeRec) fn(_ string, _ []string, args ...string) (string, error) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.calls = append(c.calls, strings.Join(args, " "))
|
||||
if err := c.fail[args[0]]; err != nil {
|
||||
return "", err
|
||||
}
|
||||
return "", nil
|
||||
}
|
||||
func (c *composeRec) list() []string {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
return append([]string(nil), c.calls...)
|
||||
}
|
||||
|
||||
// newSlice4Manager: a pinned nextcloud on the OLD version, the catalog offering the NEW one, fresh
|
||||
// proven copy, and every boundary faked. Returns the manager, its stack dir, the guards and compose.
|
||||
func newSlice4Manager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec) {
|
||||
t.Helper()
|
||||
m, dir := newPinManager(t, pinTplOld, pinTplNew,
|
||||
"deployed: true\nenv: {}\npinned_images:\n web: nextcloud:31.0.14-apache\n")
|
||||
mustWrite(t, AppliedComposePath(dir), pinTplOld)
|
||||
g := &fakeGuards{rp: UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Hour)}, stackDir: dir}
|
||||
c := &composeRec{fail: map[string]error{}}
|
||||
m.updateGuards = g
|
||||
m.updateComposeFn = c.fn
|
||||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return true, "fake healthy" }
|
||||
m.updateMemoryFn = func(int, int, int, int) (string, string) { return "", "" }
|
||||
m.updateDiskFreeFn = func() (float64, bool) { return 50, true }
|
||||
m.updateNowFn = func() time.Time { return slice4T0 } // R-457: the SAME clock the age check reads
|
||||
m.execFn = func(string, ...string) (string, error) { return "", nil }
|
||||
return m, dir, g, c
|
||||
}
|
||||
|
||||
func waitUpdateDone(t *testing.T, m *Manager, name string) *Stack {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if st, ok := m.GetStack(name); ok && !st.Updating {
|
||||
return st
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("the update never finished")
|
||||
return nil
|
||||
}
|
||||
|
||||
func pinOf(t *testing.T, dir string) string { return readPin(t, dir).PinnedImages["web"] }
|
||||
|
||||
func fileBody(t *testing.T, p string) string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile(p)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
func journalExists(m *Manager) bool {
|
||||
_, err := os.Stat(m.updateJournalPath())
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// ── A: the happy path, and the truth ─────────────────────────────────────────────────────────────
|
||||
|
||||
// TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth. The health wait BLOCKS until the test releases it;
|
||||
// while it blocks, the app must read Updating=true / phase=verifying / no error — and the safety dump
|
||||
// must have run while the pin still named the OLD version.
|
||||
//
|
||||
// COMPANION RED-PROOF 1 (REPORT.md): delete the updateHealth call from verifyAndConclude so success is
|
||||
// declared on the compose exit code. This test then fails at "Updating went false before health was
|
||||
// known" — which is R-443 exactly.
|
||||
func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
release := make(chan struct{})
|
||||
healthCalled := make(chan struct{}, 1)
|
||||
m.updateHealthFn = func(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
||||
healthCalled <- struct{}{}
|
||||
<-release
|
||||
return true, "fake healthy"
|
||||
}
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatalf("a fully-qualified update must start: %v", err)
|
||||
}
|
||||
select {
|
||||
case <-healthCalled:
|
||||
case <-time.After(5 * time.Second):
|
||||
st, _ := m.GetStack("nextcloud")
|
||||
t.Fatalf("the health wait was never reached; state: updating=%v phase=%s err=%q", st.Updating, st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
st, _ := m.GetStack("nextcloud")
|
||||
if !st.Updating {
|
||||
t.Fatal("Updating went false before health was known — success reported on the compose exit code (R-443)")
|
||||
}
|
||||
if st.UpdatePhase != UpdatePhaseVerifying || st.UpdatePhaseLabel != "Működés ellenőrzése…" {
|
||||
t.Errorf("while waiting for health the phase must be verifying, got %q / %q", st.UpdatePhase, st.UpdatePhaseLabel)
|
||||
}
|
||||
if st.UpdateError != "" {
|
||||
t.Errorf("no error may be shown while verifying, got %q", st.UpdateError)
|
||||
}
|
||||
if !journalExists(m) {
|
||||
t.Error("the journal must exist while the update is in flight (Scenario G depends on it)")
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||||
t.Errorf("by verifying, the pin must have advanced; got %q", got)
|
||||
}
|
||||
close(release)
|
||||
st = waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdatePhase != UpdatePhaseDone || st.UpdateError != "" || st.UpdatePhaseLabel != "Frissítve" {
|
||||
t.Fatalf("after health the update is done: phase=%q label=%q err=%q", st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError)
|
||||
}
|
||||
if journalExists(m) {
|
||||
t.Error("a completed update must clear its journal entry")
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(dir, preUpdateComposeFile)); err == nil {
|
||||
t.Error("the pre-update copy must be removed after success")
|
||||
}
|
||||
if g.pinAtDump != "nextcloud:31.0.14-apache" {
|
||||
t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump)
|
||||
}
|
||||
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
|
||||
t.Errorf("compose calls = %q, want %q", got, want)
|
||||
}
|
||||
// RestorePoint twice by design: once in the preflight (the refusal), once inside the job (the
|
||||
// precondition must still hold when the job actually starts).
|
||||
if got := strings.Join(g.callList(), ","); got != "RestorePoint,RestorePoint,SafetyDump" {
|
||||
t.Errorf("a fresh copy needs no backup-first; guard calls = %s", got)
|
||||
}
|
||||
}
|
||||
|
||||
// ── B: the proven copy is too old ─────────────────────────────────────────────────────────────────
|
||||
|
||||
func TestSlice4_B_StaleCopyIsRefreshedFirst(t *testing.T) {
|
||||
m, dir, g, _ := newSlice4Manager(t)
|
||||
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // > 24 h default
|
||||
g.rpAfterBackup = &UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: slice4T0.Add(-1 * time.Minute)}
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdatePhase != UpdatePhaseDone {
|
||||
t.Fatalf("with a successful backup-first the update completes, got phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
calls := strings.Join(g.callList(), ",")
|
||||
if !strings.HasPrefix(calls, "RestorePoint,RestorePoint,BackupNow,RestorePoint,SafetyDump") {
|
||||
t.Errorf("a stale copy must be backed up FIRST and the precondition re-read; calls = %s", calls)
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||||
t.Errorf("pin = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_B_BackupFailureMovesNothing(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour)
|
||||
g.backupErr = errors.New("disk full")
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if want := fmt.Sprintf(MsgUpdateBackupFailFmt, g.backupErr); st.UpdateError != want {
|
||||
t.Errorf("UpdateError = %q, want the backup's own error in the sentence %q", st.UpdateError, want)
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||||
t.Errorf("a failed backup must move nothing; pin = %q", got)
|
||||
}
|
||||
if len(c.list()) != 0 {
|
||||
t.Errorf("a failed backup must reach no compose call; got %v", c.list())
|
||||
}
|
||||
if journalExists(m) {
|
||||
t.Error("the journal must be cleared on a refusal")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_B_BackupThatYieldsNoFreshUnitRefuses(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
g.rp.ProvenAt = slice4T0.Add(-30 * time.Hour) // stays stale: rpAfterBackup nil
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdateError != MsgUpdateBackupNoUnit {
|
||||
t.Errorf("UpdateError = %q", st.UpdateError)
|
||||
}
|
||||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
|
||||
t.Error("nothing may move when the backup did not produce a fresh restorable copy")
|
||||
}
|
||||
}
|
||||
|
||||
// ── C: no backup exists that could restore this app ───────────────────────────────────────────────
|
||||
|
||||
// COMPANION RED-PROOF 2 (REPORT.md): make the precondition in UpdatePreflight proceed when
|
||||
// !rp.Restorable. This test then fails with the update started.
|
||||
func TestSlice4_C_NoRestorableCopyRefusesBeforeAnythingMoves(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
// A PROVEN, FRESH copy whose unit cannot be opened — the realistic half-copied mirror. Proven and
|
||||
// fresh on purpose: a fixture that is also unproven would be refused by the proven check alone,
|
||||
// and a red-proof that drops the restorable check would then pass inertly (observed on the first
|
||||
// run of red-proof 2, 2026-09-13).
|
||||
g.rp = UpdateRestorePoint{Restorable: false, Proven: true, ProvenAt: slice4T0.Add(-time.Hour)}
|
||||
err := m.StartGuardedUpdate("nextcloud")
|
||||
var ref *UpdateRefusal
|
||||
if !errors.As(err, &ref) || ref.Reason != "no_backup" {
|
||||
t.Fatalf("an app with no restorable copy must be REFUSED (no_backup), got %v", err)
|
||||
}
|
||||
if want := fmt.Sprintf(MsgUpdateNoBackupFmt, "nextcloud"); ref.Message != want {
|
||||
t.Errorf("message = %q", ref.Message)
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
if st, _ := m.GetStack("nextcloud"); st.Updating {
|
||||
t.Error("a refused update must not set Updating")
|
||||
}
|
||||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 {
|
||||
t.Error("a refused update must move nothing")
|
||||
}
|
||||
// A copy that exists but was never PROVEN is not a copy (R-101).
|
||||
g.rp = UpdateRestorePoint{Restorable: true, Proven: false}
|
||||
if ref := m.UpdatePreflight("nextcloud"); ref == nil || ref.Reason != "no_backup" {
|
||||
t.Errorf("an unproven copy must refuse too, got %v", ref)
|
||||
}
|
||||
}
|
||||
|
||||
// ── D: the cheap refusals, each one ─────────────────────────────────────────────────────────────────
|
||||
|
||||
func TestSlice4_D_CheapRefusals(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
setup func(m *Manager, g *fakeGuards, dir string)
|
||||
reason string
|
||||
msg string
|
||||
}{
|
||||
{"held", func(m *Manager, g *fakeGuards, _ string) { g.held, g.holdWhy = true, "THE HOLD TEXT" }, "held", "THE HOLD TEXT"},
|
||||
{"busy", func(m *Manager, g *fakeGuards, _ string) { g.busy = true }, "busy", MsgUpdateBusy},
|
||||
{"already updating", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Updating = true }, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, "nextcloud")},
|
||||
{"deploying", func(m *Manager, _ *fakeGuards, _ string) { m.stacks["nextcloud"].Deploying = true }, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, "nextcloud")},
|
||||
{"memory", func(m *Manager, _ *fakeGuards, _ string) {
|
||||
catDir := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates", "nextcloud")
|
||||
if err := os.WriteFile(filepath.Join(catDir, ".felhom.yml"), []byte("resources:\n mem_request: 900M\n"), 0o644); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
m.updateMemoryFn = func(newReq, _, _, _ int) (string, string) {
|
||||
return fmt.Sprintf("Nincs elég memória (%d MB)", newReq), ""
|
||||
}
|
||||
}, "memory", "Nincs elég memória (900 MB)"},
|
||||
{"disk", func(m *Manager, _ *fakeGuards, _ string) {
|
||||
m.updateDiskFreeFn = func() (float64, bool) { return 1.0, true }
|
||||
}, "disk", fmt.Sprintf(MsgUpdateDiskFmt, 1.0, updateDiskFloorGiB)},
|
||||
{"guards unwired", func(m *Manager, _ *fakeGuards, _ string) { m.updateGuards = nil }, "guards_unwired", MsgUpdateNoGuards},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
tc.setup(m, g, dir)
|
||||
ref := m.UpdatePreflight("nextcloud")
|
||||
if ref == nil || ref.Reason != tc.reason {
|
||||
t.Fatalf("want refusal %q, got %+v", tc.reason, ref)
|
||||
}
|
||||
if ref.Message != tc.msg {
|
||||
t.Errorf("message = %q, want %q", ref.Message, tc.msg)
|
||||
}
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err == nil {
|
||||
t.Fatal("StartGuardedUpdate must refuse the same")
|
||||
}
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
if len(c.list()) != 0 || pinOf(t, dir) != "nextcloud:31.0.14-apache" {
|
||||
t.Errorf("a cheap refusal reached the act: compose=%v pin=%s", c.list(), pinOf(t, dir))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// ── E: the pull fails ────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
c.fail["pull"] = errors.New("exit code 1\nstderr: manifest unknown")
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdateError != MsgUpdatePullFailed {
|
||||
t.Errorf("UpdateError = %q, want the Hungarian sentence and never raw stderr", st.UpdateError)
|
||||
}
|
||||
if strings.Contains(st.UpdateError, "manifest unknown") {
|
||||
t.Error("raw Docker stderr leaked into the customer sentence")
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||||
t.Errorf("the pin must be PUT BACK after a failed pull, got %q", got)
|
||||
}
|
||||
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
|
||||
t.Errorf("the live file must be re-rendered to the old version:\n%s", got)
|
||||
}
|
||||
if got := fileBody(t, AppliedComposePath(dir)); got != pinTplOld {
|
||||
t.Errorf("the stored definition must be the old one again:\n%s", got)
|
||||
}
|
||||
if got := strings.Join(c.list(), " | "); got != "pull" {
|
||||
t.Errorf("after a failed pull nothing else runs; compose calls = %q", got)
|
||||
}
|
||||
for _, call := range g.callList() {
|
||||
if call == "HoldAfterFailedUpdate" {
|
||||
t.Error("a failed PULL ran nothing and must not hold the app")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── F: the new version does not come up ─────────────────────────────────────────────────────────
|
||||
|
||||
// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test
|
||||
// then fails: no hold, and the customer is not told the route back.
|
||||
func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdatePhase != UpdatePhaseFailed {
|
||||
t.Errorf("phase = %q", st.UpdatePhase)
|
||||
}
|
||||
held, _ := g.HoldFor("nextcloud")
|
||||
if !held {
|
||||
t.Fatal("an app that did not come up must be HELD")
|
||||
}
|
||||
if !g.holdProvenAt.Equal(g.rp.ProvenAt) {
|
||||
t.Errorf("the hold must name the PROVEN copy date %s, got %s", g.rp.ProvenAt, g.holdProvenAt)
|
||||
}
|
||||
if st.UpdateError != "HELD-SENTENCE" {
|
||||
t.Errorf("the page must carry the hold's own sentence, got %q", st.UpdateError)
|
||||
}
|
||||
if st.HoldReason != "HELD-SENTENCE" {
|
||||
t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason)
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||||
t.Errorf("the pin must STAY on the new version (its migration may have run), got %q", got)
|
||||
}
|
||||
if got := strings.Join(c.list(), " | "); got != "pull | up -d --remove-orphans | down" {
|
||||
t.Errorf("the failed app must be stopped; compose calls = %q", got)
|
||||
}
|
||||
if journalExists(m) {
|
||||
t.Error("the journal is cleared once the hold (the durable record) is written")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_F_UnsavedHoldSaysSo(t *testing.T) {
|
||||
m, _, g, _ := newSlice4Manager(t)
|
||||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
|
||||
g.holdErr = errors.New("settings.json read-only")
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdateError != MsgUpdateHoldUnsaved {
|
||||
t.Errorf("an unrecorded hold must be disclosed, got %q", st.UpdateError)
|
||||
}
|
||||
}
|
||||
|
||||
// ── G: the controller restarts mid-update ────────────────────────────────────────────────────────
|
||||
|
||||
func writeTestJournal(t *testing.T, m *Manager, name string, e updateJournalEntry) {
|
||||
t.Helper()
|
||||
if err := m.writeUpdateJournal(updateJournal{Updates: map[string]updateJournalEntry{name: e}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// simulateAdvanced puts the stack in the state a crash AFTER the pin moved would leave: pin, live file
|
||||
// and stored definition all new, the pre-update copy on disk.
|
||||
func simulateAdvanced(t *testing.T, m *Manager, dir string) updateJournalEntry {
|
||||
t.Helper()
|
||||
mustWrite(t, filepath.Join(dir, preUpdateComposeFile), pinTplOld)
|
||||
mustWrite(t, filepath.Join(dir, preUpdateAppliedFile), pinTplOld)
|
||||
if err := m.advancePinToCatalog("nextcloud", dir); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return updateJournalEntry{
|
||||
StartedAt: slice4T0, PrevPin: map[string]string{"web": "nextcloud:31.0.14-apache"},
|
||||
PrevCompose: filepath.Join(dir, preUpdateComposeFile), PrevApplied: filepath.Join(dir, preUpdateAppliedFile),
|
||||
ProvenCopyAt: slice4T0.Add(-time.Hour).Format(time.RFC3339),
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_G_InterruptedBeforeUpIsPutBack(t *testing.T) {
|
||||
m, dir, _, c := newSlice4Manager(t)
|
||||
e := simulateAdvanced(t, m, dir)
|
||||
e.Phase = UpdatePhasePulling
|
||||
writeTestJournal(t, m, "nextcloud", e)
|
||||
|
||||
if resumed := m.RecoverUpdates(); len(resumed) != 0 {
|
||||
t.Fatalf("an update interrupted before `up` is not resumed, got %v", resumed)
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
|
||||
t.Errorf("the pin must be put back, got %q", got)
|
||||
}
|
||||
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
|
||||
t.Errorf("the live file must be the old definition again:\n%s", got)
|
||||
}
|
||||
st, _ := m.GetStack("nextcloud")
|
||||
if st.Updating || st.UpdateError != MsgUpdateInterrupted {
|
||||
t.Errorf("updating=%v err=%q", st.Updating, st.UpdateError)
|
||||
}
|
||||
if journalExists(m) || len(c.list()) != 0 {
|
||||
t.Error("recovery of a pre-up interruption runs nothing and clears the journal")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_G_InterruptedBeforeThePinIsDroppedUntouched(t *testing.T) {
|
||||
m, dir, _, _ := newSlice4Manager(t)
|
||||
writeTestJournal(t, m, "nextcloud", updateJournalEntry{Phase: UpdatePhaseSafetyDump, StartedAt: slice4T0})
|
||||
m.RecoverUpdates()
|
||||
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || journalExists(m) {
|
||||
t.Error("an update interrupted before the pin moved must leave the pin and clear the journal")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_G_InterruptedAfterUpResumesTheHealthWait(t *testing.T) {
|
||||
m, dir, g, c := newSlice4Manager(t)
|
||||
e := simulateAdvanced(t, m, dir)
|
||||
e.Phase = UpdatePhaseVerifying
|
||||
writeTestJournal(t, m, "nextcloud", e)
|
||||
guards := m.updateGuards
|
||||
m.updateGuards = nil // at RecoverUpdates time the backup side is NOT wired yet (main.go order)
|
||||
|
||||
resumed := m.RecoverUpdates()
|
||||
if len(resumed) != 1 || !m.IsUpdating("nextcloud") || !m.UpdatingStacks()["nextcloud"] {
|
||||
t.Fatalf("an update interrupted after `up` must be marked Updating for the boot sweep; resumed=%v", resumed)
|
||||
}
|
||||
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
|
||||
t.Errorf("after `up` the pin is NOT put back — something may have run; got %q", got)
|
||||
}
|
||||
|
||||
m.updateGuards = guards
|
||||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "still broken" }
|
||||
if n := m.ResumeInterruptedUpdates(context.Background()); n != 1 {
|
||||
t.Fatalf("resumed %d, want 1", n)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if held, _ := g.HoldFor("nextcloud"); !held || st.UpdatePhase != UpdatePhaseFailed {
|
||||
t.Errorf("a resumed update that is still unhealthy must end HELD; held=%v phase=%q", held, st.UpdatePhase)
|
||||
}
|
||||
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans | down" {
|
||||
t.Errorf("resumption re-runs `up` then stops the failed app; compose calls = %q", got)
|
||||
}
|
||||
if !g.holdProvenAt.Equal(slice4T0.Add(-time.Hour)) {
|
||||
t.Errorf("the resumed hold must name the journaled proven copy date, got %s", g.holdProvenAt)
|
||||
}
|
||||
}
|
||||
|
||||
// ── the page reads the hold from the ONE store ──────────────────────────────────────────────────
|
||||
|
||||
func TestSlice4_GetStacksCarriesTheHoldText(t *testing.T) {
|
||||
m, _, g, _ := newSlice4Manager(t)
|
||||
g.held, g.holdWhy = true, "HOLD"
|
||||
for _, st := range m.GetStacks() {
|
||||
if st.Name == "nextcloud" && st.HoldReason != "HOLD" {
|
||||
t.Errorf("GetStacks HoldReason = %q", st.HoldReason)
|
||||
}
|
||||
}
|
||||
g.held = false
|
||||
if st, _ := m.GetStack("nextcloud"); st.HoldReason != "" {
|
||||
t.Errorf("a lifted hold must disappear on the next read, got %q", st.HoldReason)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_PhaseLabelsAreTheSpecifiedCopy(t *testing.T) {
|
||||
want := map[string]string{
|
||||
UpdatePhaseChecking: "Ellenőrzés…",
|
||||
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
|
||||
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
|
||||
UpdatePhasePulling: "Új verzió letöltése…",
|
||||
UpdatePhaseStarting: "Indítás az új verzióval…",
|
||||
UpdatePhaseVerifying: "Működés ellenőrzése…",
|
||||
UpdatePhaseDone: "Frissítve",
|
||||
}
|
||||
for p, l := range want {
|
||||
if got := UpdatePhaseLabel(p); got != l {
|
||||
t.Errorf("label(%s) = %q, want %q", p, got, l)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user