v0.263.0: a failed update puts the app back by itself (09 decision 15, R-637)
gates / gates (push) Successful in 26s
gates / gates (push) Successful in 26s
The guarded update gains a folder copy of the app's named volumes, taken after the pull where the app stops anyway (decision 19, chosen by the 2026-09-23 bake-off). On a failed health check the box undoes: every copy validated by its finished-marker first, volumes refilled, definition and pin from the job's own pre-update copies, the old version checked with the OLD .felhom.yml probe. It holds only if the undo fails, and the hold sentence says so and what state the data is in. Bind-mounted folders are never touched. - R-637 built; R-638/R-640/R-641 do not arise with a folder copy; R-639 (pre-update copies incl. .felhom.yml kept until the undo is over). - journal phases copying/undoing with power-cut recovery. - app.yaml last_update_undone + one line on the app page (hu/en). - R-642: start/restart never answer "completed". - Removal deletes kept undo copies. MinAgent unchanged (0.131.0). Nine red-proofs in REPORT.md. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -23,7 +23,7 @@ import (
|
||||
// THE SEQUENCE, and the order is the point:
|
||||
//
|
||||
// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling
|
||||
// → starting → verifying → done | failed
|
||||
// → copying → starting → verifying → done | (undoing → undone | failed)
|
||||
//
|
||||
// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02,
|
||||
// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The
|
||||
@@ -37,9 +37,11 @@ import (
|
||||
// pin claiming the old version would be a record of something untrue (Scenario F). The app is
|
||||
// HELD STOPPED and the customer is told which backup it can be restored from.
|
||||
//
|
||||
// WHAT IT DELIBERATELY DOES NOT DO: put the old version back by itself. SPIKE-upgrade-test-2026-09-06
|
||||
// measured that whether the old image starts on migrated data depends on the app (PrivateBin yes,
|
||||
// Docmost and Nextcloud no) and cannot be predicted. The route back is the restore.
|
||||
// SINCE v0.263.0 IT PUTS THE OLD VERSION BACK ITSELF — with the data from before the update (09 §3
|
||||
// decision 15, undo.go). Until then it deliberately did not: SPIKE-upgrade-test-2026-09-06 measured
|
||||
// that the old image alone refuses data a new one migrated (Docmost, Nextcloud). The undo does not put
|
||||
// the old image on migrated data; it puts the pre-migration data back as well. A HOLD now happens only
|
||||
// when that undo fails too.
|
||||
//
|
||||
// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10),
|
||||
// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at
|
||||
@@ -56,6 +58,7 @@ const (
|
||||
UpdatePhaseVerifying = "verifying"
|
||||
UpdatePhaseDone = "done"
|
||||
UpdatePhaseFailed = "failed"
|
||||
// UpdatePhaseCopying, UpdatePhaseUndoing and UpdatePhaseUndone live in undo.go.
|
||||
)
|
||||
|
||||
// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the
|
||||
@@ -71,6 +74,10 @@ var updatePhaseLabels = map[string]string{
|
||||
UpdatePhaseVerifying: "Működés ellenőrzése…",
|
||||
UpdatePhaseDone: "Frissítve",
|
||||
UpdatePhaseFailed: "A frissítés nem sikerült",
|
||||
// v0.263.0 — born as bundle keys (update.phase.*); TestUndo_PhaseLabelsMatchTheBundle pins them equal.
|
||||
UpdatePhaseCopying: "Az adatok másolása a frissítés előtt…",
|
||||
UpdatePhaseUndoing: "Visszaállítás az előző változatra…",
|
||||
UpdatePhaseUndone: "Visszaállítva az előző változatra",
|
||||
}
|
||||
|
||||
// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one.
|
||||
@@ -236,7 +243,9 @@ type UpdateGuards interface {
|
||||
CanBackUp(name string) (bool, string)
|
||||
BackupNow(ctx context.Context, name string) error
|
||||
SafetyDump(ctx context.Context, name string) ([]string, error)
|
||||
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint) error
|
||||
// HoldAfterFailedUpdate records the hold. undoState (v0.263.0) says what a FAILED undo left the data
|
||||
// as (UndoState*), "" when no undo was attempted; the hold sentence says so.
|
||||
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint, undoState string) error
|
||||
}
|
||||
|
||||
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
|
||||
@@ -654,6 +663,18 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
|
||||
|
||||
// v0.263.0 — the undo's copy is PLANNED here, before anything moves: its size against the disk
|
||||
// floor (decision 19's limit). The copy itself is taken after the pull, where the app stops anyway.
|
||||
undoVols, perr := m.planUndoCopies(name)
|
||||
if perr != nil {
|
||||
if se, ok := perr.(*undoSpaceError); ok {
|
||||
fail(undoMsg("err.stacks.update_undo_space", se.need, se.free, updateDiskFloorGiB), "undo copy: "+perr.Error())
|
||||
} else {
|
||||
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy plan: "+perr.Error())
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash
|
||||
// at any later instant can put it back (Scenario G).
|
||||
prevLive, err := os.ReadFile(st.ComposePath)
|
||||
@@ -666,6 +687,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
return
|
||||
}
|
||||
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
|
||||
// R-639: the OLD .felhom.yml too — its probe is the one the old version answers.
|
||||
if md, merr := savePreUpdateMeta(dir); merr != nil {
|
||||
m.logger.Printf("[WARN] [stacks] update %s: could not keep the previous .felhom.yml (%v) — an undo would check health with the current one", name, merr)
|
||||
} else {
|
||||
entry.PrevMeta = md
|
||||
}
|
||||
if applied, aerr := LoadAppliedDefinition(dir); aerr == nil {
|
||||
if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil {
|
||||
entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile)
|
||||
@@ -687,6 +714,9 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
|
||||
return
|
||||
}
|
||||
if cfg := LoadAppConfig(dir); cfg != nil {
|
||||
entry.NewPin = cfg.PinnedImages
|
||||
}
|
||||
|
||||
env := m.stackEnv(dir)
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
|
||||
@@ -703,13 +733,32 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
return
|
||||
}
|
||||
|
||||
// COPYING (v0.263.0) — the app is stopped here anyway to be recreated, so the extra downtime is the
|
||||
// copy alone. A failed copy moves nothing: the copies go, the pin goes back, the old containers
|
||||
// start again.
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) {
|
||||
m.pinBack(name, dir, entry)
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
}
|
||||
if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy failed: %v — removing it, putting the pin back and starting the previous version", name, err)
|
||||
m.removeUndoCopies(name, entry.UndoCopies)
|
||||
m.pinBack(name, dir, entry)
|
||||
if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr)
|
||||
}
|
||||
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
|
||||
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up")
|
||||
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up", &entry)
|
||||
return
|
||||
}
|
||||
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
// Containers may already have been recreated on the new image — something may have run.
|
||||
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error())
|
||||
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error(), &entry)
|
||||
return
|
||||
}
|
||||
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
|
||||
@@ -725,12 +774,14 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
|
||||
waitStart := m.now()
|
||||
healthy, detail := m.updateHealth(ctx, name, timeout)
|
||||
if !healthy {
|
||||
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail)
|
||||
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail, entry)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
|
||||
m.recordInstalledImages(name, dir, env)
|
||||
_ = m.RefreshStatus()
|
||||
m.removeUndoCopies(name, entry.UndoCopies)
|
||||
m.recordUpdateUndone(name, dir, nil) // a successful update ends the "undone" note
|
||||
m.clearJournal(name)
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.finishUpdate(name, UpdatePhaseDone, "")
|
||||
@@ -762,22 +813,34 @@ func (m *Manager) captureHoldLogs(name, dir string, env []string) {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: kept %d bytes of the app's own log at %s before stopping it (R-621)", name, len(out), path)
|
||||
}
|
||||
|
||||
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
|
||||
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
|
||||
// failAndHold is Scenario F. Since v0.263.0 it first UNDOES (decision 15): when the job took its
|
||||
// last-second copy, the old version goes back with the data from before the update, and the app is
|
||||
// HELD only when that undo fails too. Without a copy (an update journaled by an older controller,
|
||||
// resumed after an upgrade) it holds as it always did.
|
||||
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string, entry *updateJournalEntry) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s", name, why)
|
||||
// R-621: capture the app's own logs BEFORE the `down`, because the `down` destroys them. Two
|
||||
// drill nights lost the only evidence of WHY an update failed this way — `adventurelog` ran nine
|
||||
// migrations and then never bound its port, and the log that would have said so was gone by the
|
||||
// time anyone looked. The capture is bounded and best-effort: a hold must never fail because its
|
||||
// evidence could not be written.
|
||||
m.captureHoldLogs(name, dir, env)
|
||||
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
||||
undoState := ""
|
||||
if entry != nil && entry.Copied {
|
||||
if undoState = m.tryUndo(ctx, name, dir, why, entry); undoState == "" {
|
||||
return // undone: the previous version runs on the pre-update data
|
||||
}
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: the UNDO failed too (%s) — HOLDING the app; the undo copies are kept: %v", name, undoState, entry.UndoCopies)
|
||||
} else {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: no undo copy for this update — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name)
|
||||
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
msg := MsgUpdateHoldUnsaved
|
||||
if g := m.guards(); g == nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
|
||||
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp); err != nil {
|
||||
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
@@ -789,8 +852,17 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
|
||||
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
||||
}
|
||||
|
||||
// pinBack restores the pin, the stored definition and the live file from the journaled copies.
|
||||
// pinBack restores the pin, the stored definition and the live file from the journaled copies, and
|
||||
// then removes the copies.
|
||||
func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
|
||||
m.restoreDefinition(name, dir, entry)
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
|
||||
}
|
||||
|
||||
// restoreDefinition is pinBack WITHOUT removing the copies — the undo's form, so a power cut after it
|
||||
// can run it again (RecoverUpdates → undoing) and find the copies still there.
|
||||
func (m *Manager) restoreDefinition(name, dir string, entry updateJournalEntry) {
|
||||
prevLive, lerr := os.ReadFile(entry.PrevCompose)
|
||||
if lerr != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr)
|
||||
@@ -811,13 +883,12 @@ func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
|
||||
}
|
||||
|
||||
func (m *Manager) removePreUpdateCopies(dir string) {
|
||||
_ = os.Remove(filepath.Join(dir, preUpdateComposeFile))
|
||||
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
|
||||
_ = os.RemoveAll(filepath.Join(dir, preUpdateMetaDir))
|
||||
}
|
||||
|
||||
// settleReason names WHY the wait fell back to container state, so the journal and the log do not
|
||||
@@ -834,6 +905,12 @@ func settleReason(meta Metadata) string {
|
||||
// existing probe, or — for an app with none — every container running and none restarting for
|
||||
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
|
||||
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
||||
return m.waitUpdateHealthyMeta(ctx, name, timeout, nil)
|
||||
}
|
||||
|
||||
// waitUpdateHealthyMeta is the health wait with the probe taken from `override` instead of the app's
|
||||
// current .felhom.yml — the undo's form (the old version is judged by the old probe). nil = current.
|
||||
func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeout time.Duration, override *Metadata) (bool, string) {
|
||||
deadline := m.now().Add(timeout)
|
||||
var runningSince time.Time
|
||||
warnedNoProbe := false
|
||||
@@ -857,8 +934,12 @@ func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout ti
|
||||
// 404 afterwards. A stack with no probe is not "healthy" and it is not "failing" — it is
|
||||
// SETTLED ON CONTAINER STATE (`09` §3), and never a reason to stop a running app.
|
||||
usable := false
|
||||
if hc := st.Meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
||||
c, candidates := findProbeContainerMeta(name, &st.Meta, st.Containers)
|
||||
meta := st.Meta
|
||||
if override != nil {
|
||||
meta = *override
|
||||
}
|
||||
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
||||
c, candidates := findProbeContainerMeta(name, &meta, st.Containers)
|
||||
if c != "" {
|
||||
usable = true
|
||||
res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
||||
@@ -883,7 +964,7 @@ func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout ti
|
||||
}
|
||||
if m.now().Sub(runningSince) >= updateSettleWindow {
|
||||
return true, fmt.Sprintf("all containers running, none restarting, for %s (%s)",
|
||||
updateSettleWindow, settleReason(st.Meta))
|
||||
updateSettleWindow, settleReason(meta))
|
||||
}
|
||||
last = "running, settling"
|
||||
}
|
||||
@@ -914,6 +995,13 @@ type updateJournalEntry struct {
|
||||
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
|
||||
// the right copy. 0 in a journal written by v0.238.1 or older.
|
||||
ProvenTier int `json:"proven_tier,omitempty"`
|
||||
// v0.263.0 — the undo (undo.go). UndoCopies is journaled BEFORE each copy starts; Copied is true
|
||||
// only once every copy finished (it may be true with zero copies: an app with no named volume).
|
||||
// PrevMeta is the directory holding the previous .felhom.yml; NewPin the pin the update moved to.
|
||||
UndoCopies []undoCopy `json:"undo_copies,omitempty"`
|
||||
Copied bool `json:"copied,omitempty"`
|
||||
PrevMeta string `json:"prev_meta,omitempty"`
|
||||
NewPin map[string]string `json:"new_pin,omitempty"`
|
||||
}
|
||||
|
||||
type updateJournal struct {
|
||||
@@ -1043,6 +1131,30 @@ func (m *Manager) RecoverUpdates() []string {
|
||||
m.pinBack(name, dir, e)
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
case UpdatePhaseCopying:
|
||||
// v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the
|
||||
// pin goes back, and the previous version is started again.
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while copying its data (started %s) — nothing new ran; removing the partial copy, putting the pin back and starting the previous version", name, e.StartedAt.Format(time.RFC3339))
|
||||
m.removeUndoCopies(name, e.UndoCopies)
|
||||
m.pinBack(name, dir, e)
|
||||
if _, err := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err)
|
||||
}
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
case UpdatePhaseUndoing:
|
||||
// v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from
|
||||
// the copies (still there: they are removed only after the undo succeeded) and then probes.
|
||||
// Never "done": what ran last was a failed new version.
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while UNDOING (started %s) — marking it Updating and RESUMING the undo", name, e.StartedAt.Format(time.RFC3339))
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseUndoing, UpdatePhaseLabel(UpdatePhaseUndoing)
|
||||
}
|
||||
m.updateResume = append(m.updateResume, name)
|
||||
m.mu.Unlock()
|
||||
resumed = append(resumed, name)
|
||||
case UpdatePhaseStarting, UpdatePhaseVerifying:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.mu.Lock()
|
||||
@@ -1086,9 +1198,14 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
|
||||
env := m.stackEnv(dir)
|
||||
if e.Phase == UpdatePhaseUndoing {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: resuming the UNDO after a controller restart", name)
|
||||
m.failAndHold(ctx, name, dir, env, rp, "resumed after a restart during the undo", &e)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
|
||||
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error())
|
||||
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error(), &e)
|
||||
return
|
||||
}
|
||||
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
|
||||
|
||||
Reference in New Issue
Block a user