v0.263.0: a failed update puts the app back by itself (09 decision 15, R-637)
gates / gates (push) Successful in 26s

The guarded update gains a folder copy of the app's named volumes, taken
after the pull where the app stops anyway (decision 19, chosen by the
2026-09-23 bake-off). On a failed health check the box undoes: every copy
validated by its finished-marker first, volumes refilled, definition and pin
from the job's own pre-update copies, the old version checked with the OLD
.felhom.yml probe. It holds only if the undo fails, and the hold sentence
says so and what state the data is in. Bind-mounted folders are never
touched.

- R-637 built; R-638/R-640/R-641 do not arise with a folder copy; R-639
  (pre-update copies incl. .felhom.yml kept until the undo is over).
- journal phases copying/undoing with power-cut recovery.
- app.yaml last_update_undone + one line on the app page (hu/en).
- R-642: start/restart never answer "completed".
- Removal deletes kept undo copies.

MinAgent unchanged (0.131.0). Nine red-proofs in REPORT.md.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-23 11:12:49 +02:00
parent b9deec1907
commit 8fc2b4a1a9
26 changed files with 1326 additions and 69 deletions
@@ -0,0 +1,50 @@
package api
import (
"go/ast"
"go/parser"
"go/token"
"strings"
"testing"
)
// R-642 (v0.263.0): a start or restart is never answered "completed" — measured 2026-09-23 on docmost
// and romm, both 200 "start completed" while `Restarting (1)` behind a 404 front door.
//
// COMPANION RED-PROOF (REPORT.md): put `Message: "Stack " + name + " " + action + " completed"` back in
// startAnswer — the first assertion fails.
func TestR642_StartIsNeverReportedCompleted(t *testing.T) {
for _, action := range []string{"start", "restart"} {
a := startAnswer("docmost", action, "restarting")
if strings.Contains(a.Message, "completed") {
t.Errorf("%s answered %q — a start cannot know it completed", action, a.Message)
}
if !strings.Contains(a.Message, "restarting") {
t.Errorf("%s must say the state it sees, got %q", action, a.Message)
}
if d, _ := a.Data.(map[string]interface{}); d["state"] != "restarting" {
t.Errorf("%s data = %v", action, a.Data)
}
}
}
// TestR642_TheHandlerUsesStartAnswer walks the router's AST: the stack-action handler must CALL
// startAnswer (a comment naming it would not count — the v0.154.0 / R-106 inert-seam class).
func TestR642_TheHandlerUsesStartAnswer(t *testing.T) {
f, err := parser.ParseFile(token.NewFileSet(), "router.go", nil, 0)
if err != nil {
t.Fatal(err)
}
calls := 0
ast.Inspect(f, func(n ast.Node) bool {
if c, ok := n.(*ast.CallExpr); ok {
if id, ok := c.Fun.(*ast.Ident); ok && id.Name == "startAnswer" {
calls++
}
}
return true
})
if calls != 1 {
t.Errorf("startAnswer must be called exactly once from router.go, found %d", calls)
}
}
+23 -1
View File
@@ -731,7 +731,21 @@ func (r *Router) actionStack(w http.ResponseWriter, req *http.Request, action, n
Data: map[string]interface{}{"accepted": true, "completed": false}})
return
}
writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Stack " + name + " " + action + " completed"})
// R-642 (v0.263.0): a start or restart is NEVER reported "completed". `compose up -d` returns 0
// over an app that is about to crash-loop — measured 2026-09-23 on docmost and romm, each answering
// 200 "start completed" while `Restarting (1)` behind a 404 front door. What this knows at return is
// the state the containers are in right now, so that is what it says; whether the APP works is
// the health probe's to answer, not this line's. Pinned by TestR642_StartIsNeverReportedCompleted.
if action == "start" || action == "restart" {
state := ""
_ = r.stackMgr.RefreshStatus()
if st, ok := r.stackMgr.GetStack(name); ok {
state = string(st.State)
}
writeJSON(w, http.StatusOK, startAnswer(name, action, state))
} else {
writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Stack " + name + " " + action + " completed"})
}
// Trigger integration lifecycle hooks after successful action
if r.integrationMgr != nil {
@@ -1505,3 +1519,11 @@ func writeJSON(w http.ResponseWriter, status int, v interface{}) {
func limitBody(w http.ResponseWriter, req *http.Request) {
req.Body = http.MaxBytesReader(w, req.Body, 1<<20) // 1MB
}
// startAnswer is R-642's answer for a start or restart: what was requested and the state the
// containers are in right now — never "completed", which nothing at this point can know.
func startAnswer(name, action, state string) apiResponse {
return apiResponse{OK: true,
Message: "Stack " + name + " " + action + " requested — state now: " + state,
Data: map[string]interface{}{"state": state}}
}
@@ -49,12 +49,14 @@ func (g *apiFakeGuards) RestorePoints(_ context.Context, _ string, accept func(s
}
return stacks.UpdateRestorePoint{}, false, g.points
}
func (g *apiFakeGuards) CanBackUp(string) (bool, string) { return !g.cannotBackUp, "fake: no drive" }
func (g *apiFakeGuards) CanBackUp(string) (bool, string) { return !g.cannotBackUp, "fake: no drive" }
func (g *apiFakeGuards) BackupNow(context.Context, string) error { return nil }
func (g *apiFakeGuards) SafetyDump(context.Context, string) ([]string, error) {
return nil, nil
}
func (g *apiFakeGuards) HoldAfterFailedUpdate(string, time.Time, stacks.UpdateRestorePoint) error { return nil }
func (g *apiFakeGuards) HoldAfterFailedUpdate(string, time.Time, stacks.UpdateRestorePoint, string) error {
return nil
}
const slice4AppYAML = "deployed: true\nenv: {}\npinned_images:\n app: nginx:1.27\n"
@@ -113,7 +115,7 @@ func postUpdate(t *testing.T, r *Router) (int, apiResponse) {
func TestR439_UpdateOfAHeldAppIsRefused(t *testing.T) {
r, sett, g, dir := newSlice4Router(t)
g.points, g.cannotBackUp = nil, true // the preflight's own refusal would say "no backup" — not the hold
g.blindToHolds = true // only the router's line can produce the hold's sentence
g.blindToHolds = true // only the router's line can produce the hold's sentence
if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "app", At: "2026-09-13T08:00:00Z", Reason: settings.HoldReasonUpdateFailed, CopyDate: "2026-09-13T01:30:00Z"}); err != nil {
t.Fatal(err)
}
@@ -340,18 +340,7 @@ func (m *Manager) RestoreHoldFor(stack string) (bool, string) {
// Slice 4: one storage, two reasons. An update hold names the copy it can be restored from; a
// restore hold names nothing, because the restore it refers to already consumed the copy.
if h.Reason == settings.HoldReasonUpdateFailed {
copyDate := m.note("note.reconstitute.copy_latest")
if h.CopyDate != "" {
copyDate = fmtHoldTime(h.CopyDate)
}
// R-475: name the tier when the hold recorded one; an older hold keeps its own sentence.
if label := UpdateTierLabel(h.CopyTier); label != "" && h.CopyDate != "" {
if h.CopyHolds != "" { // R-479: name what the copy holds
return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), label, copyDate, h.CopyHolds)
}
return true, fmt.Sprintf(UpdateHoldTierFmt, stack, fmtHoldTime(h.At), label, copyDate)
}
return true, fmt.Sprintf(UpdateHoldLegacyFmt, stack, fmtHoldTime(h.At), copyDate)
return true, m.undoHoldPrefix(h.UndoState) + m.updateHoldSentence(stack, h)
}
when := h.At
if t, err := time.Parse(time.RFC3339, h.At); err == nil {
@@ -360,6 +349,36 @@ func (m *Manager) RestoreHoldFor(stack string) (bool, string) {
return true, m.note("note.reconstitute.held", stack, when)
}
// undoHoldPrefix (v0.263.0) opens the hold sentence when the box already TRIED to undo the update and
// that failed too: what was tried, then what state the data is in. "" when no undo was attempted, so
// every hold written before v0.263.0 reads exactly as it did.
func (m *Manager) undoHoldPrefix(state string) string {
switch state {
case "untouched", "half", "not_started":
return m.note("hold.update.undo_failed") + " " + m.note("hold.update.undo_state."+state) + " "
case "":
return ""
}
m.logger.Printf("[WARN] [backup] unknown undo state %q on a hold — rendering the plain prefix", state)
return m.note("hold.update.undo_failed") + " "
}
// updateHoldSentence is the update hold's own sentence (slice 4, R-475, R-479), unchanged.
func (m *Manager) updateHoldSentence(stack string, h settings.RestoreHold) string {
copyDate := m.note("note.reconstitute.copy_latest")
if h.CopyDate != "" {
copyDate = fmtHoldTime(h.CopyDate)
}
// R-475: name the tier when the hold recorded one; an older hold keeps its own sentence.
if label := UpdateTierLabel(h.CopyTier); label != "" && h.CopyDate != "" {
if h.CopyHolds != "" { // R-479: name what the copy holds
return fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), label, copyDate, h.CopyHolds)
}
return fmt.Sprintf(UpdateHoldTierFmt, stack, fmtHoldTime(h.At), label, copyDate)
}
return fmt.Sprintf(UpdateHoldLegacyFmt, stack, fmtHoldTime(h.At), copyDate)
}
// holdAppAfterFailedRollback records the R-379/R-380 hold and makes sure nothing restarts the app
// behind our back.
//
@@ -74,7 +74,7 @@ func TestR479_HoldSentenceNamesWhatTheCopyHolds(t *testing.T) {
if h := m.UpdateCopyHolds("app", UpdateTierOffsite); !strings.Contains(h, "a fájlokat") || strings.Contains(h, "csak") {
t.Errorf("off-site holds the files too, got %q", h)
}
if err := m.HoldAfterFailedUpdateHolding("app", at, copyAt, UpdateTierLocal, holds); err != nil {
if err := m.HoldAfterFailedUpdateHolding("app", at, copyAt, UpdateTierLocal, holds, ""); err != nil {
t.Fatal(err)
}
_, why := m.RestoreHoldFor("app")
@@ -0,0 +1,58 @@
package backup
import (
"fmt"
"io"
"log"
"strings"
"testing"
"time"
)
// v0.263.0 — a hold after a FAILED UNDO opens with what was tried and what state the data is in, in the
// box's language; a hold with no undo attempted reads exactly as before (every pre-v0.263.0 hold).
//
// COMPANION RED-PROOF (REPORT.md): make RestoreHoldFor ignore h.UndoState (drop undoHoldPrefix) — the
// three undo cases then read the plain sentence and this test fails on the prefix.
func TestUndo_HoldSentenceSaysTheUndoWasTriedAndTheDataState(t *testing.T) {
at := time.Date(2026, 9, 23, 8, 0, 0, 0, time.UTC)
copyAt := time.Date(2026, 9, 23, 1, 30, 0, 0, time.UTC)
m := r479Manager(t, false)
m.logger = log.New(io.Discard, "", 0)
plain := fmt.Sprintf(UpdateHoldTierFmt, "app", "2026-09-23 10:00", "második meghajtó", "2026-09-23 03:30")
if err := m.HoldAfterFailedUpdateHolding("app", at, copyAt, UpdateTierSecondDrive, "", ""); err != nil {
t.Fatal(err)
}
if _, why := m.RestoreHoldFor("app"); why != plain {
t.Errorf("no undo attempted: the sentence must be unchanged, got %q", why)
}
for state, clause := range map[string]string{
"untouched": "Az adatok az új változat által hagyott állapotban vannak.",
"half": "Az adatok visszamásolása félbeszakadt",
"not_started": "Az adatok a frissítés előtti állapotba kerültek vissza, de az előző változat nem indult el.",
} {
if err := m.HoldAfterFailedUpdateHolding("app", at, copyAt, UpdateTierSecondDrive, "", state); err != nil {
t.Fatal(err)
}
_, why := m.RestoreHoldFor("app")
if !strings.HasPrefix(why, "A frissítés nem sikerült, és az automatikus visszaállítás sem. "+clause) {
t.Errorf("%s: the hold must open with the undo prefix and its state, got %q", state, why)
}
if !strings.HasSuffix(why, plain) {
t.Errorf("%s: the hold must still name the copy to restore from, got %q", state, why)
}
}
if err := m.settings.SetLanguage("en"); err != nil {
t.Fatal(err)
}
_, why := m.RestoreHoldFor("app")
if !strings.HasPrefix(why, "The update did not succeed, and the automatic undo did not either. The data is back as it was before the update") {
t.Errorf("an English box gets the English prefix, got %q", why)
}
if strings.Contains(why, "automatikus visszaállítás") {
t.Errorf("the Hungarian prefix must be GONE on an English box, got %q", why)
}
}
+11 -6
View File
@@ -505,20 +505,25 @@ func fmtHoldTime(rfc3339 string) string {
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
// failure on the page.
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error {
return m.HoldAfterFailedUpdateHolding(stackName, at, copyDate, copyTier, "")
return m.HoldAfterFailedUpdateHolding(stackName, at, copyDate, copyTier, "", "")
}
// HoldAfterFailedUpdateHolding is HoldAfterFailedUpdate with the R-479 phrase for what the copy holds;
// "" records none (the tier-only sentence). The adapter in main.go computes the phrase with
// UpdateCopyHolds at hold time.
func (m *Manager) HoldAfterFailedUpdateHolding(stackName string, at time.Time, copyDate time.Time, copyTier int, copyHolds string) error {
//
// undoState (v0.263.0) is what a FAILED undo left the data as (stacks.UndoState*), "" when no undo was
// attempted. RestoreHoldFor puts it in front of the sentence, so the household reads that the box
// already tried to put the app back, and in what state that left the data.
func (m *Manager) HoldAfterFailedUpdateHolding(stackName string, at time.Time, copyDate time.Time, copyTier int, copyHolds, undoState string) error {
if m == nil || m.settings == nil {
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
}
h := settings.RestoreHold{
Stack: stackName,
At: at.UTC().Format(time.RFC3339),
Reason: settings.HoldReasonUpdateFailed,
Stack: stackName,
At: at.UTC().Format(time.RFC3339),
Reason: settings.HoldReasonUpdateFailed,
UndoState: undoState,
}
if !copyDate.IsZero() {
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
@@ -528,7 +533,7 @@ func (m *Manager) HoldAfterFailedUpdateHolding(stackName string, at time.Time, c
if err := m.settings.SetRestoreHold(h); err != nil {
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
}
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s; holds: %q)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate, h.CopyHolds)
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s; holds: %q; undo: %q)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate, h.CopyHolds, h.UndoState)
return nil
}
+11 -1
View File
@@ -156,6 +156,7 @@
"app_info.regi_adatok_torlese": "Deleting old data",
"app_info.telepites": "Install",
"app_info.ujratelepites": "Reinstalling",
"app_info.update_undone": "The update of %s at %s did not succeed. The box put back the previous version and its data automatically — nothing was lost.",
"app_info.valassz_celtarhelyet": "Choose a target storage…",
"app_info.valassz_celtarhelyet_2": "Choose a target storage.",
"backup.contents.data": "Data",
@@ -1318,6 +1319,8 @@
"err.stacks.ujratelepites_sikertelen": "the reinstall failed (%s): %s",
"err.stacks.update_downgrade": "This version is newer than the one in the catalog — moving back needs the operator.",
"err.stacks.update_self_updating": "The controller is updating. Try again in a few minutes.",
"err.stacks.update_undo_copy_failed": "The update did not start, because the copy of the data taken before an update could not be made. The app keeps running on its previous version.",
"err.stacks.update_undo_space": "The update did not start: there is not enough free space for the copy of the data taken before an update (%.1f GB needed, %.1f GB free, and at least %.0f GB must stay free). The app keeps running unchanged.",
"err.stacks.utkozes_a_celtarolon_mar_letezik_ezeknek": "clash on the target storage — these apps already have data there: %s",
"err.web.a_formazas_allapota_nem_kerdezheto_le": "the format status cannot be read: %s",
"err.web.a_formazas_nem_indult_el_az": "the format did not start on the device (%s)",
@@ -1474,6 +1477,10 @@
"func.time.tomorrow_at": "tomorrow %s",
"func.time.yesterday": "yesterday",
"health.no_probe_container": "No health check ran: no matching container.",
"hold.update.undo_failed": "The update did not succeed, and the automatic undo did not either.",
"hold.update.undo_state.half": "Putting the data back stopped part-way, so the data is in a mixed state; the copy taken before the update is kept.",
"hold.update.undo_state.not_started": "The data is back as it was before the update, but the previous version did not start.",
"hold.update.undo_state.untouched": "The data is as the new version left it.",
"launcher.a_megosztas_jelenleg_ki_van": "Sharing is off.",
"launcher.a_megosztas_jelszoval_vedett": "The share is protected by a password.",
"launcher.alkalmazasok_telepitese": "Install apps",
@@ -2336,5 +2343,8 @@
"tier2_config.mentes": "Save",
"tier2_config.nincs_elerheto_off_drive_cel": "No off-drive target available",
"tier2_config.nincs_masik_adatmeghajto_automatikus_cel": "No other data drive — the automatic target is the internal SSD (database/configuration only). Add a 2nd\n data drive to back up all data off-drive too.",
"tier2_config.vissza_a_mentesekhez": "← Back to backups"
"tier2_config.vissza_a_mentesekhez": "← Back to backups",
"update.phase.copying": "Copying the data before the update…",
"update.phase.undoing": "Putting the previous version back…",
"update.phase.undone": "Put back to the previous version"
}
+11 -1
View File
@@ -152,6 +152,7 @@
"app_info.regi_adatok_torlese": "Régi adatok törlése",
"app_info.telepites": "Telepítés",
"app_info.ujratelepites": "Újratelepítés",
"app_info.update_undone": "A(z) %s frissítése %s-kor nem sikerült. A doboz automatikusan visszaállította az előző változatot és az adatokat — semmi nem veszett el.",
"app_info.valassz_celtarhelyet": "Válassz céltárhelyet…",
"app_info.valassz_celtarhelyet_2": "Válassz céltárhelyet.",
"backup.contents.data": "Adatok",
@@ -1309,6 +1310,8 @@
"err.stacks.ujratelepites_sikertelen": "újratelepítés sikertelen (%s): %s",
"err.stacks.update_downgrade": "Ez a változat újabb a katalógusban lévőnél — visszalépés csak az üzemeltető kérésére.",
"err.stacks.update_self_updating": "A vezérlő éppen frissül. Próbáld újra néhány perc múlva.",
"err.stacks.update_undo_copy_failed": "A frissítés nem indult el, mert a frissítés előtti adatmásolat nem készült el. Az alkalmazás a korábbi verzióval fut tovább.",
"err.stacks.update_undo_space": "A frissítés nem indult el: nincs elég szabad hely a frissítés előtti adatmásolathoz (%.1f GB kell, %.1f GB szabad, és legalább %.0f GB-nak szabadon kell maradnia). Az alkalmazás változatlanul fut tovább.",
"err.stacks.utkozes_a_celtarolon_mar_letezik_ezeknek": "ütközés a céltárolón — már létezik ezeknek az alkalmazásoknak az adata: %s",
"err.web.a_formazas_allapota_nem_kerdezheto_le": "a formázás állapota nem kérdezhető le: %s",
"err.web.a_formazas_nem_indult_el_az": "a formázás nem indult el az eszközön (%s)",
@@ -1462,6 +1465,10 @@
"func.time.tomorrow_at": "holnap %s",
"func.time.yesterday": "tegnap",
"health.no_probe_container": "Nem futott egészségellenőrzés: nincs hozzá tartozó konténer.",
"hold.update.undo_failed": "A frissítés nem sikerült, és az automatikus visszaállítás sem.",
"hold.update.undo_state.half": "Az adatok visszamásolása félbeszakadt, ezért az adatok vegyes állapotban vannak; a frissítés előtti másolat megmaradt.",
"hold.update.undo_state.not_started": "Az adatok a frissítés előtti állapotba kerültek vissza, de az előző változat nem indult el.",
"hold.update.undo_state.untouched": "Az adatok az új változat által hagyott állapotban vannak.",
"launcher.a_megosztas_jelenleg_ki_van": "A megosztás jelenleg ki van kapcsolva.",
"launcher.a_megosztas_jelszoval_vedett": "A megosztás jelszóval védett.",
"launcher.alkalmazasok_telepitese": "Alkalmazások telepítése",
@@ -2324,5 +2331,8 @@
"tier2_config.mentes": "Mentés",
"tier2_config.nincs_elerheto_off_drive_cel": "Nincs elérhető off-drive cél",
"tier2_config.nincs_masik_adatmeghajto_automatikus_cel": "Nincs másik adatmeghajtó — automatikus cél a belső SSD (csak DB/konfiguráció). Egy 2.\n adatmeghajtó hozzáadásával a teljes adat is off-drive menthető.",
"tier2_config.vissza_a_mentesekhez": "← Vissza a mentésekhez"
"tier2_config.vissza_a_mentesekhez": "← Vissza a mentésekhez",
"update.phase.copying": "Az adatok másolása a frissítés előtt…",
"update.phase.undoing": "Visszaállítás az előző változatra…",
"update.phase.undone": "Visszaállítva az előző változatra"
}
+4
View File
@@ -1718,6 +1718,10 @@ type RestoreHold struct {
// beállításokat, az adatbázist és a fájlokat tartalmazza" — recorded at hold time, because an app's
// data layout (bind-mounted files vs. named volumes) decides it and may not be readable later.
CopyHolds string `json:"copy_holds,omitempty"`
// UndoState (v0.263.0) is what a FAILED automatic undo left the data as — "untouched", "half" or
// "not_started" (stacks.UndoState*). Empty: no undo was attempted (every hold written before the
// field existed). Only set for HoldReasonUpdateFailed.
UndoState string `json:"undo_state,omitempty"`
}
// Hold reasons. See RestoreHold.Reason.
+5
View File
@@ -622,6 +622,11 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
// R-614: the app is going; its update record goes with it. Otherwise the NEXT install of the
// same name inherits a phase that belongs to an app that no longer exists.
m.ClearUpdateState(name)
// v0.263.0: an undo copy kept by a failed undo goes with the app — its volumes were just removed,
// and a copy of them must not outlive them.
if n := m.RemoveUndoCopies(name); n > 0 {
m.logger.Printf("[INFO] [stacks] RemoveStack %s: removed %d undo copy volume(s)", name, n)
}
// Step 3: the volumes that are gone now — `[]` when none, never null (R-489).
resp.VolumesRemoved = removedVolumes(volsBefore, m.projectVolumes(name))
+4
View File
@@ -154,6 +154,10 @@ type AppConfig struct {
// ABSENT MEANS UNPINNED, and unpinned means the app behaves exactly as it did before
// v0.235.0. It never means "pin to whatever the catalog says now".
PinnedImages map[string]string `yaml:"pinned_images,omitempty" json:"pinned_images,omitempty"`
// LastUpdateUndone (v0.263.0, 09 §3 decision 15) is the last update the box UNDID by itself —
// written only by a successful undo, cleared by the next successful update. The page shows one
// line from it; the future automatic caller reads it so it never re-presses the same step.
LastUpdateUndone *UpdateUndone `yaml:"last_update_undone,omitempty" json:"last_update_undone,omitempty"`
}
// InstalledImage is one compose service's observed image. See AppConfig.InstalledImages.
+12 -8
View File
@@ -235,14 +235,18 @@ type Manager struct {
execFn func(name string, args ...string) (string, error)
// --- guarded update (slice 4, update.go) ---
updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED
updateComposeFn func(dir string, env []string, args ...string) (string, error)
updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string)
updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal error, warning string)
updateDiskFreeFn func() (freeGiB float64, ok bool)
updateNowFn func() time.Time
updateJournalMu sync.Mutex
updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist
updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED
updateComposeFn func(dir string, env []string, args ...string) (string, error)
updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string)
// v0.263.0 undo seams (undo.go): the volume copier (nil ⇒ docker) and the undo's health wait,
// which receives the probe it must use (nil ⇒ waitUpdateHealthyMeta).
undoCopier volumeCopier
updateUndoHealthFn func(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string)
updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal error, warning string)
updateDiskFreeFn func() (freeGiB float64, ok bool)
updateNowFn func() time.Time
updateJournalMu sync.Mutex
updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist
// inspectRestartPolicyFn is the docker-inspect seam for the above; nil in production
// (dockerRestartPolicy). Tests inject a scripted lookup and never touch docker.
inspectRestartPolicyFn func(containerName string) (string, error)
+376
View File
@@ -0,0 +1,376 @@
package stacks
import (
"context"
"fmt"
"os"
"path/filepath"
"strconv"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
)
// ── The undo (09 §3 decision 15, v0.263.0) ─────────────────────────────────────────────────────────
//
// WHAT IT REPLACES. Until v0.262.1 a failed health check after `up` stopped the app and HELD it, and
// the household's only way back was a restore from a backup tier (§6.1, "THE ABORT DECISION"). Decision
// 15 replaced that: the box puts the old version back ITSELF, together with the data exactly as it was
// seconds before the update, and holds only if that undo fails too.
//
// WHY THE OLD RULING DOES NOT BIND THIS. The old version refuses to start on data the new one migrated
// (Nextcloud, docmost, RomM — §4, and the 2026-09-23 spike). The undo does not put the old image on
// migrated data: it puts back the PRE-MIGRATION data too, so the old version meets the data it knows.
// Measured by hand on docmost, romm and vikunja before a line of this was written
// (felhom.eu/documentation/audits/update-rulings-2026-09-23/ and undo-bakeoff-2026-09-23/).
//
// THE COPY IS A FOLDER COPY, chosen by the bake-off (decision 19). After the pull and just before
// `up` — where the app is stopped anyway to be recreated — every NAMED volume the app owns is copied
// with `cp -a` into a sibling volume by a helper container, which writes a finished-marker LAST. The
// dump-and-load route passed too, but an app with no database server gets no dump at all, so it would
// have needed this copy anyway.
//
// THREE RULES, each earned by a measurement:
//
// 1. FILES ON DISK ARE NEVER TOUCHED. Only named volumes are copied and put back. A bind-mounted
// folder — photos, documents, the household's drive — is never in the copy and never moved by an
// undo. An app whose data is only in such folders gets its old version back and nothing else.
// 2. A COPY COUNTS ONLY WITH ITS FINISHED-MARKER. Killing `docker run` does not stop the copy (the
// container runs on — measured on docmost, undo-bakeoff docmost-60); a copy container killed
// mid-way leaves fewer bytes and no marker. The marker is written after `cp -a` and `sync` both
// succeed, so its presence is the only evidence the copy is whole. Pinned by
// TestUndo_CutOffCopyIsRefusedBeforeAnythingMoves.
// 3. THE UNDO READS ITS OWN COPIES, NEVER THE RECOVERY UNIT. Measured 2026-09-23: ten seconds after
// a hold was lifted, the unit was re-captured with the NEW definition (R-645); a pin-back that read
// it started the new version on the restored data. The previous compose, applied definition, pin
// AND `.felhom.yml` are kept in the stack dir until the undo is over (R-639).
//
// And the old version is checked with the OLD `.felhom.yml` probe: the new one may name a port the
// old version never answers. Pinned by TestUndo_UsesTheOldProbe.
// Undo phases and outcomes.
const (
UpdatePhaseCopying = "copying"
UpdatePhaseUndoing = "undoing"
UpdatePhaseUndone = "undone"
// UndoState* is what a FAILED undo left the data as — carried to the hold so the household is
// told the truth about it. "" means the undo was not attempted (an update journaled by an older
// controller, resumed after an upgrade).
UndoStateUntouched = "untouched" // the copy was unusable; nothing was put back
UndoStateHalf = "half" // putting the copy back stopped part-way; the copy is kept
UndoStateNotStarted = "not_started" // data and definition put back; the old version did not come up
)
// undoCopyLabel marks every copy volume with the app it belongs to, so a removal can find and delete
// it (a copy carries no compose-project label, deliberately: it must never be mistaken for the app's
// own volume by the backup legs or the removal's volume accounting).
const undoCopyLabel = "felhom.undo-copy-of"
// undoCopyMarker is written last into a copy volume, beside the data (never inside it).
const undoCopyMarker = "felhom-undo-complete"
// preUpdateMetaDir holds the previous .felhom.yml. A DIRECTORY, so LoadMetadata can read it; the
// syncer copies only two files into the stack dir's root and never touches a subdirectory.
const preUpdateMetaDir = "pre-update-meta"
// undoHelperImage is the helper the backup legs already use for volume tars (backup.go, restore.go),
// so no new image is introduced onto a box.
const undoHelperImage = "alpine"
// undoCopy pairs an app volume with its last-second copy.
type undoCopy struct {
Volume string `json:"volume"`
Copy string `json:"copy"`
}
// UpdateUndone is app.yaml's record of the last update the box undid (decision 15: no automatic
// retry until the catalog moves; the button stays usable for a person). Written only by a
// successful undo; cleared by the next successful update.
type UpdateUndone struct {
To map[string]string `yaml:"to" json:"to"`
At string `yaml:"at" json:"at"`
Why string `yaml:"why,omitempty" json:"why,omitempty"`
}
// volumeCopier is the process boundary of the undo's copy. Production is dockerVolumeCopier; tests
// inject a fake and never touch docker.
type volumeCopier interface {
ProjectVolumes(project string) ([]string, error)
VolumeBytes(vol string) (int64, error)
Copy(src, dst, app string) error
Complete(copyVol string) bool
Restore(copyVol, vol string) error
Remove(vol string) error
CopiesOf(app string) []string
}
func (m *Manager) copier() volumeCopier {
if m.undoCopier != nil {
return m.undoCopier
}
return dockerVolumeCopier{m: m}
}
type dockerVolumeCopier struct{ m *Manager }
func (d dockerVolumeCopier) ProjectVolumes(project string) ([]string, error) {
out, err := d.m.execCommand("docker", "volume", "ls", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Name}}")
if err != nil {
return nil, err
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols, nil
}
func (d dockerVolumeCopier) VolumeBytes(vol string) (int64, error) {
out, err := d.m.execCommand("docker", "run", "--rm", "-v", vol+":/v:ro", undoHelperImage, "du", "-sb", "/v")
if err != nil {
return 0, err
}
f := strings.Fields(out)
if len(f) == 0 {
return 0, fmt.Errorf("du printed nothing for %s", vol)
}
return strconv.ParseInt(f[0], 10, 64)
}
// Copy runs in the FOREGROUND so the helper container's own exit status is the answer; the marker is
// written only after cp and sync both succeed.
func (d dockerVolumeCopier) Copy(src, dst, app string) error {
if _, err := d.m.execCommand("docker", "volume", "create", "--label", undoCopyLabel+"="+app, dst); err != nil {
return err
}
_, err := d.m.execCommand("docker", "run", "--rm", "-v", src+":/from:ro", "-v", dst+":/to", undoHelperImage,
"sh", "-c", "mkdir -p /to/data && cp -a /from/. /to/data/ && sync && touch /to/"+undoCopyMarker)
return err
}
func (d dockerVolumeCopier) Complete(copyVol string) bool {
_, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/c:ro", undoHelperImage, "test", "-f", "/c/"+undoCopyMarker)
return err == nil
}
// Restore empties the app's volume and copies the data back. The marker is checked again INSIDE the
// same helper, so a copy that lost it between the check and the restore is never poured in.
func (d dockerVolumeCopier) Restore(copyVol, vol string) error {
_, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/from:ro", "-v", vol+":/to", undoHelperImage,
"sh", "-c", "test -f /from/"+undoCopyMarker+" && find /to -mindepth 1 -delete && cp -a /from/data/. /to/ && sync")
return err
}
func (d dockerVolumeCopier) Remove(vol string) error {
_, err := d.m.execCommand("docker", "volume", "rm", "-f", vol)
return err
}
func (d dockerVolumeCopier) CopiesOf(app string) []string {
out, err := d.m.execCommand("docker", "volume", "ls", "-q", "--filter", "label="+undoCopyLabel+"="+app)
if err != nil {
return nil
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols
}
// undoCopyName is `<volume>.pre-update-<stamp>` — a legal Docker volume name, unique per update.
func undoCopyName(vol string, at time.Time) string {
return vol + ".pre-update-" + at.UTC().Format("20060102T150405Z")
}
// undoMsg renders a born-as-key sentence of the update job. Hungarian, like every other sentence the
// job writes today (R-606 localises them together).
func undoMsg(key string, args ...interface{}) string { return util.Text("hu", key, args...) }
// planUndoCopies lists the app's named volumes and refuses — before anything moves — when their copy
// would breach the disk floor the update already keeps (decision 19's disk limit).
func (m *Manager) planUndoCopies(name string) ([]string, error) {
vols, err := m.copier().ProjectVolumes(name)
if err != nil {
return nil, fmt.Errorf("listing the app's volumes: %w", err)
}
var total int64
for _, v := range vols {
b, err := m.copier().VolumeBytes(v)
if err != nil {
return nil, fmt.Errorf("sizing volume %s: %w", v, err)
}
total += b
}
needGiB := float64(total) / (1 << 30)
if free, known := m.updateDiskFree(); known && free-needGiB < updateDiskFloorGiB {
return nil, &undoSpaceError{need: needGiB, free: free}
}
m.logger.Printf("[INFO] [stacks] update %s: the undo copy will hold %d named volume(s), %.1f MiB", name, len(vols), float64(total)/(1<<20))
return vols, nil
}
type undoSpaceError struct{ need, free float64 }
func (e *undoSpaceError) Error() string {
return fmt.Sprintf("the undo copy needs %.2f GiB and %.2f GiB is free (floor %.0f GiB)", e.need, e.free, updateDiskFloorGiB)
}
// makeUndoCopies stops the app and copies every volume. The copies are journaled BEFORE each copy
// starts, so a power cut mid-copy leaves a journal that names every volume to clean up.
func (m *Manager) makeUndoCopies(name, dir string, env []string, vols []string, entry *updateJournalEntry) error {
if _, err := m.updateCompose(dir, env, "stop"); err != nil {
return fmt.Errorf("stopping the app for the copy: %w", err)
}
stamp := m.now()
for _, v := range vols {
c := undoCopy{Volume: v, Copy: undoCopyName(v, stamp)}
entry.UndoCopies = append(entry.UndoCopies, c)
if !m.enterUpdatePhase(name, entry, UpdatePhaseCopying) {
return fmt.Errorf("journal write failed before copying %s", v)
}
t0 := time.Now()
if err := m.copier().Copy(c.Volume, c.Copy, name); err != nil {
return fmt.Errorf("copying %s: %w", v, err)
}
m.logger.Printf("[INFO] [stacks] update %s: copied %s → %s in %s", name, c.Volume, c.Copy, time.Since(t0).Round(time.Millisecond))
}
entry.Copied = true
return nil
}
func (m *Manager) removeUndoCopies(name string, copies []undoCopy) {
for _, c := range copies {
if err := m.copier().Remove(c.Copy); err != nil {
m.logger.Printf("[WARN] [stacks] update %s: could not remove the undo copy %s: %v", name, c.Copy, err)
}
}
}
// RemoveUndoCopies deletes every undo copy an app still has — the removal path's hook, so a copy kept
// by a failed undo does not outlive the app.
func (m *Manager) RemoveUndoCopies(name string) int {
n := 0
for _, v := range m.copier().CopiesOf(name) {
if err := m.copier().Remove(v); err != nil {
m.logger.Printf("[WARN] [stacks] remove %s: could not delete the undo copy %s: %v", name, v, err)
continue
}
n++
}
return n
}
// savePreUpdateMeta keeps the previous .felhom.yml for the undo's health check.
func savePreUpdateMeta(dir string) (string, error) {
src, err := os.ReadFile(filepath.Join(dir, ".felhom.yml"))
if err != nil {
return "", err
}
md := filepath.Join(dir, preUpdateMetaDir)
if err := os.MkdirAll(md, 0o755); err != nil {
return "", err
}
return md, os.WriteFile(filepath.Join(md, ".felhom.yml"), src, 0o644)
}
func (m *Manager) undoHealth(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string) {
if m.updateUndoHealthFn != nil {
return m.updateUndoHealthFn(ctx, name, timeout, meta)
}
// A test that injects only the update's health seam gets the same answer for the undo (production
// injects neither and always takes waitUpdateHealthyMeta).
if m.updateHealthFn != nil {
return m.updateHealthFn(ctx, name, timeout)
}
return m.waitUpdateHealthyMeta(ctx, name, timeout, meta)
}
// tryUndo puts the pre-update data and definition back and checks the old version with its own probe.
// It returns "" when the app is healthy on its old version again, else the state the data is in.
//
// ORDER IS THE DESIGN: every copy is validated BEFORE any is poured back (a cut-off copy leaves the
// data exactly as the new version left it — UndoStateUntouched); the definition is put back only after
// the data (so a failure in between reads "half" with the pin still naming the new version, which is
// what ran on that data).
func (m *Manager) tryUndo(ctx context.Context, name, dir, why string, entry *updateJournalEntry) string {
start := m.now()
if !m.enterUpdatePhase(name, entry, UpdatePhaseUndoing) {
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the undo — undoing anyway", name)
}
m.logger.Printf("[WARN] [stacks] update %s: UNDO — putting back the previous version and its %d volume copy(ies) (reason: %s)", name, len(entry.UndoCopies), why)
if _, err := m.updateCompose(dir, m.stackEnv(dir), "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the new version before the undo failed: %v", name, err)
}
for _, c := range entry.UndoCopies {
if !m.copier().Complete(c.Copy) {
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy %s has no finished-marker — it is cut off or missing; NOTHING is put back", name, c.Copy)
return UndoStateUntouched
}
}
for _, c := range entry.UndoCopies {
if err := m.copier().Restore(c.Copy, c.Volume); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: putting %s back from %s FAILED: %v — the data is mixed; the copies are kept", name, c.Volume, c.Copy, err)
return UndoStateHalf
}
}
m.restoreDefinition(name, dir, *entry)
env := m.stackEnv(dir)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: starting the previous version failed: %v", name, err)
m.captureHoldLogs(name, dir, env)
_, _ = m.updateCompose(dir, env, "down")
return UndoStateNotStarted
}
meta := LoadMetadata(dir)
if entry.PrevMeta != "" {
if _, err := os.Stat(filepath.Join(entry.PrevMeta, ".felhom.yml")); err == nil {
meta = LoadMetadata(entry.PrevMeta)
} else {
m.logger.Printf("[WARN] [stacks] update %s: the previous .felhom.yml is missing (%v) — checking with the current one", name, err)
}
}
healthy, detail := m.undoHealth(ctx, name, m.healthTimeout(), &meta)
if !healthy {
m.logger.Printf("[ERROR] [stacks] update %s: the previous version did not come up after the undo: %s", name, detail)
m.captureHoldLogs(name, dir, env)
_, _ = m.updateCompose(dir, env, "down")
return UndoStateNotStarted
}
m.recordInstalledImages(name, dir, env)
m.removeUndoCopies(name, entry.UndoCopies)
m.recordUpdateUndone(name, dir, &UpdateUndone{To: entry.NewPin, At: start.UTC().Format(time.RFC3339), Why: why})
m.removePreUpdateCopies(dir)
_ = m.RefreshStatus()
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseUndone, "")
m.logger.Printf("[INFO] [stacks] update %s: UNDONE in %s — the previous version is running on the data from before the update (%s)", name, m.now().Sub(start).Round(time.Second), detail)
return ""
}
// recordUpdateUndone writes (or, with nil, clears) app.yaml's last_update_undone. A failed write is
// logged and never fails the undo — it is a record, and the app is already back.
func (m *Manager) recordUpdateUndone(name, dir string, u *UpdateUndone) {
cfg := LoadAppConfig(dir)
if cfg == nil || (u == nil && cfg.LastUpdateUndone == nil) {
return
}
cfg.LastUpdateUndone = u
meta := LoadMetadata(dir)
if err := SaveAppConfig(dir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: recording last_update_undone failed: %v", name, err)
return
}
m.mu.Lock()
if st, ok := m.stacks[name]; ok && st.AppConfig != nil {
st.AppConfig.LastUpdateUndone = u
}
m.mu.Unlock()
}
+417
View File
@@ -0,0 +1,417 @@
package stacks
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"sort"
"strings"
"sync"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
)
// v0.263.0 — the undo (09 §3 decision 15, undo.go). Every test runs the REAL job (runGuardedUpdate /
// RecoverUpdates / ResumeInterruptedUpdates) with the process boundaries faked, and reads the EFFECT
// back: the data in the fake volume, the pin in app.yaml, the phase, the hold, the journal.
// fakeCopier is the volume boundary. `vols` is each app volume's CONTENT, so "the data came back" is
// a string compare, and `complete` is each copy's finished-marker.
type fakeCopier struct {
mu sync.Mutex
vols map[string]string
copies map[string]string
complete map[string]bool
calls []string
copyErr error
cutOff bool // every copy is made WITHOUT its finished-marker (a copy container killed mid-way)
restoreErr error
bytes int64
}
func newFakeCopier(vols map[string]string) *fakeCopier {
return &fakeCopier{vols: vols, copies: map[string]string{}, complete: map[string]bool{}, bytes: 1 << 20}
}
func (f *fakeCopier) note(c string) { f.calls = append(f.calls, c) }
func (f *fakeCopier) ProjectVolumes(string) ([]string, error) {
f.mu.Lock()
defer f.mu.Unlock()
var out []string
for v := range f.vols {
out = append(out, v)
}
sort.Strings(out)
return out, nil
}
func (f *fakeCopier) VolumeBytes(string) (int64, error) { return f.bytes, nil }
func (f *fakeCopier) Copy(src, dst, _ string) error {
f.mu.Lock()
defer f.mu.Unlock()
f.note("copy " + src)
if f.copyErr != nil {
return f.copyErr
}
f.copies[dst] = f.vols[src]
f.complete[dst] = !f.cutOff
return nil
}
func (f *fakeCopier) Complete(c string) bool {
f.mu.Lock()
defer f.mu.Unlock()
return f.complete[c]
}
// Restore refuses a copy with no marker, exactly as the real helper does (`test -f` in the same shell).
func (f *fakeCopier) Restore(c, v string) error {
f.mu.Lock()
defer f.mu.Unlock()
f.note("restore " + v)
if f.restoreErr != nil {
return f.restoreErr
}
if !f.complete[c] {
return fmt.Errorf("no finished-marker in %s", c)
}
f.vols[v] = f.copies[c]
return nil
}
func (f *fakeCopier) Remove(v string) error {
f.mu.Lock()
defer f.mu.Unlock()
f.note("remove " + v)
delete(f.copies, v)
delete(f.complete, v)
return nil
}
func (f *fakeCopier) CopiesOf(string) []string {
f.mu.Lock()
defer f.mu.Unlock()
var out []string
for c := range f.copies {
out = append(out, c)
}
return out
}
func (f *fakeCopier) vol(v string) string { f.mu.Lock(); defer f.mu.Unlock(); return f.vols[v] }
func (f *fakeCopier) setVol(v, s string) { f.mu.Lock(); f.vols[v] = s; f.mu.Unlock() }
func (f *fakeCopier) nCopies() int { f.mu.Lock(); defer f.mu.Unlock(); return len(f.copies) }
func (f *fakeCopier) callsHave(p string) bool {
f.mu.Lock()
defer f.mu.Unlock()
for _, c := range f.calls {
if strings.HasPrefix(c, p) {
return true
}
}
return false
}
const (
undoMetaOld = "healthcheck:\n checks:\n - type: http\n port: 3000\n"
undoMetaNew = "healthcheck:\n checks:\n - type: http\n port: 3999\n" // the drill's wrong probe
undoVol = "nextcloud_db"
)
// newUndoManager: slice 4's manager plus a data volume holding "OLD", the OLD .felhom.yml in the stack
// dir, and a compose fake that plays the two things the real world does in between: the catalog's
// .felhom.yml flows in with the new probe during the pull (§5.4 — .felhom.yml is never frozen), and
// the NEW version migrates the data on its first `up`.
func newUndoManager(t *testing.T) (*Manager, string, *fakeGuards, *composeRec, *fakeCopier) {
t.Helper()
m, dir, g, c := newSlice4Manager(t)
mustWrite(t, filepath.Join(dir, ".felhom.yml"), undoMetaOld)
fc := newFakeCopier(map[string]string{undoVol: "OLD"})
m.undoCopier = fc
pulled, migrated := false, false
m.updateComposeFn = func(d string, env []string, args ...string) (string, error) {
switch args[0] {
case "pull":
pulled = true
mustWrite(t, filepath.Join(dir, ".felhom.yml"), undoMetaNew)
case "up":
// Only the NEW version migrates: the first `up` after a pull, with the undo copy taken.
// (After a failed copy, or in a resumed undo, the `up` starts the OLD version.)
if pulled && !migrated && fc.nCopies() > 0 {
migrated = true
fc.setVol(undoVol, "MIGRATED")
}
}
return c.fn(d, env, args...)
}
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "new version unhealthy" }
m.updateUndoHealthFn = func(_ context.Context, _ string, _ time.Duration, meta *Metadata) (bool, string) {
if meta != nil && meta.HealthCheck != nil && len(meta.HealthCheck.Checks) > 0 && meta.HealthCheck.Checks[0].Port == 3000 {
return true, "old probe answered"
}
return false, "probed the wrong port"
}
return m, dir, g, c, fc
}
// TestUndo_FailedUpdateIsPutBackWithItsData is decision 15 in one test: the new version migrates the
// data and fails its check; the box puts the old version back WITH THE PRE-UPDATE DATA, and nothing is
// held.
//
// COMPANION RED-PROOF 1 (REPORT.md): at v0.262.1's shape — failAndHold without the tryUndo call — the
// app ends HELD with the data still "MIGRATED", and this test fails on the first assertion.
func TestUndo_FailedUpdateIsPutBackWithItsData(t *testing.T) {
m, dir, g, _, fc := newUndoManager(t)
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); held || st.UpdatePhase != UpdatePhaseUndone {
t.Fatalf("a failed update must be UNDONE, not held; held=%v phase=%q err=%q", held, st.UpdatePhase, st.UpdateError)
}
if got := fc.vol(undoVol); got != "OLD" {
t.Errorf("the data must be the PRE-UPDATE data again, got %q", got)
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the pin must be the old version again, got %q", got)
}
if got := fileBody(t, filepath.Join(dir, "docker-compose.yml")); got != pinTplOld {
t.Errorf("the live definition must be the old one again:\n%s", got)
}
if st.UpdateError != "" || st.UpdatePhaseLabel != "Visszaállítva az előző változatra" {
t.Errorf("an undone update carries no error and the undone label; err=%q label=%q", st.UpdateError, st.UpdatePhaseLabel)
}
u := readPin(t, dir).LastUpdateUndone
if u == nil || u.To["web"] != "nextcloud:34.0.1-apache" || u.At == "" || !strings.Contains(u.Why, "unhealthy") {
t.Errorf("app.yaml must remember the undone step (to, at, why), got %+v", u)
}
if fc.nCopies() != 0 || journalExists(m) {
t.Errorf("after a successful undo the copies and the journal are gone; copies=%d journal=%v", fc.nCopies(), journalExists(m))
}
for _, f := range []string{preUpdateComposeFile, preUpdateAppliedFile, preUpdateMetaDir} {
if _, err := os.Stat(filepath.Join(dir, f)); err == nil {
t.Errorf("the pre-update copy %s must be removed once the undo is over", f)
}
}
}
// TestUndo_CutOffCopyIsRefusedBeforeAnythingMoves: a copy without its finished-marker is detected
// BEFORE anything is poured back; the outcome is today's HOLD, saying the data is as the new version
// left it — never a "success".
//
// COMPANION RED-PROOF 2 (REPORT.md): delete the Complete() validation loop in tryUndo. Restore is then
// called on the cut copy and the undo state reads "half" — this test fails on both assertions.
func TestUndo_CutOffCopyIsRefusedBeforeAnythingMoves(t *testing.T) {
m, _, g, _, fc := newUndoManager(t)
fc.cutOff = true
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); !held || st.UpdatePhase != UpdatePhaseFailed || g.undoState != UndoStateUntouched {
t.Fatalf("a cut-off copy must end HELD with state %q; held=%v phase=%q state=%q", UndoStateUntouched, held, st.UpdatePhase, g.undoState)
}
if fc.callsHave("restore ") {
t.Error("NOTHING may be poured back from a copy that is not whole")
}
if got := fc.vol(undoVol); got != "MIGRATED" {
t.Errorf("the data must be left as the new version left it, got %q", got)
}
if fc.nCopies() == 0 {
t.Error("a failed undo keeps its copies for the operator")
}
}
// TestUndo_UsesTheOldProbe: the new .felhom.yml names a port the old version never answers (the drill
// case, and a real one whenever a probe moves with a version). The undo must judge the old version
// with the OLD probe.
//
// COMPANION RED-PROOF 3 (REPORT.md): make tryUndo use LoadMetadata(dir) (the current file) instead of
// entry.PrevMeta — the undo then probes port 3999 and the app ends HELD; this test fails.
func TestUndo_UsesTheOldProbe(t *testing.T) {
m, _, g, _, _ := newUndoManager(t)
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); held || st.UpdatePhase != UpdatePhaseUndone {
t.Fatalf("with the old probe the old version is healthy and the undo completes; held=%v phase=%q state=%q", held, st.UpdatePhase, g.undoState)
}
}
// TestUndo_OldVersionThatDoesNotStartIsHeldSayingSo: data and definition put back, the old version
// still unhealthy → HOLD with the not_started state, and the copies kept.
func TestUndo_OldVersionThatDoesNotStartIsHeldSayingSo(t *testing.T) {
m, _, g, _, fc := newUndoManager(t)
m.updateUndoHealthFn = func(context.Context, string, time.Duration, *Metadata) (bool, string) {
return false, "old version broken too"
}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); !held || g.undoState != UndoStateNotStarted || st.UpdatePhase != UpdatePhaseFailed {
t.Fatalf("held=%v state=%q phase=%q", held, g.undoState, st.UpdatePhase)
}
if got := fc.vol(undoVol); got != "OLD" {
t.Errorf("the data was put back before the check, got %q", got)
}
}
// TestUndo_RestoreFailureIsHalf: putting the copy back fails part-way → HOLD, state "half".
func TestUndo_RestoreFailureIsHalf(t *testing.T) {
m, _, g, _, fc := newUndoManager(t)
fc.restoreErr = errors.New("disk full")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); !held || g.undoState != UndoStateHalf {
t.Fatalf("held=%v state=%q", held, g.undoState)
}
}
// TestUndo_CopyFailureMovesNothing: the copy is taken after the pull; if it fails the update stops
// there — the partial copy goes, the pin goes back, the old version starts, and nothing is held.
func TestUndo_CopyFailureMovesNothing(t *testing.T) {
m, dir, g, c, fc := newUndoManager(t)
fc.copyErr = errors.New("helper exited 137")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); held {
t.Fatal("a failed copy moved nothing and must not hold")
}
if st.UpdateError != util.Text("hu", "err.stacks.update_undo_copy_failed") {
t.Errorf("err = %q", st.UpdateError)
}
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the pin must go back, got %q", got)
}
if fc.vol(undoVol) != "OLD" || fc.nCopies() != 0 {
t.Errorf("data untouched and no copy left; data=%q copies=%d", fc.vol(undoVol), fc.nCopies())
}
if got := strings.Join(c.list(), " | "); got != "pull | stop | up -d --remove-orphans" {
t.Errorf("the previous version must be started again after the failed copy; compose calls = %q", got)
}
}
// TestUndo_NoRoomForTheCopyRefusesBeforeAnythingMoves: decision 19's disk limit.
func TestUndo_NoRoomForTheCopyRefusesBeforeAnythingMoves(t *testing.T) {
m, dir, _, c, fc := newUndoManager(t)
fc.bytes = 49 << 30 // 49 GiB of volumes; 50 GiB free; the 2 GiB floor must hold
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if !strings.Contains(st.UpdateError, "nincs elég szabad hely a frissítés előtti adatmásolathoz") {
t.Errorf("err = %q", st.UpdateError)
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || len(c.list()) != 0 || fc.callsHave("copy ") {
t.Errorf("nothing may move: pin=%q compose=%v", pinOf(t, dir), c.list())
}
}
// TestUndo_PowerCutDuringTheUndoResumesIt: the journal says `undoing`; after a restart the undo runs
// again from the copies and ends healthy — never "done", never dropped.
//
// COMPANION RED-PROOF 4 (REPORT.md): delete the UpdatePhaseUndoing arm from RecoverUpdates — the entry
// falls to `default` (dropped), nothing is resumed, and this test fails at the first assertion.
func TestUndo_PowerCutDuringTheUndoResumesIt(t *testing.T) {
m, dir, g, _, fc := newUndoManager(t)
e := simulateAdvanced(t, m, dir)
md, err := savePreUpdateMeta(dir)
if err != nil {
t.Fatal(err)
}
e.PrevMeta, e.Copied, e.Phase = md, true, UpdatePhaseUndoing
e.NewPin = map[string]string{"web": "nextcloud:34.0.1-apache"}
e.UndoCopies = []undoCopy{{Volume: undoVol, Copy: undoVol + ".pre-update-x"}}
fc.copies[undoVol+".pre-update-x"], fc.complete[undoVol+".pre-update-x"] = "OLD", true
fc.setVol(undoVol, "MIGRATED")
writeTestJournal(t, m, "nextcloud", e)
guards := m.updateGuards
m.updateGuards = nil
resumed := m.RecoverUpdates()
if len(resumed) != 1 || !m.IsUpdating("nextcloud") {
t.Fatalf("an interrupted undo must be resumed and the app marked Updating; resumed=%v", resumed)
}
if st, _ := m.GetStack("nextcloud"); st.UpdatePhase != UpdatePhaseUndoing {
t.Errorf("phase after recovery = %q, want %q", st.UpdatePhase, UpdatePhaseUndoing)
}
m.updateGuards = guards
if n := m.ResumeInterruptedUpdates(context.Background()); n != 1 {
t.Fatalf("resumed %d", n)
}
st := waitUpdateDone(t, m, "nextcloud")
if held, _ := g.HoldFor("nextcloud"); held || st.UpdatePhase != UpdatePhaseUndone {
t.Fatalf("the resumed undo must complete; held=%v phase=%q", held, st.UpdatePhase)
}
if fc.vol(undoVol) != "OLD" || pinOf(t, dir) != "nextcloud:31.0.14-apache" {
t.Errorf("data=%q pin=%q", fc.vol(undoVol), pinOf(t, dir))
}
}
// TestUndo_PowerCutDuringTheCopyPutsTheOldVersionBack: interrupted in `copying` — nothing new ran; the
// partial copies go, the pin goes back and the old version is started.
func TestUndo_PowerCutDuringTheCopyPutsTheOldVersionBack(t *testing.T) {
m, dir, _, c, fc := newUndoManager(t)
e := simulateAdvanced(t, m, dir)
e.Phase = UpdatePhaseCopying
e.UndoCopies = []undoCopy{{Volume: undoVol, Copy: undoVol + ".pre-update-x"}}
fc.copies[undoVol+".pre-update-x"] = "OL" // cut
writeTestJournal(t, m, "nextcloud", e)
if resumed := m.RecoverUpdates(); len(resumed) != 0 {
t.Fatalf("resumed = %v", resumed)
}
if pinOf(t, dir) != "nextcloud:31.0.14-apache" || fc.nCopies() != 0 || journalExists(m) {
t.Errorf("pin=%q copies=%d journal=%v", pinOf(t, dir), fc.nCopies(), journalExists(m))
}
if got := strings.Join(c.list(), " | "); got != "up -d --remove-orphans" {
t.Errorf("the old version must be started again; compose calls = %q", got)
}
}
// TestUndo_ASuccessfulUpdateEndsTheUndoneNote: the next update that works clears last_update_undone.
//
// COMPANION RED-PROOF 5 (REPORT.md): drop the recordUpdateUndone(nil) call from verifyAndConclude —
// the note survives a successful update and this test fails.
func TestUndo_ASuccessfulUpdateEndsTheUndoneNote(t *testing.T) {
m, dir, _, _, _ := newUndoManager(t)
m.recordUpdateUndone("nextcloud", dir, &UpdateUndone{To: map[string]string{"web": "x"}, At: "2026-09-23T10:00:00Z"})
if readPin(t, dir).LastUpdateUndone == nil {
t.Fatal("setup: the note was not written")
}
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return true, "fine" }
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("phase = %q", st.UpdatePhase)
}
if u := readPin(t, dir).LastUpdateUndone; u != nil {
t.Errorf("a successful update must end the undone note, got %+v", u)
}
}
// TestUndo_PhaseLabelsMatchTheBundle pins the Hungarian labels to the born-as-key bundle text.
func TestUndo_PhaseLabelsMatchTheBundle(t *testing.T) {
for phase, key := range map[string]string{UpdatePhaseCopying: "update.phase.copying", UpdatePhaseUndoing: "update.phase.undoing", UpdatePhaseUndone: "update.phase.undone"} {
if got, want := UpdatePhaseLabel(phase), util.Text("hu", key); got != want || got == "" || got == key {
t.Errorf("%s: label %q, bundle %q", phase, got, want)
}
}
if util.Text("en", "update.phase.undone") != "Put back to the previous version" {
t.Errorf("English label = %q", util.Text("en", "update.phase.undone"))
}
}
// TestUndo_RemovalDeletesKeptCopies: a copy kept by a failed undo goes with the app.
func TestUndo_RemovalDeletesKeptCopies(t *testing.T) {
m, _, _, _, fc := newUndoManager(t)
fc.copies["nextcloud_db.pre-update-x"] = "OLD"
if n := m.RemoveUndoCopies("nextcloud"); n != 1 || fc.nCopies() != 0 {
t.Errorf("removed %d, left %d", n, fc.nCopies())
}
}
+138 -21
View File
@@ -23,7 +23,7 @@ import (
// THE SEQUENCE, and the order is the point:
//
// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling
// → starting → verifying → done | failed
// → copying → starting → verifying → done | (undoing → undone | failed)
//
// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02,
// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The
@@ -37,9 +37,11 @@ import (
// pin claiming the old version would be a record of something untrue (Scenario F). The app is
// HELD STOPPED and the customer is told which backup it can be restored from.
//
// WHAT IT DELIBERATELY DOES NOT DO: put the old version back by itself. SPIKE-upgrade-test-2026-09-06
// measured that whether the old image starts on migrated data depends on the app (PrivateBin yes,
// Docmost and Nextcloud no) and cannot be predicted. The route back is the restore.
// SINCE v0.263.0 IT PUTS THE OLD VERSION BACK ITSELF — with the data from before the update (09 §3
// decision 15, undo.go). Until then it deliberately did not: SPIKE-upgrade-test-2026-09-06 measured
// that the old image alone refuses data a new one migrated (Docmost, Nextcloud). The undo does not put
// the old image on migrated data; it puts the pre-migration data back as well. A HOLD now happens only
// when that undo fails too.
//
// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10),
// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at
@@ -56,6 +58,7 @@ const (
UpdatePhaseVerifying = "verifying"
UpdatePhaseDone = "done"
UpdatePhaseFailed = "failed"
// UpdatePhaseCopying, UpdatePhaseUndoing and UpdatePhaseUndone live in undo.go.
)
// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the
@@ -71,6 +74,10 @@ var updatePhaseLabels = map[string]string{
UpdatePhaseVerifying: "Működés ellenőrzése…",
UpdatePhaseDone: "Frissítve",
UpdatePhaseFailed: "A frissítés nem sikerült",
// v0.263.0 — born as bundle keys (update.phase.*); TestUndo_PhaseLabelsMatchTheBundle pins them equal.
UpdatePhaseCopying: "Az adatok másolása a frissítés előtt…",
UpdatePhaseUndoing: "Visszaállítás az előző változatra…",
UpdatePhaseUndone: "Visszaállítva az előző változatra",
}
// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one.
@@ -236,7 +243,9 @@ type UpdateGuards interface {
CanBackUp(name string) (bool, string)
BackupNow(ctx context.Context, name string) error
SafetyDump(ctx context.Context, name string) ([]string, error)
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint) error
// HoldAfterFailedUpdate records the hold. undoState (v0.263.0) says what a FAILED undo left the data
// as (UndoState*), "" when no undo was attempted; the hold sentence says so.
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint, undoState string) error
}
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
@@ -654,6 +663,18 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
}
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
// v0.263.0 — the undo's copy is PLANNED here, before anything moves: its size against the disk
// floor (decision 19's limit). The copy itself is taken after the pull, where the app stops anyway.
undoVols, perr := m.planUndoCopies(name)
if perr != nil {
if se, ok := perr.(*undoSpaceError); ok {
fail(undoMsg("err.stacks.update_undo_space", se.need, se.free, updateDiskFloorGiB), "undo copy: "+perr.Error())
} else {
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy plan: "+perr.Error())
}
return
}
// PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash
// at any later instant can put it back (Scenario G).
prevLive, err := os.ReadFile(st.ComposePath)
@@ -666,6 +687,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
return
}
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
// R-639: the OLD .felhom.yml too — its probe is the one the old version answers.
if md, merr := savePreUpdateMeta(dir); merr != nil {
m.logger.Printf("[WARN] [stacks] update %s: could not keep the previous .felhom.yml (%v) — an undo would check health with the current one", name, merr)
} else {
entry.PrevMeta = md
}
if applied, aerr := LoadAppliedDefinition(dir); aerr == nil {
if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil {
entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile)
@@ -687,6 +714,9 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
return
}
if cfg := LoadAppConfig(dir); cfg != nil {
entry.NewPin = cfg.PinnedImages
}
env := m.stackEnv(dir)
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
@@ -703,13 +733,32 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
return
}
// COPYING (v0.263.0) — the app is stopped here anyway to be recreated, so the extra downtime is the
// copy alone. A failed copy moves nothing: the copies go, the pin goes back, the old containers
// start again.
if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) {
m.pinBack(name, dir, entry)
fail(MsgUpdateJournalFailed, "journal write failed")
return
}
if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy failed: %v — removing it, putting the pin back and starting the previous version", name, err)
m.removeUndoCopies(name, entry.UndoCopies)
m.pinBack(name, dir, entry)
if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil {
m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr)
}
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy: "+err.Error())
return
}
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up")
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up", &entry)
return
}
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
// Containers may already have been recreated on the new image — something may have run.
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error())
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error(), &entry)
return
}
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
@@ -725,12 +774,14 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
waitStart := m.now()
healthy, detail := m.updateHealth(ctx, name, timeout)
if !healthy {
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail)
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail, entry)
return
}
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
m.recordInstalledImages(name, dir, env)
_ = m.RefreshStatus()
m.removeUndoCopies(name, entry.UndoCopies)
m.recordUpdateUndone(name, dir, nil) // a successful update ends the "undone" note
m.clearJournal(name)
m.removePreUpdateCopies(dir)
m.finishUpdate(name, UpdatePhaseDone, "")
@@ -762,22 +813,34 @@ func (m *Manager) captureHoldLogs(name, dir string, env []string) {
m.logger.Printf("[INFO] [stacks] update %s: kept %d bytes of the app's own log at %s before stopping it (R-621)", name, len(out), path)
}
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
// failAndHold is Scenario F. Since v0.263.0 it first UNDOES (decision 15): when the job took its
// last-second copy, the old version goes back with the data from before the update, and the app is
// HELD only when that undo fails too. Without a copy (an update journaled by an older controller,
// resumed after an upgrade) it holds as it always did.
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string, entry *updateJournalEntry) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s", name, why)
// R-621: capture the app's own logs BEFORE the `down`, because the `down` destroys them. Two
// drill nights lost the only evidence of WHY an update failed this way — `adventurelog` ran nine
// migrations and then never bound its port, and the log that would have said so was gone by the
// time anyone looked. The capture is bounded and best-effort: a hold must never fail because its
// evidence could not be written.
m.captureHoldLogs(name, dir, env)
if _, err := m.updateCompose(dir, env, "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
undoState := ""
if entry != nil && entry.Copied {
if undoState = m.tryUndo(ctx, name, dir, why, entry); undoState == "" {
return // undone: the previous version runs on the pre-update data
}
m.logger.Printf("[ERROR] [stacks] update %s: the UNDO failed too (%s) — HOLDING the app; the undo copies are kept: %v", name, undoState, entry.UndoCopies)
} else {
m.logger.Printf("[ERROR] [stacks] update %s: no undo copy for this update — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name)
if _, err := m.updateCompose(dir, env, "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
}
}
msg := MsgUpdateHoldUnsaved
if g := m.guards(); g == nil {
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp); err != nil {
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
@@ -789,8 +852,17 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
m.finishUpdate(name, UpdatePhaseFailed, msg)
}
// pinBack restores the pin, the stored definition and the live file from the journaled copies.
// pinBack restores the pin, the stored definition and the live file from the journaled copies, and
// then removes the copies.
func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
m.restoreDefinition(name, dir, entry)
m.removePreUpdateCopies(dir)
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
}
// restoreDefinition is pinBack WITHOUT removing the copies — the undo's form, so a power cut after it
// can run it again (RecoverUpdates → undoing) and find the copies still there.
func (m *Manager) restoreDefinition(name, dir string, entry updateJournalEntry) {
prevLive, lerr := os.ReadFile(entry.PrevCompose)
if lerr != nil {
m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr)
@@ -811,13 +883,12 @@ func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err)
}
}
m.removePreUpdateCopies(dir)
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
}
func (m *Manager) removePreUpdateCopies(dir string) {
_ = os.Remove(filepath.Join(dir, preUpdateComposeFile))
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
_ = os.RemoveAll(filepath.Join(dir, preUpdateMetaDir))
}
// settleReason names WHY the wait fell back to container state, so the journal and the log do not
@@ -834,6 +905,12 @@ func settleReason(meta Metadata) string {
// existing probe, or — for an app with none — every container running and none restarting for
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
return m.waitUpdateHealthyMeta(ctx, name, timeout, nil)
}
// waitUpdateHealthyMeta is the health wait with the probe taken from `override` instead of the app's
// current .felhom.yml — the undo's form (the old version is judged by the old probe). nil = current.
func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeout time.Duration, override *Metadata) (bool, string) {
deadline := m.now().Add(timeout)
var runningSince time.Time
warnedNoProbe := false
@@ -857,8 +934,12 @@ func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout ti
// 404 afterwards. A stack with no probe is not "healthy" and it is not "failing" — it is
// SETTLED ON CONTAINER STATE (`09` §3), and never a reason to stop a running app.
usable := false
if hc := st.Meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
c, candidates := findProbeContainerMeta(name, &st.Meta, st.Containers)
meta := st.Meta
if override != nil {
meta = *override
}
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
c, candidates := findProbeContainerMeta(name, &meta, st.Containers)
if c != "" {
usable = true
res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
@@ -883,7 +964,7 @@ func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout ti
}
if m.now().Sub(runningSince) >= updateSettleWindow {
return true, fmt.Sprintf("all containers running, none restarting, for %s (%s)",
updateSettleWindow, settleReason(st.Meta))
updateSettleWindow, settleReason(meta))
}
last = "running, settling"
}
@@ -914,6 +995,13 @@ type updateJournalEntry struct {
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
// the right copy. 0 in a journal written by v0.238.1 or older.
ProvenTier int `json:"proven_tier,omitempty"`
// v0.263.0 — the undo (undo.go). UndoCopies is journaled BEFORE each copy starts; Copied is true
// only once every copy finished (it may be true with zero copies: an app with no named volume).
// PrevMeta is the directory holding the previous .felhom.yml; NewPin the pin the update moved to.
UndoCopies []undoCopy `json:"undo_copies,omitempty"`
Copied bool `json:"copied,omitempty"`
PrevMeta string `json:"prev_meta,omitempty"`
NewPin map[string]string `json:"new_pin,omitempty"`
}
type updateJournal struct {
@@ -1043,6 +1131,30 @@ func (m *Manager) RecoverUpdates() []string {
m.pinBack(name, dir, e)
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
case UpdatePhaseCopying:
// v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the
// pin goes back, and the previous version is started again.
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while copying its data (started %s) — nothing new ran; removing the partial copy, putting the pin back and starting the previous version", name, e.StartedAt.Format(time.RFC3339))
m.removeUndoCopies(name, e.UndoCopies)
m.pinBack(name, dir, e)
if _, err := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); err != nil {
m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err)
}
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
case UpdatePhaseUndoing:
// v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from
// the copies (still there: they are removed only after the undo succeeded) and then probes.
// Never "done": what ran last was a failed new version.
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while UNDOING (started %s) — marking it Updating and RESUMING the undo", name, e.StartedAt.Format(time.RFC3339))
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Updating, s.UpdateError, s.updateHeld = true, "", false
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseUndoing, UpdatePhaseLabel(UpdatePhaseUndoing)
}
m.updateResume = append(m.updateResume, name)
m.mu.Unlock()
resumed = append(resumed, name)
case UpdatePhaseStarting, UpdatePhaseVerifying:
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.mu.Lock()
@@ -1086,9 +1198,14 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
dir := filepath.Dir(st.ComposePath)
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
env := m.stackEnv(dir)
if e.Phase == UpdatePhaseUndoing {
m.logger.Printf("[INFO] [stacks] update %s: resuming the UNDO after a controller restart", name)
m.failAndHold(ctx, name, dir, env, rp, "resumed after a restart during the undo", &e)
return
}
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error())
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error(), &e)
return
}
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
+20 -10
View File
@@ -36,6 +36,7 @@ type fakeGuards struct {
holdRP UpdateRestorePoint
pinAtDump string
stackDir string
undoState string // what the last hold said a failed undo left (v0.263.0)
}
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
@@ -85,14 +86,14 @@ func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
}
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
}
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint) error {
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint, undoState string) error {
f.note("HoldAfterFailedUpdate")
f.mu.Lock()
defer f.mu.Unlock()
if f.holdErr != nil {
return f.holdErr
}
f.held, f.holdWhy, f.holdRP = true, "HELD-SENTENCE", rp
f.held, f.holdWhy, f.holdRP, f.undoState = true, "HELD-SENTENCE", rp, undoState
return nil
}
@@ -222,7 +223,8 @@ func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
if g.pinAtDump != "nextcloud:31.0.14-apache" {
t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump)
}
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
// v0.263.0: `stop` between the pull and `up` is the undo's copy window (the app stops there anyway).
if got, want := strings.Join(c.list(), " | "), "pull | stop | up -d --remove-orphans"; got != want {
t.Errorf("compose calls = %q, want %q", got, want)
}
// R-475: the preflight asks only whether a backup could be taken; the job reads the copies once.
@@ -407,6 +409,11 @@ func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) {
// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test
// then fails: no hold, and the customer is not told the route back.
//
// v0.263.0: a failed health check is UNDONE first (undo.go). Here the old version fails too (the same
// fake health answers the undo), so this is now "the undo failed as well → HOLD, saying so". The pin is
// BACK on the old version because the undo put the old definition back before it checked it; the case
// where no undo can be attempted — the pin stays new — is TestSlice4_G_InterruptedAfterUpResumesTheHealthWait.
func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
@@ -430,14 +437,17 @@ func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
if st.HoldReason != "HELD-SENTENCE" {
t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("the pin must STAY on the new version (its migration may have run), got %q", got)
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the undo put the old definition back before its check failed; the pin must say so, got %q", got)
}
// The `logs` call between `up` and `down` is R-621: the hold keeps the app's own log BEFORE the
// `down` destroys it. The order is the assertion — a capture after the `down` would read empty,
// which is exactly how two drill nights lost the only evidence of why an update failed.
if got := strings.Join(c.list(), " | "); got != "pull | up -d --remove-orphans | logs --no-color --tail 400 | down" {
t.Errorf("the failed app must be stopped, and its log kept FIRST; compose calls = %q", got)
if g.undoState != UndoStateNotStarted {
t.Errorf("the hold must say the undo was tried and the old version did not start, got undo state %q", g.undoState)
}
// The `logs` call before each `down` is R-621: the hold keeps the app's own log BEFORE the `down`
// destroys it — once for the new version, once for the old one the undo tried. The order is the
// assertion — a capture after the `down` would read empty.
if got := strings.Join(c.list(), " | "); got != "pull | stop | up -d --remove-orphans | logs --no-color --tail 400 | down | up -d --remove-orphans | logs --no-color --tail 400 | down" {
t.Errorf("the failed app must be stopped, its log kept FIRST, the undo tried and its log kept too; compose calls = %q", got)
}
if journalExists(m) {
t.Error("the journal is cleared once the hold (the durable record) is written")
+17
View File
@@ -833,6 +833,13 @@ func (s *Server) appDetailHandler(w http.ResponseWriter, r *http.Request, slug s
}
}
// v0.263.0 (09 §3 decision 15): the box undid a failed update by itself — one line under the badge,
// in the REQUEST's language (a finished sentence passed as page data must be rendered per request,
// never taken from a Hungarian store — the composed-sentence class, R-573/R-590/R-596/R-598).
if found.Deployed && found.AppConfig != nil && found.AppConfig.LastUpdateUndone != nil {
data["UpdateUndoneLine"] = s.msg(r, "app_info.update_undone", found.Name, undoneWhen(found.AppConfig.LastUpdateUndone.At))
}
s.executeTemplate(w, r, "app_info", data)
}
@@ -3503,3 +3510,13 @@ func generateFileBrowserCompose(domain string, storageMounts []string) string {
func generateFileBrowserConfig(paths []settings.StoragePath, importSource bool) string {
return infra.RenderFileBrowserConfig(paths, importSource)
}
// undoneWhen renders last_update_undone.at for the page line, in the box's time zone; the raw value
// when it does not parse (a record is never hidden for its format).
func undoneWhen(rfc3339 string) string {
t, err := time.Parse(time.RFC3339, rfc3339)
if err != nil {
return rfc3339
}
return t.In(getTimezone()).Format("2006-01-02 15:04")
}
@@ -34,6 +34,8 @@
<div class="alert alert-error" style="margin-top:1rem" data-held="true">{{.Stack.HoldReason}} <a href="/backups/apps" class="btn btn-sm btn-outline">{{T "app_info.mentesek"}}</a></div>
{{else if .Stack.UpdateError}}
<div class="alert alert-warning" style="margin-top:1rem" data-update-error="true">{{.Stack.UpdateError}}</div>
{{else if .UpdateUndoneLine}}
<div class="alert alert-info" style="margin-top:1rem" data-update-undone="true">{{.UpdateUndoneLine}}</div>
{{end}}
{{if .MissingStorageLabel}}
+48
View File
@@ -0,0 +1,48 @@
package web
import (
"html"
"net/http/httptest"
"os"
"path/filepath"
"strings"
"testing"
)
// v0.263.0 — after the box UNDID a failed update, the app page carries one line saying so, in the
// REQUEST's language. Driven through the REAL handler (appDetailHandler) and the REAL template, both
// languages, and the other language's sentence asserted GONE — the composed-sentence class
// (R-573/R-590/R-596/R-598) is only found this way.
//
// COMPANION RED-PROOF (REPORT.md): drop the UpdateUndoneLine block from appDetailHandler — neither
// language then carries the line, and this test fails on the first assertion.
func TestUndo_PageSaysTheBoxPutTheAppBack(t *testing.T) {
s, _ := credsHarness(t)
sd := filepath.Join(s.cfg.Paths.StacksDir, "crafty")
app := "deployed: true\nlast_update_undone:\n to:\n web: crafty:2.0.0\n at: \"2026-09-23T08:03:00Z\"\n why: \"not healthy\"\n"
if err := os.WriteFile(filepath.Join(sd, "app.yaml"), []byte(app), 0o644); err != nil {
t.Fatal(err)
}
_ = s.stackMgr.ScanStacks()
s.loadTemplates()
render := func(lang string) string {
rr := httptest.NewRecorder()
s.appDetailHandler(rr, httptest.NewRequest("GET", "/apps/crafty?lang="+lang, nil), "crafty")
return html.UnescapeString(rr.Body.String())
}
en, hu := render("en"), render("hu")
const enLine, huLine = "The box put back the previous version and its data automatically", "A doboz automatikusan visszaállította az előző változatot és az adatokat"
if !strings.Contains(en, enLine) || !strings.Contains(en, "crafty") {
t.Errorf("the English page must carry the undone line")
}
if strings.Contains(en, huLine) {
t.Errorf("the Hungarian sentence must be GONE from the English page")
}
if !strings.Contains(hu, huLine) || strings.Contains(hu, enLine) {
t.Errorf("the Hungarian page must carry the Hungarian line and not the English one")
}
if !strings.Contains(hu, "data-update-undone") {
t.Errorf("the line must render in its own marked alert")
}
}