v0.263.0: a failed update puts the app back by itself (09 decision 15, R-637)
gates / gates (push) Successful in 26s

The guarded update gains a folder copy of the app's named volumes, taken
after the pull where the app stops anyway (decision 19, chosen by the
2026-09-23 bake-off). On a failed health check the box undoes: every copy
validated by its finished-marker first, volumes refilled, definition and pin
from the job's own pre-update copies, the old version checked with the OLD
.felhom.yml probe. It holds only if the undo fails, and the hold sentence
says so and what state the data is in. Bind-mounted folders are never
touched.

- R-637 built; R-638/R-640/R-641 do not arise with a folder copy; R-639
  (pre-update copies incl. .felhom.yml kept until the undo is over).
- journal phases copying/undoing with power-cut recovery.
- app.yaml last_update_undone + one line on the app page (hu/en).
- R-642: start/restart never answer "completed".
- Removal deletes kept undo copies.

MinAgent unchanged (0.131.0). Nine red-proofs in REPORT.md.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-23 11:12:49 +02:00
parent b9deec1907
commit 8fc2b4a1a9
26 changed files with 1326 additions and 69 deletions
+20 -10
View File
@@ -36,6 +36,7 @@ type fakeGuards struct {
holdRP UpdateRestorePoint
pinAtDump string
stackDir string
undoState string // what the last hold said a failed undo left (v0.263.0)
}
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
@@ -85,14 +86,14 @@ func (f *fakeGuards) SafetyDump(context.Context, string) ([]string, error) {
}
return []string{"/fake/pre-restore-x.sql"}, f.dumpErr
}
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint) error {
func (f *fakeGuards) HoldAfterFailedUpdate(_ string, _ time.Time, rp UpdateRestorePoint, undoState string) error {
f.note("HoldAfterFailedUpdate")
f.mu.Lock()
defer f.mu.Unlock()
if f.holdErr != nil {
return f.holdErr
}
f.held, f.holdWhy, f.holdRP = true, "HELD-SENTENCE", rp
f.held, f.holdWhy, f.holdRP, f.undoState = true, "HELD-SENTENCE", rp, undoState
return nil
}
@@ -222,7 +223,8 @@ func TestSlice4_A_SuccessIsDeclaredOnlyAfterHealth(t *testing.T) {
if g.pinAtDump != "nextcloud:31.0.14-apache" {
t.Errorf("the safety dump must run BEFORE the pin moves (\"a minute ago\"); the pin at dump time was %q", g.pinAtDump)
}
if got, want := strings.Join(c.list(), " | "), "pull | up -d --remove-orphans"; got != want {
// v0.263.0: `stop` between the pull and `up` is the undo's copy window (the app stops there anyway).
if got, want := strings.Join(c.list(), " | "), "pull | stop | up -d --remove-orphans"; got != want {
t.Errorf("compose calls = %q, want %q", got, want)
}
// R-475: the preflight asks only whether a backup could be taken; the job reads the copies once.
@@ -407,6 +409,11 @@ func TestSlice4_E_PullFailurePutsThePinBack(t *testing.T) {
// COMPANION RED-PROOF 3 (REPORT.md): remove the HoldAfterFailedUpdate call from failAndHold. This test
// then fails: no hold, and the customer is not told the route back.
//
// v0.263.0: a failed health check is UNDONE first (undo.go). Here the old version fails too (the same
// fake health answers the undo), so this is now "the undo failed as well → HOLD, saying so". The pin is
// BACK on the old version because the undo put the old definition back before it checked it; the case
// where no undo can be attempted — the pin stays new — is TestSlice4_G_InterruptedAfterUpResumesTheHealthWait.
func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
m, dir, g, c := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
@@ -430,14 +437,17 @@ func TestSlice4_F_HealthFailureHoldsTheAppAndKeepsTheNewPin(t *testing.T) {
if st.HoldReason != "HELD-SENTENCE" {
t.Errorf("GetStack must carry the hold text, got %q", st.HoldReason)
}
if got := pinOf(t, dir); got != "nextcloud:34.0.1-apache" {
t.Errorf("the pin must STAY on the new version (its migration may have run), got %q", got)
if got := pinOf(t, dir); got != "nextcloud:31.0.14-apache" {
t.Errorf("the undo put the old definition back before its check failed; the pin must say so, got %q", got)
}
// The `logs` call between `up` and `down` is R-621: the hold keeps the app's own log BEFORE the
// `down` destroys it. The order is the assertion — a capture after the `down` would read empty,
// which is exactly how two drill nights lost the only evidence of why an update failed.
if got := strings.Join(c.list(), " | "); got != "pull | up -d --remove-orphans | logs --no-color --tail 400 | down" {
t.Errorf("the failed app must be stopped, and its log kept FIRST; compose calls = %q", got)
if g.undoState != UndoStateNotStarted {
t.Errorf("the hold must say the undo was tried and the old version did not start, got undo state %q", g.undoState)
}
// The `logs` call before each `down` is R-621: the hold keeps the app's own log BEFORE the `down`
// destroys it — once for the new version, once for the old one the undo tried. The order is the
// assertion — a capture after the `down` would read empty.
if got := strings.Join(c.list(), " | "); got != "pull | stop | up -d --remove-orphans | logs --no-color --tail 400 | down | up -d --remove-orphans | logs --no-color --tail 400 | down" {
t.Errorf("the failed app must be stopped, its log kept FIRST, the undo tried and its log kept too; compose calls = %q", got)
}
if journalExists(m) {
t.Error("the journal is cleared once the hold (the durable record) is written")