package stacks import ( "context" "fmt" "os" "path/filepath" "strconv" "strings" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/util" ) // ── The undo (09 §3 decision 15, v0.263.0) ───────────────────────────────────────────────────────── // // WHAT IT REPLACES. Until v0.262.1 a failed health check after `up` stopped the app and HELD it, and // the household's only way back was a restore from a backup tier (§6.1, "THE ABORT DECISION"). Decision // 15 replaced that: the box puts the old version back ITSELF, together with the data exactly as it was // seconds before the update, and holds only if that undo fails too. // // WHY THE OLD RULING DOES NOT BIND THIS. The old version refuses to start on data the new one migrated // (Nextcloud, docmost, RomM — §4, and the 2026-09-23 spike). The undo does not put the old image on // migrated data: it puts back the PRE-MIGRATION data too, so the old version meets the data it knows. // Measured by hand on docmost, romm and vikunja before a line of this was written // (felhom.eu/documentation/audits/update-rulings-2026-09-23/ and undo-bakeoff-2026-09-23/). // // THE COPY IS A FOLDER COPY, chosen by the bake-off (decision 19). After the pull and just before // `up` — where the app is stopped anyway to be recreated — every NAMED volume the app owns is copied // with `cp -a` into a sibling volume by a helper container, which writes a finished-marker LAST. The // dump-and-load route passed too, but an app with no database server gets no dump at all, so it would // have needed this copy anyway. // // THREE RULES, each earned by a measurement: // // 1. FILES ON DISK ARE NEVER TOUCHED. Only named volumes are copied and put back. A bind-mounted // folder — photos, documents, the household's drive — is never in the copy and never moved by an // undo. An app whose data is only in such folders gets its old version back and nothing else. // 2. A COPY COUNTS ONLY WITH ITS FINISHED-MARKER. Killing `docker run` does not stop the copy (the // container runs on — measured on docmost, undo-bakeoff docmost-60); a copy container killed // mid-way leaves fewer bytes and no marker. The marker is written after `cp -a` and `sync` both // succeed, so its presence is the only evidence the copy is whole. Pinned by // TestUndo_CutOffCopyIsRefusedBeforeAnythingMoves. // 3. THE UNDO READS ITS OWN COPIES, NEVER THE RECOVERY UNIT. Measured 2026-09-23: ten seconds after // a hold was lifted, the unit was re-captured with the NEW definition (R-645); a pin-back that read // it started the new version on the restored data. The previous compose, applied definition, pin // AND `.felhom.yml` are kept in the stack dir until the undo is over (R-639). // // And the old version is checked with the OLD `.felhom.yml` probe: the new one may name a port the // old version never answers. Pinned by TestUndo_UsesTheOldProbe. // Undo phases and outcomes. const ( UpdatePhaseCopying = "copying" UpdatePhaseUndoing = "undoing" UpdatePhaseUndone = "undone" // UndoState* is what a FAILED undo left the data as — carried to the hold so the household is // told the truth about it. "" means the undo was not attempted (an update journaled by an older // controller, resumed after an upgrade). UndoStateUntouched = "untouched" // the copy was unusable; nothing was put back UndoStateHalf = "half" // putting the copy back stopped part-way; the copy is kept UndoStateNotStarted = "not_started" // data and definition put back; the old version did not come up ) // undoCopyLabel marks every copy volume with the app it belongs to, so a removal can find and delete // it (a copy carries no compose-project label, deliberately: it must never be mistaken for the app's // own volume by the backup legs or the removal's volume accounting). const undoCopyLabel = "felhom.undo-copy-of" // undoCopyMarker is written last into a copy volume, beside the data (never inside it). const undoCopyMarker = "felhom-undo-complete" // preUpdateMetaDir holds the previous .felhom.yml. A DIRECTORY, so LoadMetadata can read it; the // syncer copies only two files into the stack dir's root and never touches a subdirectory. const preUpdateMetaDir = "pre-update-meta" // undoHelperImage is the helper the backup legs already use for volume tars (backup.go, restore.go), // so no new image is introduced onto a box. const undoHelperImage = "alpine" // undoCopy pairs an app volume with its last-second copy. type undoCopy struct { Volume string `json:"volume"` Copy string `json:"copy"` } // UpdateUndone is app.yaml's record of the last update the box undid (decision 15: no automatic // retry until the catalog moves; the button stays usable for a person). Written only by a // successful undo; cleared by the next successful update. type UpdateUndone struct { To map[string]string `yaml:"to" json:"to"` At string `yaml:"at" json:"at"` Why string `yaml:"why,omitempty" json:"why,omitempty"` } // volumeCopier is the process boundary of the undo's copy. Production is dockerVolumeCopier; tests // inject a fake and never touch docker. type volumeCopier interface { ProjectVolumes(project string) ([]string, error) VolumeBytes(vol string) (int64, error) Copy(src, dst, app string) error Complete(copyVol string) bool Restore(copyVol, vol string) error Remove(vol string) error CopiesOf(app string) []string } func (m *Manager) copier() volumeCopier { if m.undoCopier != nil { return m.undoCopier } return dockerVolumeCopier{m: m} } type dockerVolumeCopier struct{ m *Manager } func (d dockerVolumeCopier) ProjectVolumes(project string) ([]string, error) { out, err := d.m.execCommand("docker", "volume", "ls", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Name}}") if err != nil { return nil, err } var vols []string for _, l := range strings.Split(out, "\n") { if l = strings.TrimSpace(l); l != "" { vols = append(vols, l) } } return vols, nil } func (d dockerVolumeCopier) VolumeBytes(vol string) (int64, error) { out, err := d.m.execCommand("docker", "run", "--rm", "-v", vol+":/v:ro", undoHelperImage, "du", "-sb", "/v") if err != nil { return 0, err } f := strings.Fields(out) if len(f) == 0 { return 0, fmt.Errorf("du printed nothing for %s", vol) } return strconv.ParseInt(f[0], 10, 64) } // Copy runs in the FOREGROUND so the helper container's own exit status is the answer; the marker is // written only after cp and sync both succeed. func (d dockerVolumeCopier) Copy(src, dst, app string) error { if _, err := d.m.execCommand("docker", "volume", "create", "--label", undoCopyLabel+"="+app, dst); err != nil { return err } _, err := d.m.execCommand("docker", "run", "--rm", "-v", src+":/from:ro", "-v", dst+":/to", undoHelperImage, "sh", "-c", "mkdir -p /to/data && cp -a /from/. /to/data/ && sync && touch /to/"+undoCopyMarker) return err } func (d dockerVolumeCopier) Complete(copyVol string) bool { _, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/c:ro", undoHelperImage, "test", "-f", "/c/"+undoCopyMarker) return err == nil } // Restore empties the app's volume and copies the data back. The marker is checked again INSIDE the // same helper, so a copy that lost it between the check and the restore is never poured in. func (d dockerVolumeCopier) Restore(copyVol, vol string) error { _, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/from:ro", "-v", vol+":/to", undoHelperImage, "sh", "-c", "test -f /from/"+undoCopyMarker+" && find /to -mindepth 1 -delete && cp -a /from/data/. /to/ && sync") return err } func (d dockerVolumeCopier) Remove(vol string) error { _, err := d.m.execCommand("docker", "volume", "rm", "-f", vol) return err } func (d dockerVolumeCopier) CopiesOf(app string) []string { out, err := d.m.execCommand("docker", "volume", "ls", "-q", "--filter", "label="+undoCopyLabel+"="+app) if err != nil { return nil } var vols []string for _, l := range strings.Split(out, "\n") { if l = strings.TrimSpace(l); l != "" { vols = append(vols, l) } } return vols } // undoCopyName is `.pre-update-` — a legal Docker volume name, unique per update. func undoCopyName(vol string, at time.Time) string { return vol + ".pre-update-" + at.UTC().Format("20060102T150405Z") } // undoMsg renders a born-as-key sentence of the update job. Hungarian, like every other sentence the // job writes today (R-606 localises them together). func undoMsg(key string, args ...interface{}) string { return util.Text("hu", key, args...) } // planUndoCopies lists the app's named volumes and refuses — before anything moves — when their copy // would breach the disk floor the update already keeps (decision 19's disk limit). func (m *Manager) planUndoCopies(name string) ([]string, error) { vols, err := m.copier().ProjectVolumes(name) if err != nil { return nil, fmt.Errorf("listing the app's volumes: %w", err) } var total int64 for _, v := range vols { b, err := m.copier().VolumeBytes(v) if err != nil { return nil, fmt.Errorf("sizing volume %s: %w", v, err) } total += b } needGiB := float64(total) / (1 << 30) if free, known := m.updateDiskFree(); known && free-needGiB < updateDiskFloorGiB { return nil, &undoSpaceError{need: needGiB, free: free} } m.logger.Printf("[INFO] [stacks] update %s: the undo copy will hold %d named volume(s), %.1f MiB", name, len(vols), float64(total)/(1<<20)) return vols, nil } type undoSpaceError struct{ need, free float64 } func (e *undoSpaceError) Error() string { return fmt.Sprintf("the undo copy needs %.2f GiB and %.2f GiB is free (floor %.0f GiB)", e.need, e.free, updateDiskFloorGiB) } // makeUndoCopies stops the app and copies every volume. The copies are journaled BEFORE each copy // starts, so a power cut mid-copy leaves a journal that names every volume to clean up. func (m *Manager) makeUndoCopies(name, dir string, env []string, vols []string, entry *updateJournalEntry) error { if _, err := m.updateCompose(dir, env, "stop"); err != nil { return fmt.Errorf("stopping the app for the copy: %w", err) } stamp := m.now() for _, v := range vols { c := undoCopy{Volume: v, Copy: undoCopyName(v, stamp)} entry.UndoCopies = append(entry.UndoCopies, c) if !m.enterUpdatePhase(name, entry, UpdatePhaseCopying) { return fmt.Errorf("journal write failed before copying %s", v) } t0 := time.Now() if err := m.copier().Copy(c.Volume, c.Copy, name); err != nil { return fmt.Errorf("copying %s: %w", v, err) } m.logger.Printf("[INFO] [stacks] update %s: copied %s → %s in %s", name, c.Volume, c.Copy, time.Since(t0).Round(time.Millisecond)) } entry.Copied = true return nil } func (m *Manager) removeUndoCopies(name string, copies []undoCopy) { for _, c := range copies { if err := m.copier().Remove(c.Copy); err != nil { m.logger.Printf("[WARN] [stacks] update %s: could not remove the undo copy %s: %v", name, c.Copy, err) } } } // RemoveUndoCopies deletes every undo copy an app still has — the removal path's hook, so a copy kept // by a failed undo does not outlive the app. func (m *Manager) RemoveUndoCopies(name string) int { n := 0 for _, v := range m.copier().CopiesOf(name) { if err := m.copier().Remove(v); err != nil { m.logger.Printf("[WARN] [stacks] remove %s: could not delete the undo copy %s: %v", name, v, err) continue } n++ } return n } // savePreUpdateMeta keeps the previous .felhom.yml for the undo's health check. func savePreUpdateMeta(dir string) (string, error) { src, err := os.ReadFile(filepath.Join(dir, ".felhom.yml")) if err != nil { return "", err } md := filepath.Join(dir, preUpdateMetaDir) if err := os.MkdirAll(md, 0o755); err != nil { return "", err } return md, os.WriteFile(filepath.Join(md, ".felhom.yml"), src, 0o644) } func (m *Manager) undoHealth(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string) { if m.updateUndoHealthFn != nil { return m.updateUndoHealthFn(ctx, name, timeout, meta) } // A test that injects only the update's health seam gets the same answer for the undo (production // injects neither and always takes waitUpdateHealthyMeta). if m.updateHealthFn != nil { return m.updateHealthFn(ctx, name, timeout) } return m.waitUpdateHealthyMeta(ctx, name, timeout, meta) } // tryUndo puts the pre-update data and definition back and checks the old version with its own probe. // It returns "" when the app is healthy on its old version again, else the state the data is in. // // ORDER IS THE DESIGN: every copy is validated BEFORE any is poured back (a cut-off copy leaves the // data exactly as the new version left it — UndoStateUntouched); the definition is put back only after // the data (so a failure in between reads "half" with the pin still naming the new version, which is // what ran on that data). func (m *Manager) tryUndo(ctx context.Context, name, dir, why string, entry *updateJournalEntry) string { start := m.now() if !m.enterUpdatePhase(name, entry, UpdatePhaseUndoing) { m.logger.Printf("[ERROR] [stacks] update %s: could not journal the undo — undoing anyway", name) } m.logger.Printf("[WARN] [stacks] update %s: UNDO — putting back the previous version and its %d volume copy(ies) (reason: %s)", name, len(entry.UndoCopies), why) if _, err := m.updateCompose(dir, m.stackEnv(dir), "down"); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: stopping the new version before the undo failed: %v", name, err) } for _, c := range entry.UndoCopies { if !m.copier().Complete(c.Copy) { m.logger.Printf("[ERROR] [stacks] update %s: the undo copy %s has no finished-marker — it is cut off or missing; NOTHING is put back", name, c.Copy) return UndoStateUntouched } } for _, c := range entry.UndoCopies { if err := m.copier().Restore(c.Copy, c.Volume); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: putting %s back from %s FAILED: %v — the data is mixed; the copies are kept", name, c.Volume, c.Copy, err) return UndoStateHalf } } m.restoreDefinition(name, dir, *entry) env := m.stackEnv(dir) if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: starting the previous version failed: %v", name, err) m.captureHoldLogs(name, dir, env) _, _ = m.updateCompose(dir, env, "down") return UndoStateNotStarted } meta := LoadMetadata(dir) if entry.PrevMeta != "" { if _, err := os.Stat(filepath.Join(entry.PrevMeta, ".felhom.yml")); err == nil { meta = LoadMetadata(entry.PrevMeta) } else { m.logger.Printf("[WARN] [stacks] update %s: the previous .felhom.yml is missing (%v) — checking with the current one", name, err) } } healthy, detail := m.undoHealth(ctx, name, m.healthTimeout(), &meta) if !healthy { m.logger.Printf("[ERROR] [stacks] update %s: the previous version did not come up after the undo: %s", name, detail) m.captureHoldLogs(name, dir, env) _, _ = m.updateCompose(dir, env, "down") return UndoStateNotStarted } m.recordInstalledImages(name, dir, env) m.removeUndoCopies(name, entry.UndoCopies) m.recordUpdateUndone(name, dir, &UpdateUndone{To: entry.NewPin, At: start.UTC().Format(time.RFC3339), Why: why}) m.removePreUpdateCopies(dir) _ = m.RefreshStatus() m.clearJournal(name) m.finishUpdate(name, UpdatePhaseUndone, "") m.logger.Printf("[INFO] [stacks] update %s: UNDONE in %s — the previous version is running on the data from before the update (%s)", name, m.now().Sub(start).Round(time.Second), detail) return "" } // recordUpdateUndone writes (or, with nil, clears) app.yaml's last_update_undone. A failed write is // logged and never fails the undo — it is a record, and the app is already back. func (m *Manager) recordUpdateUndone(name, dir string, u *UpdateUndone) { cfg := LoadAppConfig(dir) if cfg == nil || (u == nil && cfg.LastUpdateUndone == nil) { return } cfg.LastUpdateUndone = u meta := LoadMetadata(dir) if err := SaveAppConfig(dir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: recording last_update_undone failed: %v", name, err) return } m.mu.Lock() if st, ok := m.stacks[name]; ok && st.AppConfig != nil { st.AppConfig.LastUpdateUndone = u } m.mu.Unlock() }