Files
felhom-controller/controller/internal/stacks/undo.go
T
admin 2cd66663f3
gates / gates (push) Successful in 25s
v0.263.2: the undo keeps the probe of the pinned version (R-637)
Found live on 9202 (romm): .felhom.yml flows into the stack dir on every
catalog sync, so "the old .felhom.yml" saved at update time was already the
new one, and the serving old version was judged with the new probe.

New record applied-meta/.felhom.yml, written whenever a version is pinned
(deploy, adoption, pin advance) and put back by the undo, like
applied-compose.yml. The fixture now places the new file at sync time.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-23 11:53:53 +02:00

427 lines
19 KiB
Go

package stacks
import (
"context"
"fmt"
"os"
"path/filepath"
"strconv"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
)
// ── The undo (09 §3 decision 15, v0.263.0) ─────────────────────────────────────────────────────────
//
// WHAT IT REPLACES. Until v0.262.1 a failed health check after `up` stopped the app and HELD it, and
// the household's only way back was a restore from a backup tier (§6.1, "THE ABORT DECISION"). Decision
// 15 replaced that: the box puts the old version back ITSELF, together with the data exactly as it was
// seconds before the update, and holds only if that undo fails too.
//
// WHY THE OLD RULING DOES NOT BIND THIS. The old version refuses to start on data the new one migrated
// (Nextcloud, docmost, RomM — §4, and the 2026-09-23 spike). The undo does not put the old image on
// migrated data: it puts back the PRE-MIGRATION data too, so the old version meets the data it knows.
// Measured by hand on docmost, romm and vikunja before a line of this was written
// (felhom.eu/documentation/audits/update-rulings-2026-09-23/ and undo-bakeoff-2026-09-23/).
//
// THE COPY IS A FOLDER COPY, chosen by the bake-off (decision 19). After the pull and just before
// `up` — where the app is stopped anyway to be recreated — every NAMED volume the app owns is copied
// with `cp -a` into a sibling volume by a helper container, which writes a finished-marker LAST. The
// dump-and-load route passed too, but an app with no database server gets no dump at all, so it would
// have needed this copy anyway.
//
// THREE RULES, each earned by a measurement:
//
// 1. FILES ON DISK ARE NEVER TOUCHED. Only named volumes are copied and put back. A bind-mounted
// folder — photos, documents, the household's drive — is never in the copy and never moved by an
// undo. An app whose data is only in such folders gets its old version back and nothing else.
// 2. A COPY COUNTS ONLY WITH ITS FINISHED-MARKER. Killing `docker run` does not stop the copy (the
// container runs on — measured on docmost, undo-bakeoff docmost-60); a copy container killed
// mid-way leaves fewer bytes and no marker. The marker is written after `cp -a` and `sync` both
// succeed, so its presence is the only evidence the copy is whole. Pinned by
// TestUndo_CutOffCopyIsRefusedBeforeAnythingMoves.
// 3. THE UNDO READS ITS OWN COPIES, NEVER THE RECOVERY UNIT. Measured 2026-09-23: ten seconds after
// a hold was lifted, the unit was re-captured with the NEW definition (R-645); a pin-back that read
// it started the new version on the restored data. The previous compose, applied definition, pin
// AND `.felhom.yml` are kept in the stack dir until the undo is over (R-639).
//
// And the old version is checked with the OLD `.felhom.yml` probe: the new one may name a port the
// old version never answers. Pinned by TestUndo_UsesTheOldProbe.
// Undo phases and outcomes.
const (
UpdatePhaseCopying = "copying"
UpdatePhaseUndoing = "undoing"
UpdatePhaseUndone = "undone"
// UndoState* is what a FAILED undo left the data as — carried to the hold so the household is
// told the truth about it. "" means the undo was not attempted (an update journaled by an older
// controller, resumed after an upgrade).
UndoStateUntouched = "untouched" // the copy was unusable; nothing was put back
UndoStateHalf = "half" // putting the copy back stopped part-way; the copy is kept
UndoStateNotStarted = "not_started" // data and definition put back; the old version did not come up
)
// undoCopyLabel marks every copy volume with the app it belongs to, so a removal can find and delete
// it (a copy carries no compose-project label, deliberately: it must never be mistaken for the app's
// own volume by the backup legs or the removal's volume accounting).
const undoCopyLabel = "felhom.undo-copy-of"
// undoCopyMarker is written last into a copy volume, beside the data (never inside it).
const undoCopyMarker = "felhom-undo-complete"
// preUpdateMetaDir holds the previous version's .felhom.yml for the undo. A DIRECTORY, so
// LoadMetadata can read it; the syncer copies only two files into the stack dir's root and never
// touches a subdirectory.
const preUpdateMetaDir = "pre-update-meta"
// appliedMetaDir (v0.263.2) holds the .felhom.yml that belongs to the PINNED version — the probe the
// running version answers. Written when a version is pinned (deploy, adoption, a guarded update) and
// put back by an undo. WHY IT EXISTS: `.felhom.yml` flows into the stack dir on every catalog sync
// (§5.4 — it is never frozen), so by the time anyone presses Update the stack dir already holds the
// NEW version's file. MEASURED LIVE on 9202 2026-09-23: v0.263.1 saved "the old .felhom.yml" at update
// time, got the new probe (romm :8999), and judged the serving old version "did not start".
const appliedMetaDir = "applied-meta"
// storeAppliedMeta records the .felhom.yml of the version just pinned. A failure is logged by the
// caller and never fails the act — without it the undo falls back to the current file, loudly.
func storeAppliedMeta(stackDir string, src []byte) error {
if len(src) == 0 {
return fmt.Errorf("refusing to store an empty applied .felhom.yml")
}
d := filepath.Join(stackDir, appliedMetaDir)
if err := os.MkdirAll(d, 0o755); err != nil {
return err
}
tmp := filepath.Join(d, ".felhom.yml.tmp")
if err := os.WriteFile(tmp, src, 0o644); err != nil {
return err
}
return os.Rename(tmp, filepath.Join(d, ".felhom.yml"))
}
// storeAppliedMetaFrom copies a .felhom.yml file into the applied record, logging (never failing).
func (m *Manager) storeAppliedMetaFrom(name, stackDir, src string) {
b, err := os.ReadFile(src)
if err == nil {
err = storeAppliedMeta(stackDir, b)
}
if err != nil {
m.logger.Printf("[WARN] [stacks] pin %s: could not record the pinned version's .felhom.yml (%v) — an undo would check health with whatever file is current then", name, err)
}
}
// undoHelperImage is the helper the backup legs already use for volume tars (backup.go, restore.go),
// so no new image is introduced onto a box.
const undoHelperImage = "alpine"
// undoCopy pairs an app volume with its last-second copy.
type undoCopy struct {
Volume string `json:"volume"`
Copy string `json:"copy"`
}
// UpdateUndone is app.yaml's record of the last update the box undid (decision 15: no automatic
// retry until the catalog moves; the button stays usable for a person). Written only by a
// successful undo; cleared by the next successful update.
type UpdateUndone struct {
To map[string]string `yaml:"to" json:"to"`
At string `yaml:"at" json:"at"`
Why string `yaml:"why,omitempty" json:"why,omitempty"`
}
// volumeCopier is the process boundary of the undo's copy. Production is dockerVolumeCopier; tests
// inject a fake and never touch docker.
type volumeCopier interface {
ProjectVolumes(project string) ([]string, error)
VolumeBytes(vol string) (int64, error)
Copy(src, dst, app string) error
Complete(copyVol string) bool
Restore(copyVol, vol string) error
Remove(vol string) error
CopiesOf(app string) []string
}
func (m *Manager) copier() volumeCopier {
if m.undoCopier != nil {
return m.undoCopier
}
return dockerVolumeCopier{m: m}
}
type dockerVolumeCopier struct{ m *Manager }
func (d dockerVolumeCopier) ProjectVolumes(project string) ([]string, error) {
out, err := d.m.execCommand("docker", "volume", "ls", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Name}}")
if err != nil {
return nil, err
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols, nil
}
func (d dockerVolumeCopier) VolumeBytes(vol string) (int64, error) {
out, err := d.m.execCommand("docker", "run", "--rm", "-v", vol+":/v:ro", undoHelperImage, "du", "-sb", "/v")
if err != nil {
return 0, err
}
f := strings.Fields(out)
if len(f) == 0 {
return 0, fmt.Errorf("du printed nothing for %s", vol)
}
return strconv.ParseInt(f[0], 10, 64)
}
// Copy runs in the FOREGROUND so the helper container's own exit status is the answer; the marker is
// written only after cp and sync both succeed.
func (d dockerVolumeCopier) Copy(src, dst, app string) error {
if _, err := d.m.execCommand("docker", "volume", "create", "--label", undoCopyLabel+"="+app, dst); err != nil {
return err
}
_, err := d.m.execCommand("docker", "run", "--rm", "-v", src+":/from:ro", "-v", dst+":/to", undoHelperImage,
"sh", "-c", "mkdir -p /to/data && cp -a /from/. /to/data/ && sync && touch /to/"+undoCopyMarker)
return err
}
func (d dockerVolumeCopier) Complete(copyVol string) bool {
_, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/c:ro", undoHelperImage, "test", "-f", "/c/"+undoCopyMarker)
return err == nil
}
// Restore empties the app's volume and copies the data back. The marker is checked again INSIDE the
// same helper, so a copy that lost it between the check and the restore is never poured in.
func (d dockerVolumeCopier) Restore(copyVol, vol string) error {
_, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/from:ro", "-v", vol+":/to", undoHelperImage,
"sh", "-c", "test -f /from/"+undoCopyMarker+" && find /to -mindepth 1 -delete && cp -a /from/data/. /to/ && sync")
return err
}
func (d dockerVolumeCopier) Remove(vol string) error {
_, err := d.m.execCommand("docker", "volume", "rm", "-f", vol)
return err
}
func (d dockerVolumeCopier) CopiesOf(app string) []string {
out, err := d.m.execCommand("docker", "volume", "ls", "-q", "--filter", "label="+undoCopyLabel+"="+app)
if err != nil {
return nil
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols
}
// undoCopyName is `<volume>.pre-update-<stamp>` — a legal Docker volume name, unique per update.
func undoCopyName(vol string, at time.Time) string {
return vol + ".pre-update-" + at.UTC().Format("20060102T150405Z")
}
// undoMsg renders a born-as-key sentence of the update job. Hungarian, like every other sentence the
// job writes today (R-606 localises them together).
func undoMsg(key string, args ...interface{}) string { return util.Text("hu", key, args...) }
// planUndoCopies lists the app's named volumes and refuses — before anything moves — when their copy
// would breach the disk floor the update already keeps (decision 19's disk limit).
func (m *Manager) planUndoCopies(name string) ([]string, error) {
vols, err := m.copier().ProjectVolumes(name)
if err != nil {
return nil, fmt.Errorf("listing the app's volumes: %w", err)
}
var total int64
for _, v := range vols {
b, err := m.copier().VolumeBytes(v)
if err != nil {
return nil, fmt.Errorf("sizing volume %s: %w", v, err)
}
total += b
}
needGiB := float64(total) / (1 << 30)
if free, known := m.updateDiskFree(); known && free-needGiB < updateDiskFloorGiB {
return nil, &undoSpaceError{need: needGiB, free: free}
}
m.logger.Printf("[INFO] [stacks] update %s: the undo copy will hold %d named volume(s), %.1f MiB", name, len(vols), float64(total)/(1<<20))
return vols, nil
}
type undoSpaceError struct{ need, free float64 }
func (e *undoSpaceError) Error() string {
return fmt.Sprintf("the undo copy needs %.2f GiB and %.2f GiB is free (floor %.0f GiB)", e.need, e.free, updateDiskFloorGiB)
}
// makeUndoCopies stops the app and copies every volume. The copies are journaled BEFORE each copy
// starts, so a power cut mid-copy leaves a journal that names every volume to clean up.
func (m *Manager) makeUndoCopies(name, dir string, env []string, vols []string, entry *updateJournalEntry) error {
if _, err := m.updateCompose(dir, env, "stop"); err != nil {
return fmt.Errorf("stopping the app for the copy: %w", err)
}
stamp := m.now()
for _, v := range vols {
c := undoCopy{Volume: v, Copy: undoCopyName(v, stamp)}
entry.UndoCopies = append(entry.UndoCopies, c)
if !m.enterUpdatePhase(name, entry, UpdatePhaseCopying) {
return fmt.Errorf("journal write failed before copying %s", v)
}
t0 := time.Now()
if err := m.copier().Copy(c.Volume, c.Copy, name); err != nil {
return fmt.Errorf("copying %s: %w", v, err)
}
m.logger.Printf("[INFO] [stacks] update %s: copied %s → %s in %s", name, c.Volume, c.Copy, time.Since(t0).Round(time.Millisecond))
}
entry.Copied = true
return nil
}
func (m *Manager) removeUndoCopies(name string, copies []undoCopy) {
for _, c := range copies {
if err := m.copier().Remove(c.Copy); err != nil {
m.logger.Printf("[WARN] [stacks] update %s: could not remove the undo copy %s: %v", name, c.Copy, err)
}
}
}
// RemoveUndoCopies deletes every undo copy an app still has — the removal path's hook, so a copy kept
// by a failed undo does not outlive the app.
func (m *Manager) RemoveUndoCopies(name string) int {
n := 0
for _, v := range m.copier().CopiesOf(name) {
if err := m.copier().Remove(v); err != nil {
m.logger.Printf("[WARN] [stacks] remove %s: could not delete the undo copy %s: %v", name, v, err)
continue
}
n++
}
return n
}
// savePreUpdateMeta keeps the PINNED version's .felhom.yml for the undo's health check: the applied
// record when there is one; otherwise — an app pinned before v0.263.2 — the current file, which the
// catalog sync may already have replaced with the new version's (said in the returned note).
func savePreUpdateMeta(dir string) (string, error) {
src, err := os.ReadFile(filepath.Join(dir, appliedMetaDir, ".felhom.yml"))
if err != nil {
if src, err = os.ReadFile(filepath.Join(dir, ".felhom.yml")); err != nil {
return "", err
}
err = errNoAppliedMeta
}
note := err
md := filepath.Join(dir, preUpdateMetaDir)
if err := os.MkdirAll(md, 0o755); err != nil {
return "", err
}
if err := os.WriteFile(filepath.Join(md, ".felhom.yml"), src, 0o644); err != nil {
return "", err
}
return md, note
}
// errNoAppliedMeta: the app has no applied .felhom.yml record (pinned before v0.263.2), so the undo's
// probe is taken from the current file. The copy WAS made — callers log this and carry on.
var errNoAppliedMeta = fmt.Errorf("no applied .felhom.yml for the pinned version (pinned before v0.263.2) — the undo will probe with the current file")
func (m *Manager) undoHealth(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string) {
if m.updateUndoHealthFn != nil {
return m.updateUndoHealthFn(ctx, name, timeout, meta)
}
// A test that injects only the update's health seam gets the same answer for the undo (production
// injects neither and always takes waitUpdateHealthyMeta).
if m.updateHealthFn != nil {
return m.updateHealthFn(ctx, name, timeout)
}
return m.waitUpdateHealthyMeta(ctx, name, timeout, meta)
}
// tryUndo puts the pre-update data and definition back and checks the old version with its own probe.
// It returns "" when the app is healthy on its old version again, else the state the data is in.
//
// ORDER IS THE DESIGN: every copy is validated BEFORE any is poured back (a cut-off copy leaves the
// data exactly as the new version left it — UndoStateUntouched); the definition is put back only after
// the data (so a failure in between reads "half" with the pin still naming the new version, which is
// what ran on that data).
func (m *Manager) tryUndo(ctx context.Context, name, dir, why string, entry *updateJournalEntry) string {
start := m.now()
if !m.enterUpdatePhase(name, entry, UpdatePhaseUndoing) {
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the undo — undoing anyway", name)
}
m.logger.Printf("[WARN] [stacks] update %s: UNDO — putting back the previous version and its %d volume copy(ies) (reason: %s)", name, len(entry.UndoCopies), why)
if _, err := m.updateCompose(dir, m.stackEnv(dir), "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the new version before the undo failed: %v", name, err)
}
for _, c := range entry.UndoCopies {
if !m.copier().Complete(c.Copy) {
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy %s has no finished-marker — it is cut off or missing; NOTHING is put back", name, c.Copy)
return UndoStateUntouched
}
}
for _, c := range entry.UndoCopies {
if err := m.copier().Restore(c.Copy, c.Volume); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: putting %s back from %s FAILED: %v — the data is mixed; the copies are kept", name, c.Volume, c.Copy, err)
return UndoStateHalf
}
}
m.restoreDefinition(name, dir, *entry)
env := m.stackEnv(dir)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: starting the previous version failed: %v", name, err)
m.captureHoldLogs(name, dir, env)
_, _ = m.updateCompose(dir, env, "down")
return UndoStateNotStarted
}
meta := LoadMetadata(dir)
if entry.PrevMeta != "" {
if _, err := os.Stat(filepath.Join(entry.PrevMeta, ".felhom.yml")); err == nil {
meta = LoadMetadata(entry.PrevMeta)
} else {
m.logger.Printf("[WARN] [stacks] update %s: the previous .felhom.yml is missing (%v) — checking with the current one", name, err)
}
}
healthy, detail := m.undoHealth(ctx, name, m.healthTimeout(), &meta)
if !healthy {
m.logger.Printf("[ERROR] [stacks] update %s: the previous version did not come up after the undo: %s", name, detail)
m.captureHoldLogs(name, dir, env)
_, _ = m.updateCompose(dir, env, "down")
return UndoStateNotStarted
}
m.recordInstalledImages(name, dir, env)
m.removeUndoCopies(name, entry.UndoCopies)
m.recordUpdateUndone(name, dir, &UpdateUndone{To: entry.NewPin, At: start.UTC().Format(time.RFC3339), Why: why})
m.removePreUpdateCopies(dir)
_ = m.RefreshStatus()
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseUndone, "")
m.logger.Printf("[INFO] [stacks] update %s: UNDONE in %s — the previous version is running on the data from before the update (%s)", name, m.now().Sub(start).Round(time.Second), detail)
return ""
}
// recordUpdateUndone writes (or, with nil, clears) app.yaml's last_update_undone. A failed write is
// logged and never fails the undo — it is a record, and the app is already back.
func (m *Manager) recordUpdateUndone(name, dir string, u *UpdateUndone) {
cfg := LoadAppConfig(dir)
if cfg == nil || (u == nil && cfg.LastUpdateUndone == nil) {
return
}
cfg.LastUpdateUndone = u
meta := LoadMetadata(dir)
if err := SaveAppConfig(dir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: recording last_update_undone failed: %v", name, err)
return
}
m.mu.Lock()
if st, ok := m.stacks[name]; ok && st.AppConfig != nil {
st.AppConfig.LastUpdateUndone = u
}
m.mu.Unlock()
}