Files
felhom-controller/controller/internal/stacks/undo.go
T
admin 2caae38a71
gates / gates (push) Successful in 25s
stacks: the box converts a PostgreSQL major as a guarded-update step (09 6.4 part 10, decisions 35/37/38)
A step whose ladder entry carries engine_conversion {service, engine, from, to}
converts the database: the old engine alone, the check (owners, roles,
extensions, per-table row counts), pg_dumpall validated by its completion line,
the volume emptied only after the undo copy's marker is validated again, the new
engine alone, the load with ON_ERROR_STOP, the check again + PG_VERSION. Any
failure goes to the existing undo; a restart during converting is undone.
A PostgreSQL major move without the mark is refused before anything moves.
The old datadir's copy is kept until a backup is proven after the conversion.
17 tests, 9 red-proofs (audits/night-2026-09-26/B/).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-25 12:57:03 +02:00

678 lines
29 KiB
Go

package stacks
import (
"context"
"fmt"
"os"
"path/filepath"
"sort"
"strconv"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"gopkg.in/yaml.v3"
)
// ── The undo (09 §3 decision 15, v0.263.0) ─────────────────────────────────────────────────────────
//
// WHAT IT REPLACES. Until v0.262.1 a failed health check after `up` stopped the app and HELD it, and
// the household's only way back was a restore from a backup tier (§6.1, "THE ABORT DECISION"). Decision
// 15 replaced that: the box puts the old version back ITSELF, together with the data exactly as it was
// seconds before the update, and holds only if that undo fails too.
//
// WHY THE OLD RULING DOES NOT BIND THIS. The old version refuses to start on data the new one migrated
// (Nextcloud, docmost, RomM — §4, and the 2026-09-23 spike). The undo does not put the old image on
// migrated data: it puts back the PRE-MIGRATION data too, so the old version meets the data it knows.
// Measured by hand on docmost, romm and vikunja before a line of this was written
// (felhom.eu/documentation/audits/update-rulings-2026-09-23/ and undo-bakeoff-2026-09-23/).
//
// THE COPY IS A FOLDER COPY, chosen by the bake-off (decision 19). After the pull and just before
// `up` — where the app is stopped anyway to be recreated — every NAMED volume the app owns is copied
// with `cp -a` into a sibling volume by a helper container, which writes a finished-marker LAST. The
// dump-and-load route passed too, but an app with no database server gets no dump at all, so it would
// have needed this copy anyway.
//
// THREE RULES, each earned by a measurement:
//
// 1. FILES ON DISK ARE NEVER TOUCHED. Only named volumes are copied and put back. A bind-mounted
// folder — photos, documents, the household's drive — is never in the copy and never moved by an
// undo. An app whose data is only in such folders gets its old version back and nothing else.
// 2. A COPY COUNTS ONLY WITH ITS FINISHED-MARKER. Killing `docker run` does not stop the copy (the
// container runs on — measured on docmost, undo-bakeoff docmost-60); a copy container killed
// mid-way leaves fewer bytes and no marker. The marker is written after `cp -a` and `sync` both
// succeed, so its presence is the only evidence the copy is whole. Pinned by
// TestUndo_CutOffCopyIsRefusedBeforeAnythingMoves.
// 3. THE UNDO READS ITS OWN COPIES, NEVER THE RECOVERY UNIT. Measured 2026-09-23: ten seconds after
// a hold was lifted, the unit was re-captured with the NEW definition (R-645); a pin-back that read
// it started the new version on the restored data. The previous compose, applied definition, pin
// AND `.felhom.yml` are kept in the stack dir until the undo is over (R-639).
//
// And the old version is checked with the OLD `.felhom.yml` probe: the new one may name a port the
// old version never answers. Pinned by TestUndo_UsesTheOldProbe.
// Undo phases and outcomes.
const (
UpdatePhaseCopying = "copying"
UpdatePhaseUndoing = "undoing"
UpdatePhaseUndone = "undone"
// UndoState* is what a FAILED undo left the data as — carried to the hold so the household is
// told the truth about it. "" means the undo was not attempted (an update journaled by an older
// controller, resumed after an upgrade).
UndoStateUntouched = "untouched" // the copy was unusable; nothing was put back
UndoStateHalf = "half" // putting the copy back stopped part-way; the copy is kept
UndoStateNotStarted = "not_started" // data and definition put back; the old version did not come up
)
// undoCopyLabel marks every copy volume with the app it belongs to, so a removal can find and delete
// it (a copy carries no compose-project label, deliberately: it must never be mistaken for the app's
// own volume by the backup legs or the removal's volume accounting).
const undoCopyLabel = "felhom.undo-copy-of"
// undoCopyMarker is written last into a copy volume, beside the data (never inside it).
const undoCopyMarker = "felhom-undo-complete"
// preUpdateMetaDir holds the previous version's .felhom.yml for the undo. A DIRECTORY, so
// LoadMetadata can read it; the syncer copies only two files into the stack dir's root and never
// touches a subdirectory.
const preUpdateMetaDir = "pre-update-meta"
// appliedMetaDir (v0.263.2) holds the .felhom.yml that belongs to the PINNED version — the probe the
// running version answers. Written when a version is pinned (deploy, adoption, a guarded update) and
// put back by an undo. WHY IT EXISTS: `.felhom.yml` flows into the stack dir on every catalog sync
// (§5.4 — it is never frozen), so by the time anyone presses Update the stack dir already holds the
// NEW version's file. MEASURED LIVE on 9202 2026-09-23: v0.263.1 saved "the old .felhom.yml" at update
// time, got the new probe (romm :8999), and judged the serving old version "did not start".
const appliedMetaDir = "applied-meta"
// AppliedMetaFile is the pinned version's .felhom.yml record (R-669, v0.270.0: the recovery unit captures
// it, so a restore brings back the PINNED version's health check, not the sync's newer one).
func AppliedMetaFile(stackDir string) string {
return filepath.Join(stackDir, appliedMetaDir, ".felhom.yml")
}
// RecordRestoredAppliedMeta makes the .felhom.yml a restore just wrote the pinned version's record
// (R-669, v0.270.0). MEASURED 2026-09-24 (A2): after a failed step ended in a restore, applied-meta still
// held the FAILED step's probe, and the next failed update's undo judged the correct old version with it
// and held the app. Pinned by TestR669_RestoreResetsTheAppliedRecord.
func (m *Manager) RecordRestoredAppliedMeta(name, stackDir string) {
m.storeAppliedMetaFrom(name, stackDir, filepath.Join(stackDir, ".felhom.yml"))
}
// storeAppliedMeta records the .felhom.yml of the version just pinned. A failure is logged by the
// caller and never fails the act — without it the undo falls back to the current file, loudly.
func storeAppliedMeta(stackDir string, src []byte) error {
if len(src) == 0 {
return fmt.Errorf("refusing to store an empty applied .felhom.yml")
}
d := filepath.Join(stackDir, appliedMetaDir)
if err := os.MkdirAll(d, 0o755); err != nil {
return err
}
tmp := filepath.Join(d, ".felhom.yml.tmp")
if err := os.WriteFile(tmp, src, 0o644); err != nil {
return err
}
return os.Rename(tmp, filepath.Join(d, ".felhom.yml"))
}
// storeAppliedMetaFrom copies a .felhom.yml file into the applied record, logging (never failing).
func (m *Manager) storeAppliedMetaFrom(name, stackDir, src string) {
b, err := os.ReadFile(src)
if err == nil {
err = storeAppliedMeta(stackDir, b)
}
if err != nil {
m.logger.Printf("[WARN] [stacks] pin %s: could not record the pinned version's .felhom.yml (%v) — an undo would check health with whatever file is current then", name, err)
}
}
// undoHelperImage is the helper the backup legs already use for volume tars (backup.go, restore.go),
// so no new image is introduced onto a box.
const undoHelperImage = "alpine"
// undoCopy pairs an app volume with its last-second copy.
type undoCopy struct {
Volume string `json:"volume"`
Copy string `json:"copy"`
}
// UpdateUndone is app.yaml's record of the last update the box undid (decision 15: no automatic
// retry until the catalog moves; the button stays usable for a person). Written only by a
// successful undo; cleared by the next successful update.
type UpdateUndone struct {
To map[string]string `yaml:"to" json:"to"`
At string `yaml:"at" json:"at"`
Why string `yaml:"why,omitempty" json:"why,omitempty"`
}
// volumeCopier is the process boundary of the undo's copy. Production is dockerVolumeCopier; tests
// inject a fake and never touch docker.
type volumeCopier interface {
// ProjectVolumes lists the volumes carrying the compose project label. Since v0.268.0 (R-658) it
// is a CROSS-CHECK only, logged — never the selector: a restore recreated volumes without it.
ProjectVolumes(project string) ([]string, error)
// VolumeExists reports whether Docker holds a volume by exactly this name.
VolumeExists(vol string) (bool, error)
VolumeBytes(vol string) (int64, error)
Copy(src, dst, app string) error
Complete(copyVol string) bool
Restore(copyVol, vol string) error
Remove(vol string) error
CopiesOf(app string) []string
}
func (m *Manager) copier() volumeCopier {
if m.undoCopier != nil {
return m.undoCopier
}
return dockerVolumeCopier{m: m}
}
type dockerVolumeCopier struct{ m *Manager }
func (d dockerVolumeCopier) ProjectVolumes(project string) ([]string, error) {
out, err := d.m.execCommand("docker", "volume", "ls", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Name}}")
if err != nil {
return nil, err
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols, nil
}
func (d dockerVolumeCopier) VolumeExists(vol string) (bool, error) {
// `--filter name=` matches a SUBSTRING, so the listing is only narrowed by it; the exact compare
// below is the test.
out, err := d.m.execCommand("docker", "volume", "ls", "-q", "--filter", "name="+vol)
if err != nil {
return false, err
}
for _, l := range strings.Split(out, "\n") {
if strings.TrimSpace(l) == vol {
return true, nil
}
}
return false, nil
}
func (d dockerVolumeCopier) VolumeBytes(vol string) (int64, error) {
out, err := d.m.execCommand("docker", "run", "--rm", "-v", vol+":/v:ro", undoHelperImage, "du", "-sb", "/v")
if err != nil {
return 0, err
}
f := strings.Fields(out)
if len(f) == 0 {
return 0, fmt.Errorf("du printed nothing for %s", vol)
}
return strconv.ParseInt(f[0], 10, 64)
}
// Copy runs in the FOREGROUND so the helper container's own exit status is the answer; the marker is
// written only after cp and sync both succeed.
func (d dockerVolumeCopier) Copy(src, dst, app string) error {
if _, err := d.m.execCommand("docker", "volume", "create", "--label", undoCopyLabel+"="+app, dst); err != nil {
return err
}
_, err := d.m.execCommand("docker", "run", "--rm", "-v", src+":/from:ro", "-v", dst+":/to", undoHelperImage,
"sh", "-c", "mkdir -p /to/data && cp -a /from/. /to/data/ && sync && touch /to/"+undoCopyMarker)
return err
}
func (d dockerVolumeCopier) Complete(copyVol string) bool {
_, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/c:ro", undoHelperImage, "test", "-f", "/c/"+undoCopyMarker)
return err == nil
}
// Restore empties the app's volume and copies the data back. The marker is checked again INSIDE the
// same helper, so a copy that lost it between the check and the restore is never poured in.
func (d dockerVolumeCopier) Restore(copyVol, vol string) error {
_, err := d.m.execCommand("docker", "run", "--rm", "-v", copyVol+":/from:ro", "-v", vol+":/to", undoHelperImage,
"sh", "-c", "test -f /from/"+undoCopyMarker+" && find /to -mindepth 1 -delete && cp -a /from/data/. /to/ && sync")
return err
}
func (d dockerVolumeCopier) Remove(vol string) error {
_, err := d.m.execCommand("docker", "volume", "rm", "-f", vol)
return err
}
func (d dockerVolumeCopier) CopiesOf(app string) []string {
out, err := d.m.execCommand("docker", "volume", "ls", "-q", "--filter", "label="+undoCopyLabel+"="+app)
if err != nil {
return nil
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols
}
// undoCopyName is `<volume>.pre-update-<stamp>` — a legal Docker volume name, unique per update.
func undoCopyName(vol string, at time.Time) string {
return vol + ".pre-update-" + at.UTC().Format("20060102T150405Z")
}
// undoMsg renders a born-as-key sentence of the update job. Hungarian, like every other sentence the
// job writes today (R-606 localises them together).
func undoMsg(key string, args ...interface{}) string { return util.Text("hu", key, args...) }
// planUndoCopies lists the app's named volumes and refuses — before anything moves — when their copy
// would breach the disk floor the update already keeps (decision 19's disk limit).
func (m *Manager) planUndoCopies(name, dir string) ([]string, error) {
vols, err := m.appVolumes(name, dir)
if err != nil {
return nil, fmt.Errorf("listing the app's volumes: %w", err)
}
var total int64
for _, v := range vols {
b, err := m.copier().VolumeBytes(v)
if err != nil {
return nil, fmt.Errorf("sizing volume %s: %w", v, err)
}
total += b
}
needGiB := float64(total) / (1 << 30)
if free, known := m.updateDiskFree(); known && free-needGiB < updateDiskFloorGiB {
return nil, &undoSpaceError{need: needGiB, free: free}
}
m.logger.Printf("[INFO] [stacks] update %s: the undo copy will hold %d named volume(s), %.1f MiB", name, len(vols), float64(total)/(1<<20))
return vols, nil
}
// ── Which volumes are the app's (R-658, v0.268.0) ─────────────────────────────────────────────────
//
// FOUND 2026-09-23 night by the chaos hour on 9202: the unit restore recreates each named volume with
// a bare `docker volume create <name>`, which carries no compose label, and until v0.267.0 the undo
// selected the volumes it copies BY THAT LABEL. So after any restore the undo copied NOTHING and
// reported the failed update "undone" — the old binary on the new version's migrated data (round 9,
// vikunja: `the undo copy will hold 0 named volume(s)`). Measured with a control: the three restored
// apps had 0 of 3 / 2 / 2 volumes labelled, the three never restored had all of theirs.
//
// So the selector is now the app's OWN DEFINITION: the named volumes its compose file declares,
// resolved to the names compose gives them, each checked to exist. The label is a cross-check that
// is logged and decides nothing. Pinned by TestR658_UndoCopiesUnlabelledVolumes (red-proof: the old
// label selector copies 0 of 2).
// composeVolumesDoc is the part of a compose file that names the app's volumes.
type composeVolumesDoc struct {
Name string `yaml:"name"`
Volumes map[string]*composeVolDef `yaml:"volumes"`
}
type composeVolDef struct {
Name string `yaml:"name"`
External interface{} `yaml:"external"`
}
func (d *composeVolDef) external() bool {
if d == nil {
return false
}
switch v := d.External.(type) {
case bool:
return v
case map[string]interface{}:
return true // the legacy `external: {name: …}` form
}
return false
}
// DeclaredVolumeNames returns the Docker names of the named volumes a compose file declares, the way
// compose names them: the volume's own `name:` when set, else `<project>_<key>`, where the project is
// the file's top-level `name:` or, as the manager runs compose, the stack directory's name. External
// volumes are not the app's and are left out (returned second, for the log). Sorted.
func DeclaredVolumeNames(composePath string) (own, external []string, err error) {
data, err := os.ReadFile(composePath)
if err != nil {
return nil, nil, fmt.Errorf("reading compose file: %w", err)
}
var doc composeVolumesDoc
if err := yaml.Unmarshal(data, &doc); err != nil {
return nil, nil, fmt.Errorf("parsing compose file %s: %w", composePath, err)
}
project := strings.TrimSpace(doc.Name)
if project == "" {
project = filepath.Base(filepath.Dir(composePath))
}
for key, def := range doc.Volumes {
full := project + "_" + key
if def != nil && strings.TrimSpace(def.Name) != "" {
full = strings.TrimSpace(def.Name)
}
if def.external() {
external = append(external, full)
continue
}
own = append(own, full)
}
sort.Strings(own)
sort.Strings(external)
return own, external, nil
}
// appVolumes is the undo's selector: the declared volumes that exist, with the label cross-checked
// and every disagreement logged by name.
func (m *Manager) appVolumes(name, dir string) ([]string, error) {
declared, external, err := DeclaredVolumeNames(ComposePathIn(dir))
if err != nil {
return nil, err
}
if len(external) > 0 {
m.logger.Printf("[INFO] [stacks] update %s: external volume(s) %v are not the app's — never copied", name, external)
}
var vols []string
for _, v := range declared {
ok, err := m.copier().VolumeExists(v)
if err != nil {
return nil, fmt.Errorf("checking volume %s: %w", v, err)
}
if !ok {
m.logger.Printf("[WARN] [stacks] update %s: the definition declares volume %s but Docker holds none by that name — nothing to copy for it", name, v)
continue
}
vols = append(vols, v)
}
labeled, lerr := m.copier().ProjectVolumes(filepath.Base(dir))
if lerr != nil {
m.logger.Printf("[WARN] [stacks] update %s: the label cross-check could not list volumes (%v) — the definition decides anyway", name, lerr)
return vols, nil
}
has := map[string]bool{}
for _, v := range labeled {
has[v] = true
}
var unlabeled []string
for _, v := range vols {
if !has[v] {
unlabeled = append(unlabeled, v)
}
delete(has, v)
}
if len(unlabeled) > 0 {
m.logger.Printf("[WARN] [stacks] update %s: volume(s) %v carry no compose label (recreated by a restore before v0.268.0 — R-658) — copied by name", name, unlabeled)
}
if len(has) > 0 {
var extra []string
for v := range has {
extra = append(extra, v)
}
sort.Strings(extra)
m.logger.Printf("[INFO] [stacks] update %s: volume(s) %v carry the app's label but the definition does not declare them — not copied", name, extra)
}
return vols, nil
}
type undoSpaceError struct{ need, free float64 }
func (e *undoSpaceError) Error() string {
return fmt.Sprintf("the undo copy needs %.2f GiB and %.2f GiB is free (floor %.0f GiB)", e.need, e.free, updateDiskFloorGiB)
}
// makeUndoCopies stops the app and copies every volume. The copies are journaled BEFORE each copy
// starts, so a power cut mid-copy leaves a journal that names every volume to clean up.
func (m *Manager) makeUndoCopies(name, dir string, env []string, vols []string, entry *updateJournalEntry) error {
if _, err := m.updateCompose(dir, env, "stop"); err != nil {
return fmt.Errorf("stopping the app for the copy: %w", err)
}
stamp := m.now()
for _, v := range vols {
c := undoCopy{Volume: v, Copy: undoCopyName(v, stamp)}
entry.UndoCopies = append(entry.UndoCopies, c)
if !m.enterUpdatePhase(name, entry, UpdatePhaseCopying) {
return fmt.Errorf("journal write failed before copying %s", v)
}
t0 := time.Now()
if err := m.copier().Copy(c.Volume, c.Copy, name); err != nil {
return fmt.Errorf("copying %s: %w", v, err)
}
m.logger.Printf("[INFO] [stacks] update %s: copied %s → %s in %s", name, c.Volume, c.Copy, time.Since(t0).Round(time.Millisecond))
}
entry.Copied = true
return nil
}
func (m *Manager) removeUndoCopies(name string, copies []undoCopy) {
for _, c := range copies {
if err := m.copier().Remove(c.Copy); err != nil {
m.logger.Printf("[WARN] [stacks] update %s: could not remove the undo copy %s: %v", name, c.Copy, err)
}
}
}
// RemoveUndoCopies deletes every undo copy an app still has — the removal path's hook, so a copy kept
// by a failed undo does not outlive the app.
func (m *Manager) RemoveUndoCopies(name string) int {
n := 0
for _, v := range m.copier().CopiesOf(name) {
if err := m.copier().Remove(v); err != nil {
m.logger.Printf("[WARN] [stacks] remove %s: could not delete the undo copy %s: %v", name, v, err)
continue
}
n++
}
return n
}
// savePreUpdateMeta keeps the PINNED version's .felhom.yml for the undo's health check: the applied
// record when there is one; otherwise — an app pinned before v0.263.2 — the current file, which the
// catalog sync may already have replaced with the new version's (said in the returned note).
func savePreUpdateMeta(dir string) (string, error) {
src, err := os.ReadFile(filepath.Join(dir, appliedMetaDir, ".felhom.yml"))
if err != nil {
if src, err = os.ReadFile(filepath.Join(dir, ".felhom.yml")); err != nil {
return "", err
}
err = errNoAppliedMeta
}
note := err
md := filepath.Join(dir, preUpdateMetaDir)
if err := os.MkdirAll(md, 0o755); err != nil {
return "", err
}
if err := os.WriteFile(filepath.Join(md, ".felhom.yml"), src, 0o644); err != nil {
return "", err
}
return md, note
}
// errNoAppliedMeta: the app has no applied .felhom.yml record (pinned before v0.263.2), so the undo's
// probe is taken from the current file. The copy WAS made — callers log this and carry on.
var errNoAppliedMeta = fmt.Errorf("no applied .felhom.yml for the pinned version (pinned before v0.263.2) — the undo will probe with the current file")
func (m *Manager) undoHealth(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string) {
if m.updateUndoHealthFn != nil {
return m.updateUndoHealthFn(ctx, name, timeout, meta)
}
// A test that injects only the update's health seam gets the same answer for the undo (production
// injects neither and always takes waitUpdateHealthyMeta).
if m.updateHealthFn != nil {
return m.updateHealthFn(ctx, name, timeout)
}
return m.waitUpdateHealthyMeta(ctx, name, timeout, meta)
}
// tryUndo puts the pre-update data and definition back and checks the old version with its own probe.
// It returns "" when the app is healthy on its old version again, else the state the data is in.
//
// ORDER IS THE DESIGN: every copy is validated BEFORE any is poured back (a cut-off copy leaves the
// data exactly as the new version left it — UndoStateUntouched); the definition is put back only after
// the data (so a failure in between reads "half" with the pin still naming the new version, which is
// what ran on that data).
func (m *Manager) tryUndo(ctx context.Context, name, dir, why string, entry *updateJournalEntry) string {
start := m.now()
if !m.enterUpdatePhase(name, entry, UpdatePhaseUndoing) {
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the undo — undoing anyway", name)
}
m.logger.Printf("[WARN] [stacks] update %s: UNDO — putting back the previous version and its %d volume copy(ies) (reason: %s)", name, len(entry.UndoCopies), why)
if _, err := m.updateCompose(dir, m.stackEnv(dir), "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: stopping the new version before the undo failed: %v", name, err)
}
for _, c := range entry.UndoCopies {
if !m.copier().Complete(c.Copy) {
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy %s has no finished-marker — it is cut off or missing; NOTHING is put back", name, c.Copy)
return UndoStateUntouched
}
}
for _, c := range entry.UndoCopies {
if err := m.copier().Restore(c.Copy, c.Volume); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: putting %s back from %s FAILED: %v — the data is mixed; the copies are kept", name, c.Volume, c.Copy, err)
return UndoStateHalf
}
}
m.restoreDefinition(name, dir, *entry)
env := m.stackEnv(dir)
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: starting the previous version failed: %v", name, err)
m.captureHoldLogs(name, dir, env)
_, _ = m.updateCompose(dir, env, "down")
return UndoStateNotStarted
}
meta := LoadMetadata(dir)
if entry.PrevMeta != "" {
if _, err := os.Stat(filepath.Join(entry.PrevMeta, ".felhom.yml")); err == nil {
meta = LoadProbeMetadata(entry.PrevMeta) // R-670: a probe copy, no compose beside it
} else {
m.logger.Printf("[WARN] [stacks] update %s: the previous .felhom.yml is missing (%v) — checking with the current one", name, err)
}
}
healthy, detail := m.undoHealth(ctx, name, m.healthTimeout(), &meta)
if !healthy {
m.logger.Printf("[ERROR] [stacks] update %s: the previous version did not come up after the undo: %s", name, detail)
m.captureHoldLogs(name, dir, env)
_, _ = m.updateCompose(dir, env, "down")
return UndoStateNotStarted
}
m.recordInstalledImages(name, dir, env)
m.removeUndoCopies(name, entry.UndoCopies)
m.recordUpdateUndone(name, dir, &UpdateUndone{To: entry.NewPin, At: start.UTC().Format(time.RFC3339), Why: why})
m.recordFailedStep(name, dir, entry.NewPin, "undone") // R-680
m.removePreUpdateCopies(dir)
_ = os.RemoveAll(filepath.Join(dir, preUpdateConvertDir)) // v0.273.0: the conversion's dump, if any
_ = m.RefreshStatus()
m.clearJournal(name)
if err := m.ScanStacks(); err != nil { // R-678: the page reads the pin the undo put back
m.logger.Printf("[WARN] [stacks] update %s: rescan after the undo failed: %v", name, err)
}
m.finishUpdate(name, UpdatePhaseUndone, "")
m.emitUpdateEvent(UpdateEventUndone, name, entry, UpdateRestorePoint{}, true)
m.logger.Printf("[INFO] [stacks] update %s: UNDONE in %s — the previous version is running on the data from before the update (%s)", name, m.now().Sub(start).Round(time.Second), detail)
return ""
}
// recordUpdateUndone writes (or, with nil, clears) app.yaml's last_update_undone. A failed write is
// logged and never fails the undo — it is a record, and the app is already back.
func (m *Manager) recordUpdateUndone(name, dir string, u *UpdateUndone) {
cfg := LoadAppConfig(dir)
if cfg == nil || (u == nil && cfg.LastUpdateUndone == nil) {
return
}
cfg.LastUpdateUndone = u
meta := LoadMetadata(dir)
if err := SaveAppConfig(dir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: recording last_update_undone failed: %v", name, err)
return
}
m.mu.Lock()
if st, ok := m.stacks[name]; ok && st.AppConfig != nil {
st.AppConfig.LastUpdateUndone = u
}
m.mu.Unlock()
}
// ── The household and the operator are TOLD (v0.264.0, `09` §3 decision 15, §6.4 part 2) ─────────
// Update event kinds — the hub event types, one register for both repos (hub allowedEventTypes).
const (
UpdateEventUndone = "app_update_undone" // the box put the previous version back — warning
UpdateEventHeld = "app_update_held" // the update (or its undo) failed and the app is HELD — error
)
// UpdateEvent is what the update job hands to the notifier. The stacks package composes no sentence
// for it: the notifier renders the undone line from a bundle key, and the held line is the hold's own
// sentence, rendered by the backup side in each language (main.go wires both).
type UpdateEvent struct {
Kind string
App string
From, To map[string]string
At time.Time
CopyTier int
CopyDate time.Time
// HoldRecorded is false when the hold could not be persisted — the event still goes (the operator
// must hear about exactly that), and the household sentence is then update.error.hold_unsaved.
HoldRecorded bool
}
// SetUpdateEventSink wires the notifier (main.go). INIT-ONLY. Nil = no events (tests, setup mode).
// Pinned by TestUpdateEventSinkIsWiredAtStartup (an AST walk of main.go).
func (m *Manager) SetUpdateEventSink(fn func(UpdateEvent)) {
m.mu.Lock()
m.updateEventSink = fn
m.mu.Unlock()
}
// emitUpdateEvent sends ONE event for an undone or held update. Called exactly once per outcome by
// tryUndo (undone) and failAndHold (held) — the job never retries (decision 15), so one outcome is one
// event; the hub's per-app cooldown is the second belt against a storm (R-629).
func (m *Manager) emitUpdateEvent(kind, name string, entry *updateJournalEntry, rp UpdateRestorePoint, holdRecorded bool) {
m.mu.RLock()
sink := m.updateEventSink
m.mu.RUnlock()
if sink == nil {
return
}
ev := UpdateEvent{Kind: kind, App: name, At: m.now(), CopyTier: rp.Tier, CopyDate: rp.ProvenAt, HoldRecorded: holdRecorded}
if entry != nil {
ev.From, ev.To = entry.PrevPin, entry.NewPin
}
if kind == UpdateEventHeld {
m.logger.Printf("[INFO] [stacks] update %s: event %s (hold recorded: %v)", name, kind, holdRecorded)
} else { // R-647: "hold recorded" means nothing for an undone update
m.logger.Printf("[INFO] [stacks] update %s: event %s", name, kind)
}
sink(ev)
}
// BackfillAppliedMeta is R-646's startup pass (v0.264.0), beside AdoptPins. An app pinned before
// v0.263.2 has no applied-meta record, so its FIRST undo would judge the old version with whatever
// .felhom.yml the catalog sync last put in place. For an app that is CURRENT with the catalog, that file
// IS the pinned version's own — so it is recorded now. An app that is Behind or Unknown is skipped and
// NAMED: its pinned version's file is already gone, and guessing it would be worse than saying so.
// Idempotent; never overwrites an existing record. Returns (recorded, skipped) app names.
func (m *Manager) BackfillAppliedMeta() (recorded, skipped []string) {
for _, st := range m.GetStacks() {
if !st.Deployed || st.AppConfig == nil || len(st.AppConfig.PinnedImages) == 0 {
continue
}
dir := filepath.Dir(st.ComposePath)
if _, err := os.Stat(filepath.Join(dir, appliedMetaDir, ".felhom.yml")); err == nil {
continue // already recorded — never overwritten
}
if CatalogOrder(st) != UpdateOrderCurrent {
skipped = append(skipped, st.Name)
continue
}
b, err := os.ReadFile(filepath.Join(dir, ".felhom.yml"))
if err == nil {
err = storeAppliedMeta(dir, b)
}
if err != nil {
m.logger.Printf("[WARN] [stacks] applied-meta backfill: %s: %v", st.Name, err)
skipped = append(skipped, st.Name)
continue
}
recorded = append(recorded, st.Name)
}
m.logger.Printf("[INFO] [stacks] applied-meta backfill (R-646): recorded %d %v; skipped %d %v — not current with the catalog, their pinned version's .felhom.yml is no longer on the box", len(recorded), recorded, len(skipped), skipped)
return recorded, skipped
}