7861bf9dde
gates / gates (push) Successful in 28s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
1525 lines
71 KiB
Go
1525 lines
71 KiB
Go
package stacks
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
|
)
|
|
|
|
// ── The guarded update (update arc slice 4, controller v0.237.0) ─────────────────────────────────
|
|
//
|
|
// WHAT IT REPLACED. `UpdateStack` advanced the pin, pulled, ran `up -d` and returned — no copy first,
|
|
// no check of memory, disk, a running backup or a held app, and a success the moment `up` returned.
|
|
// SPIKE-app-update-2026-09-01 §4 measured that as HTTP 200 over an app that was already crash-looping
|
|
// (R-443), and R-439 is that a held app could be updated at all.
|
|
//
|
|
// THE SEQUENCE, and the order is the point:
|
|
//
|
|
// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling
|
|
// → copying → starting → verifying → done | (undoing → undone | failed)
|
|
//
|
|
// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02,
|
|
// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The
|
|
// same predicate that permits the destructive „Teljes visszaállítás" permits the update — the
|
|
// route back IS that restore, so an update without it has no route back.
|
|
// 2. The safety dump is taken BEFORE the pin moves: it is "the state the customer was in a minute ago",
|
|
// and a minute later the migration may have run.
|
|
// 3. The pin moves BEFORE the pull (v0.235.0's reason: pull and up act on the file on disk).
|
|
// 4. A PULL failure puts the pin BACK — nothing ran, so reverting is safe and honest (Scenario E).
|
|
// 5. A HEALTH failure leaves the pin where it is — the new version's migration may have run, and a
|
|
// pin claiming the old version would be a record of something untrue (Scenario F). The app is
|
|
// HELD STOPPED and the customer is told which backup it can be restored from.
|
|
//
|
|
// SINCE v0.263.0 IT PUTS THE OLD VERSION BACK ITSELF — with the data from before the update (09 §3
|
|
// decision 15, undo.go). Until then it deliberately did not: SPIKE-upgrade-test-2026-09-06 measured
|
|
// that the old image alone refuses data a new one migrated (Docmost, Nextcloud). The undo does not put
|
|
// the old image on migrated data; it puts the pre-migration data back as well. A HOLD now happens only
|
|
// when that undo fails too.
|
|
//
|
|
// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10),
|
|
// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at
|
|
// the next startup (Scenario G) — the AppStopGuard pattern.
|
|
|
|
// Update phases, as recorded in the journal and served as Stack.UpdatePhase.
|
|
const (
|
|
UpdatePhaseChecking = "checking"
|
|
UpdatePhaseBackingUp = "backing-up"
|
|
UpdatePhaseSafetyDump = "safety-dump"
|
|
UpdatePhasePinning = "pinning"
|
|
UpdatePhasePulling = "pulling"
|
|
UpdatePhaseStarting = "starting"
|
|
UpdatePhaseVerifying = "verifying"
|
|
UpdatePhaseDone = "done"
|
|
UpdatePhaseFailed = "failed"
|
|
// UpdatePhaseCopying, UpdatePhaseUndoing and UpdatePhaseUndone live in undo.go.
|
|
)
|
|
|
|
// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the
|
|
// specification — it is instantaneous and is the first step of fetching the new version, so it shares
|
|
// the pull's label rather than inventing a sentence nobody would read.
|
|
var updatePhaseLabels = map[string]string{
|
|
UpdatePhaseChecking: "Ellenőrzés…",
|
|
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
|
|
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
|
|
UpdatePhasePinning: "Új verzió letöltése…",
|
|
UpdatePhasePulling: "Új verzió letöltése…",
|
|
UpdatePhaseStarting: "Indítás az új verzióval…",
|
|
UpdatePhaseVerifying: "Működés ellenőrzése…",
|
|
UpdatePhaseDone: "Frissítve",
|
|
UpdatePhaseFailed: "A frissítés nem sikerült",
|
|
// v0.263.0 — born as bundle keys (update.phase.*); TestUndo_PhaseLabelsMatchTheBundle pins them equal.
|
|
UpdatePhaseCopying: "Az adatok másolása a frissítés előtt…",
|
|
UpdatePhaseUndoing: "Visszaállítás az előző változatra…",
|
|
UpdatePhaseUndone: "Visszaállítva az előző változatra",
|
|
// v0.273.0 — born as a bundle key (update.phase.converting).
|
|
UpdatePhaseConverting: "Adatbázis átalakítása",
|
|
}
|
|
|
|
// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one.
|
|
func UpdatePhaseLabel(phase string) string { return updatePhaseLabels[phase] }
|
|
|
|
// Customer sentences. Named so tests compare against the constant, never a retyped literal (R-364).
|
|
const (
|
|
MsgUpdateNoGuards = "A frissítés nem indítható: a frissítés előtti biztonsági ellenőrzés nem érhető el ezen a szerveren."
|
|
MsgUpdateNotDeployed = "Az alkalmazás nincs telepítve, ezért nem frissíthető."
|
|
MsgUpdateDeployingFmt = "A(z) %s telepítése még folyamatban van — a frissítés utána indítható."
|
|
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
|
|
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
|
|
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
|
|
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani, és most új mentés sem készíthető róla. Ellenőrizd a Mentések oldalon, hogy az alkalmazás meghajtója elérhető-e — utána a frissítés elindítható."
|
|
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
|
|
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdateDumpFailFmt = "A frissítés nem indult el, mert az adatbázis pillanatkép nem készült el: %v. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdatePinFailed = "A frissítés nem indult el: az új verzió leírása nem olvasható be. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdateJournalFailed = "A frissítés nem indult el: a frissítés naplója nem menthető. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdatePullFailed = "Az új verzió letöltése nem sikerült, ezért a frissítés elmaradt. Az alkalmazás a korábbi verzióval fut tovább."
|
|
MsgUpdateInterrupted = "A frissítés megszakadt, mert a vezérlő újraindult, mielőtt az új verzió elindult volna. Az alkalmazás a korábbi verzióval fut tovább."
|
|
MsgUpdateHoldUnsaved = "A frissítés nem sikerült, az alkalmazás le lett állítva, de a leállítás rögzítése nem sikerült. Ne indítsd újra — vedd fel velünk a kapcsolatot."
|
|
)
|
|
|
|
// updateDiskFloorGiB is the free space the Docker data root must have before a pull. A FIXED FLOOR,
|
|
// stated as such: the new image set's size is not known without a registry query (the catalog
|
|
// records tags, not sizes, and §8.1 of 09 already declines registry calls on the customer box), so
|
|
// the rule is "not less than 2 GB", not "enough for these images".
|
|
const updateDiskFloorGiB = 2.0
|
|
|
|
// updateSettleWindow is the rule for an app with no .felhom.yml health check: every container
|
|
// running, none restarting, for this long.
|
|
const updateSettleWindow = 60 * time.Second
|
|
|
|
// updatePollEvery is how often the health wait re-reads the stack.
|
|
const updatePollEvery = 5 * time.Second
|
|
|
|
// Backup tiers (R-475), mirroring backup.UpdateTier* — stacks cannot import backup, so
|
|
// TestR475_TierConstantsAgree (cmd/controller) pins the two sets equal.
|
|
const (
|
|
UpdateTierLocal = 1 // the app's own recovery unit
|
|
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
|
|
UpdateTierOffsite = 3 // off-site
|
|
)
|
|
|
|
// UpdateRestorePoint is one proven, restorable copy the update may lean on. The backup side returns
|
|
// only proven, restorable copies (never an attempt clock, never an unopenable unit), so there is no
|
|
// "maybe" field here to forget to check.
|
|
type UpdateRestorePoint struct {
|
|
Tier int // UpdateTierSecondDrive / UpdateTierLocal / UpdateTierOffsite
|
|
ProvenAt time.Time // when the data in that copy was last proven written
|
|
}
|
|
|
|
func updateTierName(tier int) string {
|
|
switch tier {
|
|
case UpdateTierSecondDrive:
|
|
return "Tier 2 (second drive)"
|
|
case UpdateTierLocal:
|
|
return "Tier 1 (own recovery unit)"
|
|
case UpdateTierOffsite:
|
|
return "Tier 3 (off-site)"
|
|
}
|
|
return fmt.Sprintf("tier %d", tier)
|
|
}
|
|
|
|
// freshRestorePoint is THE age rule (backup_max_age), applied to whichever tier is being considered
|
|
// — R-475 Scenario M: a stale copy on ANY tier is stale.
|
|
//
|
|
// COMPANION RED-PROOF M (REPORT.md): check the age only for Tier 2 (let any other tier through
|
|
// whatever its age). TestR475_M_TheAgeRuleAppliesToTheChosenTier then fails: a 30-hour-old copy of
|
|
// the app's own unit carries the update with no backup first.
|
|
func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePoint) bool {
|
|
return func(p UpdateRestorePoint) bool {
|
|
return !p.ProvenAt.IsZero() && now.Sub(p.ProvenAt) <= maxAge
|
|
}
|
|
}
|
|
|
|
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
|
|
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
|
|
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
|
|
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
|
|
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
|
|
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
|
|
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
|
|
fresh := freshRestorePoint(now, maxAge)
|
|
return func(p UpdateRestorePoint) bool {
|
|
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
|
|
return false
|
|
}
|
|
return fresh(p)
|
|
}
|
|
}
|
|
|
|
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
|
|
func (m *Manager) currentDeployTime(name string) time.Time {
|
|
st, ok := m.GetStack(name)
|
|
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
|
|
return time.Time{}
|
|
}
|
|
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
|
|
if err != nil {
|
|
return time.Time{}
|
|
}
|
|
return t
|
|
}
|
|
|
|
// ClearUpdateState wipes everything a finished-or-abandoned update left on a stack. R-614: a fresh
|
|
// install of an app that had previously failed an update showed the OLD phase — „Frissítve" on a
|
|
// deploy that had just happened — because a remove cleared the directory and not the in-memory
|
|
// record. The name is the only thing the next install shares with the last one, so the record has to
|
|
// go when the app does.
|
|
func (m *Manager) ClearUpdateState(name string) {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
s, ok := m.stacks[name]
|
|
if !ok {
|
|
return
|
|
}
|
|
s.Updating = false
|
|
s.UpdatePhase = ""
|
|
s.UpdatePhaseLabel = ""
|
|
s.UpdateError = ""
|
|
s.updateHeld = false
|
|
s.HealthProbe = nil
|
|
}
|
|
|
|
func (m *Manager) markUpdateHeld(name string) {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.updateHeld = true
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
func fmtDeployTime(t time.Time) string {
|
|
if t.IsZero() {
|
|
return "unknown"
|
|
}
|
|
return t.UTC().Format(time.RFC3339)
|
|
}
|
|
|
|
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
|
|
if len(pts) == 0 {
|
|
return "none"
|
|
}
|
|
parts := make([]string, 0, len(pts))
|
|
for _, p := range pts {
|
|
parts = append(parts, fmt.Sprintf("%s at %s (%s old)", updateTierName(p.Tier), p.ProvenAt.UTC().Format(time.RFC3339), now.Sub(p.ProvenAt).Round(time.Minute)))
|
|
}
|
|
return strings.Join(parts, "; ")
|
|
}
|
|
|
|
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
|
|
// backup, so cmd/controller/main.go wires an adapter (TestSlice4_UpdateGuardsAreWiredAtStartup).
|
|
type UpdateGuards interface {
|
|
HoldFor(name string) (bool, string)
|
|
Busy(name string) (bool, string)
|
|
// RestorePoints walks the tiers in preference order (2, 1, 3) and returns the first copy accept
|
|
// admits (nil = any), whether one was found, and every copy looked at (R-475).
|
|
RestorePoints(ctx context.Context, name string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint)
|
|
// CanBackUp reports whether "back up first" can run for this app now (R-475 Scenario L).
|
|
CanBackUp(name string) (bool, string)
|
|
BackupNow(ctx context.Context, name string) error
|
|
SafetyDump(ctx context.Context, name string) ([]string, error)
|
|
// HoldAfterFailedUpdate records the hold. undoState (v0.263.0) says what a FAILED undo left the data
|
|
// as (UndoState*), "" when no undo was attempted; the hold sentence says so.
|
|
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint, undoState string) error
|
|
}
|
|
|
|
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
|
|
// an update that cannot see the backup cannot promise a route back.
|
|
func (m *Manager) SetUpdateGuards(g UpdateGuards) {
|
|
m.mu.Lock()
|
|
m.updateGuards = g
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
// SetSelfUpdatingCheck wires the OTHER half of the v0.261.0 lock: is the CONTROLLER swapping itself?
|
|
//
|
|
// A plain callback rather than a member of UpdateGuards, for two reasons. UpdateGuards is the BACKUP
|
|
// side's interface and this has nothing to do with backups; and `stacks` must never import
|
|
// `selfupdate` (selfupdate reaches the agent, and the import would run the wrong way), so the
|
|
// dependency is inverted here and satisfied in main.go with `updater.IsUpdateRunning`.
|
|
//
|
|
// Nil is safe and means the pre-v0.261.0 behaviour: no self-update gate.
|
|
func (m *Manager) SetSelfUpdatingCheck(fn func() bool) {
|
|
m.mu.Lock()
|
|
m.selfUpdating = fn
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
func (m *Manager) selfUpdatingNow() bool {
|
|
m.mu.RLock()
|
|
fn := m.selfUpdating
|
|
m.mu.RUnlock()
|
|
return fn != nil && fn()
|
|
}
|
|
|
|
func (m *Manager) guards() UpdateGuards {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
return m.updateGuards
|
|
}
|
|
|
|
func fillHoldReason(g UpdateGuards, st *Stack) {
|
|
if st == nil {
|
|
return
|
|
}
|
|
held := false
|
|
if g != nil && st.Deployed {
|
|
if h, why := g.HoldFor(st.Name); h {
|
|
st.HoldReason, held = why, true
|
|
if nw, ok := g.(holdWholeCopy); ok {
|
|
st.HoldNoWholeCopy = nw.HoldNoWholeCopy(st.Name)
|
|
}
|
|
if hk, ok := g.(holdKinder); ok {
|
|
st.HoldKind = hk.HoldKind(st.Name)
|
|
}
|
|
}
|
|
}
|
|
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
|
|
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
|
|
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
|
|
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
|
|
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
|
|
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
|
|
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
|
|
}
|
|
}
|
|
|
|
// holdKinder is the OPTIONAL half of UpdateGuards that names the hold's kind (v0.269.0).
|
|
type holdKinder interface {
|
|
HoldKind(name string) string
|
|
}
|
|
|
|
// holdWholeCopy is the OPTIONAL half of UpdateGuards that says a hold names no copy (R-659).
|
|
type holdWholeCopy interface {
|
|
HoldNoWholeCopy(name string) bool
|
|
}
|
|
|
|
// UpdateRefusal is a refusal taken before anything moved. Reason is a stable key for logs and tests;
|
|
// Message is the customer sentence.
|
|
type UpdateRefusal struct {
|
|
Reason string
|
|
Message string
|
|
// Cause carries the refusal as an ERROR when the producer made one (v0.253.0, R-557). A
|
|
// util.MsgError there knows its bundle key, so `api.Router.errText` renders the refusal in the
|
|
// household's language; Message stays the Hungarian fallback for every refusal built from a
|
|
// literal. Unwrap is what lets errors.Is and util.AsMsg see through this wrapper.
|
|
Cause error
|
|
}
|
|
|
|
func (r *UpdateRefusal) Error() string {
|
|
if r.Cause != nil {
|
|
return r.Cause.Error()
|
|
}
|
|
return r.Message
|
|
}
|
|
|
|
func (r *UpdateRefusal) Unwrap() error { return r.Cause }
|
|
|
|
func (m *Manager) now() time.Time {
|
|
if m.updateNowFn != nil {
|
|
return m.updateNowFn()
|
|
}
|
|
return time.Now()
|
|
}
|
|
|
|
func (m *Manager) refuseUpdate(name, reason, msg, detail string) *UpdateRefusal {
|
|
m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail)
|
|
return &UpdateRefusal{Reason: reason, Message: msg}
|
|
}
|
|
|
|
// refuseUpdateErr is refuseUpdate for a refusal the producer already built as an error — it keeps the
|
|
// error whole, so its kind and its message key both survive to the API.
|
|
func (m *Manager) refuseUpdateErr(name, reason string, cause error, detail string) *UpdateRefusal {
|
|
m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail)
|
|
return &UpdateRefusal{Reason: reason, Message: cause.Error(), Cause: cause}
|
|
}
|
|
|
|
// UpdatePreflight runs every CHEAP refusal (slice 4 Part 1 + the precondition's existence), in order,
|
|
// and returns the first. Nothing is moved and nothing is recorded by it. The router calls it before
|
|
// recording the customer's intent, so an update that was never going to happen records nothing.
|
|
func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack")
|
|
}
|
|
if !st.Deployed {
|
|
return m.refuseUpdateErr(name, "not_deployed", util.MsgError("update.refusal.not_deployed"), "not deployed")
|
|
}
|
|
g := m.guards()
|
|
if g == nil {
|
|
return m.refuseUpdateErr(name, "guards_unwired", util.MsgError("update.error.no_guards"), "no UpdateGuards wired — fail closed")
|
|
}
|
|
if st.Deploying {
|
|
return m.refuseUpdateErr(name, "deploying", util.MsgError("update.refusal.deploying", name), "a deploy is in progress")
|
|
}
|
|
if st.Updating {
|
|
return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "an update is already in progress")
|
|
}
|
|
if held, why := g.HoldFor(name); held {
|
|
return m.refuseUpdate(name, "held", why, "the app is held")
|
|
}
|
|
if busy, why := g.Busy(name); busy {
|
|
return m.refuseUpdateErr(name, "busy", util.MsgError("update.refusal.busy"), why)
|
|
}
|
|
if m.IsMigrating() {
|
|
return m.refuseUpdateErr(name, "migrating", util.MsgError("update.refusal.migrating"), "a data migration is running")
|
|
}
|
|
// v0.261.0 — the other half of the self-update lock. The controller's swap restarts this process;
|
|
// starting an app update into that is how an update loses its own supervisor mid-flight. TRANSIENT:
|
|
// the household is told to try again in a few minutes, and the caller in `09` §6.2 reads the
|
|
// machine-readable `downgrade`-style reason and retries rather than giving up.
|
|
if m.selfUpdatingNow() {
|
|
return m.refuseUpdateErr(name, "self_updating", util.MsgError("err.stacks.update_self_updating"),
|
|
"the controller is swapping itself — refusing to start an app update into a restart")
|
|
}
|
|
// R-524 — THE PIN NEVER MOVES BACKWARDS WITHOUT THE OPERATOR. MEASURED 2026-09-15 (BIGNIGHT
|
|
// Phase 6): privatebin was updated 2.0.5 → 2.0.6, the catalog was reverted to 2.0.5, and the
|
|
// „Frissítés" button behind the badge would have advanced the pin to the OLDER image — on a
|
|
// datadir the newer version may already have migrated, with §4's ruling saying that cannot be
|
|
// undone. A catalog revert is an operator act on our side; it must never become a data event on
|
|
// the customer's side by itself.
|
|
//
|
|
// It refuses ONLY the provable case (stacks.CatalogOrder's Ahead arm: every differing service
|
|
// orderable and newer). Anything unorderable, mixed or equal falls through to the behaviour that
|
|
// shipped in v0.237.0 — this gate can block an update, so it errs towards letting one run.
|
|
//
|
|
// COMPANION RED-PROOF (REPORT.md): make CatalogOrder's Ahead arm return Behind.
|
|
// TestR524_PreflightRefusesDowngrade then fails — the update is allowed to move the pin back.
|
|
if CatalogOrder(*st) == UpdateOrderAhead {
|
|
return m.refuseUpdateErr(name, "downgrade", util.MsgError("err.stacks.update_downgrade"),
|
|
fmt.Sprintf("installed is provably NEWER than the catalog on every differing service (installed=%v catalog=%v)",
|
|
st.AppConfig.InstalledImages, st.CatalogImages))
|
|
}
|
|
// R-679 (v0.270.0) — AN APP ALREADY AT THE HEAD IS NOT UPDATED. MEASURED 2026-09-24 (night, Part C):
|
|
// navidrome at the catalog head was pressed four times; each press made a safety dump, pulled and
|
|
// restarted the app (8.5 s of downtime) and changed nothing. "Current" is CatalogOrder's own verdict:
|
|
// the installed images equal the catalog's AND no newer tested digest exists for a floating tag — so a
|
|
// re-tested digest still updates. Unknown / unorderable / behind fall through as before. Before this, a
|
|
// same-version Update was called "the repair path" in a test comment; Restart is the repair path now.
|
|
//
|
|
// COMPANION RED-PROOF (REPORT.md): drop this check → TestR679_CurrentAppIsRefused fails at "an app at
|
|
// the head was allowed to update".
|
|
if CatalogOrder(*st) == UpdateOrderCurrent {
|
|
return m.refuseUpdateErr(name, "already_current", util.MsgError("err.stacks.update_already_current"),
|
|
fmt.Sprintf("installed equals the catalog head on every service and no newer tested digest (installed=%v catalog=%v)",
|
|
st.AppConfig.InstalledImages, st.CatalogImages))
|
|
}
|
|
// R-475: any tier counts, and an app with no copy at all is backed up first by the job. So the only
|
|
// refusal left here is Scenario L — no copy on any tier AND no way to make one now. (With a copy
|
|
// but no way to back up, the job still applies the age rule and refuses then if the copy is stale.)
|
|
// Tier 3 is looked at only on this branch, so an ordinary update never waits on the network here.
|
|
//
|
|
// COMPANION RED-PROOF (REPORT.md): drop the `!found` condition. TestR475_L then fails — an app
|
|
// with a copy but no way to back up is refused too.
|
|
if canBackUp, why := g.CanBackUp(name); !canBackUp {
|
|
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
|
|
return m.refuseUpdateErr(name, "no_backup", util.MsgError("update.refusal.no_backup", name),
|
|
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
|
|
}
|
|
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
|
|
}
|
|
// v0.273.0 (B1) — a PostgreSQL major move without the test's mark is refused before anything moves.
|
|
if st.AppConfig != nil && len(st.AppConfig.PinnedImages) > 0 {
|
|
if step, err := nextLadderStep(filepath.Dir(m.CatalogTemplatePath(name, "docker-compose.yml")), st.AppConfig.PinnedImages); err == nil {
|
|
if _, _, key, cerr := m.conversionFor(st, st.AppConfig.PinnedImages, step); cerr != nil && key == "err.stacks.update_engine_no_test" {
|
|
return m.refuseUpdateErr(name, "engine_no_test", util.MsgError(key, appLabel(st)), cerr.Error())
|
|
}
|
|
}
|
|
}
|
|
if ref := m.updateMemoryRefusal(name, st); ref != nil {
|
|
return ref
|
|
}
|
|
free, known := m.updateDiskFree()
|
|
switch {
|
|
case !known:
|
|
m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB)
|
|
case free < updateDiskFloorGiB:
|
|
return m.refuseUpdateErr(name, "disk", util.MsgError("update.refusal.disk", free, updateDiskFloorGiB),
|
|
fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// updateMemoryRefusal applies the deploy's memory check to the NEW template's request, releasing the
|
|
// app's CURRENT request first (an update replaces it). An unknown new request proceeds with a WARN.
|
|
func (m *Manager) updateMemoryRefusal(name string, st *Stack) *UpdateRefusal {
|
|
catPath := m.CatalogTemplatePath(name, ".felhom.yml")
|
|
if _, err := os.Stat(catPath); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update %s: the new template's memory request is unknown (%v) — proceeding without the memory check", name, err)
|
|
return nil
|
|
}
|
|
newMeta := LoadMetadata(filepath.Dir(catPath))
|
|
// R-664 (v0.269.0): on a ladder the NEXT STEP's own memory request is what this press installs.
|
|
if st.AppConfig != nil && len(st.AppConfig.PinnedImages) > 0 {
|
|
if step, err := nextLadderStep(filepath.Dir(catPath), st.AppConfig.PinnedImages); err == nil && step.Meta != catPath {
|
|
if mm, merr := loadMetadataFile(step.Meta); merr == nil {
|
|
newMeta = mm
|
|
}
|
|
}
|
|
}
|
|
newReq, newLim := ParseMemoryMB(newMeta.Resources.MemRequest), ParseMemoryMB(newMeta.Resources.MemLimit)
|
|
if newReq == 0 {
|
|
m.logger.Printf("[WARN] [stacks] update %s: the new template declares no memory request — proceeding without the memory check", name)
|
|
return nil
|
|
}
|
|
oldReq, oldLim := ParseMemoryMB(st.Meta.Resources.MemRequest), ParseMemoryMB(st.Meta.Resources.MemLimit)
|
|
verdict := m.updateMemoryFn
|
|
if verdict == nil {
|
|
verdict = m.memoryVerdict
|
|
}
|
|
if refusal, _ := verdict(newReq, newLim, oldReq, oldLim); refusal != nil {
|
|
return m.refuseUpdateErr(name, "memory", refusal, fmt.Sprintf("new_req=%dMB replacing %dMB does not fit", newReq, oldReq))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func (m *Manager) updateDiskFree() (float64, bool) {
|
|
if m.updateDiskFreeFn != nil {
|
|
return m.updateDiskFreeFn()
|
|
}
|
|
du := system.GetDiskUsage(system.DockerVolumePath)
|
|
if du == nil {
|
|
return 0, false
|
|
}
|
|
return du.AvailGB, true
|
|
}
|
|
|
|
// StartGuardedUpdate re-checks the cheap refusals, claims the Updating flag atomically and launches the
|
|
// job. It returns as soon as the job has STARTED — the result arrives on GET /api/stacks/{name}.
|
|
func (m *Manager) StartGuardedUpdate(name string) error {
|
|
if ref := m.UpdatePreflight(name); ref != nil {
|
|
return ref
|
|
}
|
|
m.mu.Lock()
|
|
s, ok := m.stacks[name]
|
|
if !ok {
|
|
m.mu.Unlock()
|
|
return &UpdateRefusal{Reason: "not_found", Message: fmt.Sprintf("stack %q not found", name)}
|
|
}
|
|
// A second press between the preflight and here is the race this lock closes.
|
|
if s.Updating || s.Deploying {
|
|
m.mu.Unlock()
|
|
return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "lost the race for the Updating flag")
|
|
}
|
|
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
|
m.mu.Unlock()
|
|
|
|
m.logger.Printf("[INFO] [stacks] update %s: accepted — guarded update started", name)
|
|
go m.runGuardedUpdate(context.Background(), name)
|
|
return nil
|
|
}
|
|
|
|
// IsUpdating reports whether a guarded update is in progress for the app.
|
|
func (m *Manager) IsUpdating(name string) bool {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
s, ok := m.stacks[name]
|
|
return ok && s.Updating
|
|
}
|
|
|
|
// AnyUpdating reports whether a guarded update is in flight for ANY app.
|
|
//
|
|
// ── WHY IT EXISTS (v0.261.0) ────────────────────────────────────────────────────────────────────
|
|
//
|
|
// The CONTROLLER updates itself too — daily at `self_update.auto_update_time` (04:30 by default) and,
|
|
// once the hub serves a floor above this box, after any report, at any hour. That swap restarts the
|
|
// controller container. Until v0.261.0 its ONLY busy gate was `backupRunning`, so a self-update could
|
|
// land in the middle of a guarded app update.
|
|
//
|
|
// MEASURED, and the gap is narrower than it looks but real: the update's `backing-up` phase takes the
|
|
// backup single-flight (`RunAppBackupNow` → `acquireRunning`), so `backupMgr.IsRunning()` ALREADY
|
|
// covered that one phase. It covers none of the others — `checking`, `safety-dump`, `pinning`,
|
|
// `pulling`, `starting`, `verifying` — and `starting`/`verifying` are exactly where the new version
|
|
// may already have touched the app's data.
|
|
//
|
|
// ⚠ IT MUST ANSWER FALSE FOR A HELD APP. `Stack.Updating` is cleared by `finishUpdate` on done,
|
|
// failed AND held, so a held app does not hold this lock — otherwise one app that cannot come up
|
|
// would block the controller's own updates for ever, which is a worse failure than the one this
|
|
// prevents. TestR608_LockReleasesAfterHold pins that.
|
|
func (m *Manager) AnyUpdating() bool {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
for _, s := range m.stacks {
|
|
if s.Updating {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// UpdatingStacks is the set of apps an update is currently moving — for the dead-app alarm, which
|
|
// must not count an app the update itself is recreating (R-330's class, a third mechanism).
|
|
func (m *Manager) UpdatingStacks() map[string]bool {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
var out map[string]bool
|
|
for name, s := range m.stacks {
|
|
if s.Updating {
|
|
if out == nil {
|
|
out = map[string]bool{}
|
|
}
|
|
out[name] = true
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
func (m *Manager) setUpdatePhase(name, phase string) {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure — a
|
|
// finished one (the hold's own sentence, or none). A sentence the job owns goes through finishUpdateKey.
|
|
func (m *Manager) finishUpdate(name, phase, msg string) {
|
|
m.finishUpdateKey(name, phase, "", msg)
|
|
}
|
|
|
|
// finishUpdateKey is finishUpdate with the sentence as a bundle KEY (v0.264.0, R-606): UpdateError
|
|
// keeps the Hungarian (byte-identical to the MsgUpdate* literal it replaced), and the key + args ride
|
|
// beside it for the page. key "" = `plain` is a finished sentence and is stored as it is.
|
|
func (m *Manager) finishUpdateKey(name, phase, key, plain string, args ...interface{}) {
|
|
msg := plain
|
|
if key != "" && key != UpdateErrorKeyHeld { // the held key names no bundle entry: plain IS the sentence
|
|
msg = util.Text(i18n.Default, key, args...)
|
|
}
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Updating = false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
|
s.UpdateError, s.UpdateErrorKey, s.UpdateErrorArgs = msg, key, args
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
// UpdatePhaseLabelIn is a phase's label in lang (v0.264.0, R-606). Hungarian is the updatePhaseLabels
|
|
// map itself; another language reads `update.phase.<phase>` and falls back to the Hungarian.
|
|
func UpdatePhaseLabelIn(lang, phase string) string {
|
|
if lang == i18n.Default || phase == "" {
|
|
return UpdatePhaseLabel(phase)
|
|
}
|
|
if b, err := i18n.Shared(); err == nil && b.Has(lang, "update.phase."+phase) {
|
|
return b.Msg(lang, "update.phase."+phase)
|
|
}
|
|
return UpdatePhaseLabel(phase)
|
|
}
|
|
|
|
// UpdateErrorIn is the stack's update sentence in lang (v0.264.0, R-606).
|
|
func (s Stack) UpdateErrorIn(lang string) string {
|
|
if s.UpdateErrorKey == "" || s.UpdateErrorKey == UpdateErrorKeyHeld || lang == i18n.Default {
|
|
return s.UpdateError
|
|
}
|
|
return util.Text(lang, s.UpdateErrorKey, s.UpdateErrorArgs...)
|
|
}
|
|
|
|
// UpdateErrorKeyHeld marks an UpdateError that IS the hold sentence (R-647, v0.265.0). It names no
|
|
// bundle entry: the sentence is composed by the backup side, so UpdateErrorFor asks it for the
|
|
// reader's language, and UpdateErrorIn (no manager to ask) returns the stored Hungarian.
|
|
const UpdateErrorKeyHeld = "update.error.held"
|
|
|
|
// holdLanguage is the OPTIONAL half of UpdateGuards that renders the hold sentence in a given
|
|
// language (the production adapter implements it; the test fakes need not).
|
|
type holdLanguage interface {
|
|
HoldForLang(name, lang string) (bool, string)
|
|
}
|
|
|
|
func holdForLang(g UpdateGuards, name, lang string) (bool, string) {
|
|
if g == nil {
|
|
return false, ""
|
|
}
|
|
if hl, ok := g.(holdLanguage); ok {
|
|
return hl.HoldForLang(name, lang)
|
|
}
|
|
return g.HoldFor(name)
|
|
}
|
|
|
|
// HoldReasonFor is a stack's hold sentence in the READER's language (R-647): "" when nothing holds it.
|
|
// A Hungarian reader on a Hungarian box gets exactly Stack.HoldReason.
|
|
func (m *Manager) HoldReasonFor(st Stack, lang string) string {
|
|
if st.HoldReason == "" {
|
|
return ""
|
|
}
|
|
if held, why := holdForLang(m.guards(), st.Name, lang); held && why != "" {
|
|
return why
|
|
}
|
|
return st.HoldReason
|
|
}
|
|
|
|
// UpdateErrorFor is a stack's update error in the READER's language (R-606 + R-647). A held update's
|
|
// error is the hold sentence, rendered by the backup side for this reader.
|
|
func (m *Manager) UpdateErrorFor(st Stack, lang string) string {
|
|
if st.UpdateErrorKey == UpdateErrorKeyHeld {
|
|
if held, why := holdForLang(m.guards(), st.Name, lang); held && why != "" {
|
|
return why
|
|
}
|
|
return st.UpdateError
|
|
}
|
|
return st.UpdateErrorIn(lang)
|
|
}
|
|
|
|
func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) {
|
|
defer dockerexec.BeginImageWork(args)() // R-863
|
|
if m.updateComposeFn != nil {
|
|
return m.updateComposeFn(dir, env, args...)
|
|
}
|
|
return m.composeExecCustomEnv(dir, env, args...)
|
|
}
|
|
|
|
func (m *Manager) updateHealth(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
|
return m.updateHealthFor(ctx, name, timeout, "")
|
|
}
|
|
|
|
// updateHealthFor is the verify with the NEW version's own .felhom.yml (R-665/R-664): metaFile is the
|
|
// catalog's (or the step's) file journaled at the start of the job; "" = the stack dir's (an update
|
|
// journaled by an older controller).
|
|
func (m *Manager) updateHealthFor(ctx context.Context, name string, timeout time.Duration, metaFile string) (bool, string) {
|
|
if m.updateHealthFn != nil && m.updateHealthMetaFn == nil {
|
|
return m.updateHealthFn(ctx, name, timeout)
|
|
}
|
|
if metaFile != "" {
|
|
if _, err := os.Stat(metaFile); err == nil {
|
|
meta := LoadMetadata(filepath.Dir(metaFile))
|
|
if filepath.Base(metaFile) != ".felhom.yml" {
|
|
if mm, err := loadMetadataFile(metaFile); err == nil {
|
|
meta = mm
|
|
}
|
|
}
|
|
m.lastVerifyMeta = metaFile
|
|
if m.updateHealthMetaFn != nil {
|
|
return m.updateHealthMetaFn(ctx, name, timeout, &meta)
|
|
}
|
|
return m.waitUpdateHealthyMeta(ctx, name, timeout, &meta)
|
|
}
|
|
m.logger.Printf("[WARN] [stacks] update %s: the new version's .felhom.yml %s is gone — judging with the stack dir's", name, metaFile)
|
|
}
|
|
return m.waitUpdateHealthy(ctx, name, timeout)
|
|
}
|
|
|
|
func (m *Manager) healthTimeout() time.Duration {
|
|
if m.cfg == nil {
|
|
return 5 * time.Minute
|
|
}
|
|
return m.cfg.Update.HealthTimeoutDuration()
|
|
}
|
|
|
|
func (m *Manager) backupMaxAge() time.Duration {
|
|
if m.cfg == nil {
|
|
return 24 * time.Hour
|
|
}
|
|
return m.cfg.Update.BackupMaxAgeDuration()
|
|
}
|
|
|
|
// pre-update copies, kept in the stack dir so they travel with it. Neither name is one the syncer
|
|
// copies (it copies exactly docker-compose.yml and .felhom.yml).
|
|
const (
|
|
preUpdateComposeFile = "pre-update-compose.yml"
|
|
preUpdateAppliedFile = "pre-update-applied.yml"
|
|
)
|
|
|
|
func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
|
start := m.now()
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
m.finishUpdate(name, UpdatePhaseFailed, fmt.Sprintf("stack %q not found", name))
|
|
return
|
|
}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
g := m.guards()
|
|
entry := updateJournalEntry{StartedAt: start}
|
|
// decision 53: what the app ran before this press — kept as the previous image if the update ends done.
|
|
if st.AppConfig != nil && len(st.AppConfig.InstalledImages) > 0 {
|
|
entry.BeforeImages = make(map[string]InstalledImage, len(st.AppConfig.InstalledImages))
|
|
for k, v := range st.AppConfig.InstalledImages {
|
|
entry.BeforeImages[k] = v
|
|
}
|
|
}
|
|
fail := func(key, detail string, args ...interface{}) {
|
|
m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail)
|
|
m.clearJournal(name)
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, key, "", args...)
|
|
}
|
|
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) {
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.journal_failed", "")
|
|
return
|
|
}
|
|
if g == nil {
|
|
fail("update.error.no_guards", "no UpdateGuards wired")
|
|
return
|
|
}
|
|
// v0.268.0 — THE LADDER (`09` §3 decision 14): which definition this ONE press pins. Decided
|
|
// first, before anything moves, so a step the catalog promises and does not carry refuses here
|
|
// rather than jumping past it. An unpinned app is left to today's behaviour (advancePinTo no-ops).
|
|
stepSrc := m.CatalogTemplatePath(name, "docker-compose.yml")
|
|
stepMeta := m.CatalogTemplatePath(name, ".felhom.yml")
|
|
if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 {
|
|
step, serr := nextLadderStep(filepath.Dir(stepSrc), cfg.PinnedImages)
|
|
if serr != nil {
|
|
fail("update.error.pin_failed", "update ladder: "+serr.Error())
|
|
return
|
|
}
|
|
stepSrc, stepMeta = step.Source, step.Meta
|
|
m.logger.Printf("[INFO] [stacks] update %s: ladder — %s", name, step.Why)
|
|
// v0.273.0 — A POSTGRESQL MAJOR MOVES ONLY WITH THE TEST'S MARK (`09` §6.4 part 10, B1). Decided
|
|
// here, before anything moves; the preflight asked the same question (conversionRefusal).
|
|
conv, vol, key, cerr := m.conversionFor(st, cfg.PinnedImages, step)
|
|
if cerr != nil {
|
|
if key == "" {
|
|
fail("update.error.pin_failed", "engine conversion: "+cerr.Error())
|
|
} else {
|
|
fail(key, "engine conversion: "+cerr.Error(), appLabel(st))
|
|
}
|
|
return
|
|
}
|
|
if conv != nil {
|
|
entry.Convert, entry.ConvertVolume = conv, vol
|
|
m.logger.Printf("[INFO] [stacks] update %s: this step CONVERTS %s PostgreSQL %d → %d (the ladder entry's mark); database volume %s", name, conv.Service, conv.From, conv.To, vol)
|
|
}
|
|
}
|
|
// R-665 (v0.269.0): the new version is judged by ITS OWN .felhom.yml, journaled so a resumed verify
|
|
// uses it too — never by the stack dir's, which a restore rewrites with the unit's older file until
|
|
// the next catalog sync (measured on 9202 2026-09-24: a failing edge passed on the restored probe).
|
|
entry.NewMeta = stepMeta
|
|
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
|
|
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
|
|
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
|
|
maxAge := m.backupMaxAge()
|
|
deployedAt := m.currentDeployTime(name)
|
|
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
|
|
if ok {
|
|
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
|
|
} else {
|
|
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
|
fail("update.error.journal_failed", "journal write failed")
|
|
return
|
|
}
|
|
if err := g.BackupNow(ctx, name); err != nil {
|
|
fail("update.error.backup_failed", "pre-update backup: "+err.Error(), err.Error())
|
|
return
|
|
}
|
|
now := m.now()
|
|
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
|
|
if !ok {
|
|
fail("update.error.backup_no_unit", fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
|
|
}
|
|
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
|
|
entry.ProvenTier = rp.Tier
|
|
|
|
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
|
|
fail("update.error.journal_failed", "journal write failed")
|
|
return
|
|
}
|
|
paths, err := g.SafetyDump(ctx, name)
|
|
if err != nil {
|
|
fail("update.error.dump_failed", "safety dump: "+err.Error(), err.Error())
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
|
|
|
|
// v0.263.0 — the undo's copy is PLANNED here, before anything moves: its size against the disk
|
|
// floor (decision 19's limit). The copy itself is taken after the pull, where the app stops anyway.
|
|
undoVols, perr := m.planUndoCopies(name, dir)
|
|
if perr != nil {
|
|
if se, ok := perr.(*undoSpaceError); ok {
|
|
fail("err.stacks.update_undo_space", "undo copy: "+perr.Error(), se.need, se.free, updateDiskFloorGiB)
|
|
} else {
|
|
fail("err.stacks.update_undo_copy_failed", "undo copy plan: "+perr.Error())
|
|
}
|
|
return
|
|
}
|
|
if entry.Convert != nil {
|
|
covered := false
|
|
for _, v := range undoVols {
|
|
covered = covered || v == entry.ConvertVolume
|
|
}
|
|
if !covered {
|
|
fail("update.error.pin_failed", fmt.Sprintf("engine conversion: the database volume %s is not in the undo copy %v — refusing before anything moves", entry.ConvertVolume, undoVols))
|
|
return
|
|
}
|
|
if err := m.planConversionSpace(name, dir, entry.ConvertVolume); err != nil {
|
|
if se, ok := err.(*convertSpaceError); ok {
|
|
fail("err.stacks.update_convert_space", "engine conversion: "+err.Error(), appLabel(st), humanBytes(se.need), humanBytes(se.free))
|
|
} else {
|
|
fail("err.stacks.update_undo_copy_failed", "engine conversion space: "+err.Error())
|
|
}
|
|
return
|
|
}
|
|
}
|
|
|
|
// PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash
|
|
// at any later instant can put it back (Scenario G).
|
|
prevLive, err := os.ReadFile(st.ComposePath)
|
|
if err != nil {
|
|
fail("update.error.pin_failed", "reading the live compose file: "+err.Error())
|
|
return
|
|
}
|
|
if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil {
|
|
fail("update.error.journal_failed", "saving the pre-update compose copy: "+err.Error())
|
|
return
|
|
}
|
|
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
|
|
// R-639: the OLD .felhom.yml too — its probe is the one the old version answers.
|
|
md, merr := savePreUpdateMeta(dir)
|
|
switch {
|
|
case md == "":
|
|
m.logger.Printf("[WARN] [stacks] update %s: could not keep the previous .felhom.yml (%v) — an undo would check health with the current one", name, merr)
|
|
case merr != nil:
|
|
m.logger.Printf("[WARN] [stacks] update %s: %v", name, merr)
|
|
entry.PrevMeta = md
|
|
default:
|
|
entry.PrevMeta = md
|
|
}
|
|
if applied, aerr := LoadAppliedDefinition(dir); aerr == nil {
|
|
if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil {
|
|
entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile)
|
|
}
|
|
}
|
|
if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 {
|
|
entry.PrevPin = map[string]string{}
|
|
for k, v := range cfg.PinnedImages {
|
|
entry.PrevPin[k] = v
|
|
}
|
|
}
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) {
|
|
m.removePreUpdateCopies(dir)
|
|
fail("update.error.journal_failed", "journal write failed")
|
|
return
|
|
}
|
|
if err := m.advancePinTo(name, dir, stepSrc, stepMeta); err != nil {
|
|
m.pinBack(name, dir, entry)
|
|
fail("update.error.pin_failed", "advancing the pin: "+err.Error())
|
|
return
|
|
}
|
|
if cfg := LoadAppConfig(dir); cfg != nil {
|
|
entry.NewPin = cfg.PinnedImages
|
|
}
|
|
|
|
env := m.stackEnv(dir)
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
|
|
m.pinBack(name, dir, entry)
|
|
fail("update.error.journal_failed", "journal write failed")
|
|
return
|
|
}
|
|
if _, err := m.updateCompose(dir, env, "pull"); err != nil {
|
|
// Scenario E: NOTHING RAN. The containers are the old ones and still running, so the honest
|
|
// state is the old pin and the old file — put both back.
|
|
m.pinBack(name, dir, entry)
|
|
m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err)
|
|
fail("update.error.pull_failed", "pull failed: "+err.Error())
|
|
return
|
|
}
|
|
|
|
// COPYING (v0.263.0) — the app is stopped here anyway to be recreated, so the extra downtime is the
|
|
// copy alone. A failed copy moves nothing: the copies go, the pin goes back, the old containers
|
|
// start again.
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) {
|
|
m.pinBack(name, dir, entry)
|
|
fail("update.error.journal_failed", "journal write failed")
|
|
return
|
|
}
|
|
if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy failed: %v — removing it, putting the pin back and starting the previous version", name, err)
|
|
m.removeUndoCopies(name, entry.UndoCopies)
|
|
m.pinBack(name, dir, entry)
|
|
if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr)
|
|
}
|
|
fail("err.stacks.update_undo_copy_failed", "undo copy: "+err.Error())
|
|
return
|
|
}
|
|
|
|
// CONVERTING (v0.273.0) — only on a step the ladder marks. Any failure is undone like a failed health
|
|
// check: every volume back from its copy (the old datadir included), old pin, old definition.
|
|
if entry.Convert != nil {
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseConverting) {
|
|
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before converting", &entry)
|
|
return
|
|
}
|
|
if err := m.convertEngine(ctx, name, dir, env, &entry); err != nil {
|
|
m.failAndHold(ctx, name, dir, env, rp, "conversion failed: "+err.Error(), &entry)
|
|
return
|
|
}
|
|
}
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
|
|
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up", &entry)
|
|
return
|
|
}
|
|
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
|
// Containers may already have been recreated on the new image — something may have run.
|
|
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error(), &entry)
|
|
return
|
|
}
|
|
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
|
|
}
|
|
|
|
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
|
|
// known, and a failure holds the app.
|
|
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, start time.Time, entry *updateJournalEntry) {
|
|
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
|
|
}
|
|
timeout := m.healthTimeout()
|
|
waitStart := m.now()
|
|
healthy, detail := m.updateHealthFor(ctx, name, timeout, entry.NewMeta)
|
|
if !healthy {
|
|
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail, entry)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
|
|
m.recordInstalledImages(name, dir, env)
|
|
_ = m.RefreshStatus()
|
|
if entry.Convert != nil {
|
|
// B5 (v0.273.0): the OLD datadir's copy stays until a backup of the converted app is proven
|
|
// (ReleaseConversionCopies); every other copy goes as usual.
|
|
var keep *undoCopy
|
|
var rest []undoCopy
|
|
for i, c := range entry.UndoCopies {
|
|
if c.Volume == entry.ConvertVolume {
|
|
keep = &entry.UndoCopies[i]
|
|
continue
|
|
}
|
|
rest = append(rest, c)
|
|
}
|
|
m.removeUndoCopies(name, rest)
|
|
if keep != nil {
|
|
m.recordConversionCopy(name, dir, &ConversionCopy{Volume: keep.Volume, Copy: keep.Copy, At: m.now().UTC().Format(time.RFC3339), From: entry.Convert.From, To: entry.Convert.To, Service: entry.Convert.Service})
|
|
m.logger.Printf("[INFO] [stacks] update %s: KEEPING the pre-conversion datadir copy %s (PostgreSQL %d) until a backup of the converted app is proven", name, keep.Copy, entry.Convert.From)
|
|
}
|
|
_ = os.RemoveAll(filepath.Join(dir, preUpdateConvertDir))
|
|
} else {
|
|
m.removeUndoCopies(name, entry.UndoCopies)
|
|
}
|
|
m.recordUpdateUndone(name, dir, nil) // a successful update ends the "undone" note
|
|
m.clearFailedStep(name, dir) // R-680: and the failed-step record
|
|
m.clearJournal(name)
|
|
m.removePreUpdateCopies(dir)
|
|
// R-678 (v0.271.0): the app's catalog fields — ladder_steps_left, the badge's inputs, the pin — are
|
|
// re-read NOW, before Updating goes false, so neither a person nor the automatic leg ever reads the
|
|
// pre-update values after `done`. MEASURED 2026-09-24: ~50 s stale, six re-presses by the caller.
|
|
// COMPANION RED-PROOF (REPORT.md): drop this scan — TestR678_StepsLeftFreshAtDone fails.
|
|
if err := m.ScanStacks(); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update %s: rescan after the update failed: %v — the page catches up at the next scan", name, err)
|
|
}
|
|
m.finishUpdate(name, UpdatePhaseDone, "")
|
|
m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second))
|
|
m.retainAfterUpdate(name, entry.BeforeImages) // decision 53 (R-736)
|
|
}
|
|
|
|
// holdLogTailLines is how much of each service's log the hold keeps. 400 lines is enough to hold a
|
|
// migration run and a startup failure, and small enough that a hold never fills a disk.
|
|
const holdLogTailLines = "400"
|
|
|
|
// captureHoldLogs writes each service's log into <stackdir>/hold-logs/<ts>/ before the app is
|
|
// stopped. Best-effort by design (R-621): the hold itself must happen either way.
|
|
func (m *Manager) captureHoldLogs(name, dir string, env []string) {
|
|
ts := m.now().UTC().Format("20060102T150405Z")
|
|
outDir := filepath.Join(dir, "hold-logs", ts)
|
|
if err := os.MkdirAll(outDir, 0o755); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: cannot create the hold-log directory %s: %v — the hold still proceeds", name, outDir, err)
|
|
return
|
|
}
|
|
out, err := m.updateCompose(dir, env, "logs", "--no-color", "--tail", holdLogTailLines)
|
|
if err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update %s: `compose logs` before the hold failed: %v — writing what came back anyway", name, err)
|
|
}
|
|
path := filepath.Join(outDir, "compose-logs.txt")
|
|
if werr := os.WriteFile(path, []byte(out), 0o644); werr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: could not write %s: %v", name, path, werr)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: kept %d bytes of the app's own log at %s before stopping it (R-621)", name, len(out), path)
|
|
}
|
|
|
|
// failAndHold is Scenario F. Since v0.263.0 it first UNDOES (decision 15): when the job took its
|
|
// last-second copy, the old version goes back with the data from before the update, and the app is
|
|
// HELD only when that undo fails too. Without a copy (an update journaled by an older controller,
|
|
// resumed after an upgrade) it holds as it always did.
|
|
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string, entry *updateJournalEntry) {
|
|
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s", name, why)
|
|
// R-621: capture the app's own logs BEFORE the `down`, because the `down` destroys them. Two
|
|
// drill nights lost the only evidence of WHY an update failed this way — `adventurelog` ran nine
|
|
// migrations and then never bound its port, and the log that would have said so was gone by the
|
|
// time anyone looked. The capture is bounded and best-effort: a hold must never fail because its
|
|
// evidence could not be written.
|
|
m.captureHoldLogs(name, dir, env)
|
|
undoState := ""
|
|
if entry != nil && entry.Copied {
|
|
if undoState = m.tryUndo(ctx, name, dir, why, entry); undoState == "" {
|
|
return // undone: the previous version runs on the pre-update data
|
|
}
|
|
m.logger.Printf("[ERROR] [stacks] update %s: the UNDO failed too (%s) — HOLDING the app; the undo copies are kept: %v", name, undoState, entry.UndoCopies)
|
|
} else {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: no undo copy for this update — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name)
|
|
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
|
}
|
|
}
|
|
holdWhy := ""
|
|
if g := m.guards(); g == nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
|
|
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
|
} else if _, w := holdForLang(g, name, i18n.Default); w != "" {
|
|
holdWhy = w
|
|
m.markUpdateHeld(name)
|
|
}
|
|
if entry != nil { // R-680: the automatic leg never presses this step again on this ladder
|
|
m.recordFailedStep(name, dir, entry.NewPin, "held")
|
|
}
|
|
_ = m.RefreshStatus()
|
|
m.clearJournal(name)
|
|
m.removePreUpdateCopies(dir)
|
|
if holdWhy == "" {
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.hold_unsaved", "")
|
|
} else {
|
|
// R-647 (v0.265.0): stored as the HELD key + the Hungarian sentence, so every reader — the
|
|
// `?lang=` switch, an operator reading a box in the other language — gets the hold sentence in
|
|
// THEIR language (UpdateErrorFor), and the stored text is Hungarian as for every other key.
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, UpdateErrorKeyHeld, holdWhy)
|
|
}
|
|
m.emitUpdateEvent(UpdateEventHeld, name, entry, rp, holdWhy != "")
|
|
}
|
|
|
|
// pinBack restores the pin, the stored definition and the live file from the journaled copies, and
|
|
// then removes the copies.
|
|
func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
|
|
m.restoreDefinition(name, dir, entry)
|
|
m.removePreUpdateCopies(dir)
|
|
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
|
|
}
|
|
|
|
// restoreDefinition is pinBack WITHOUT removing the copies — the undo's form, so a power cut after it
|
|
// can run it again (RecoverUpdates → undoing) and find the copies still there. It also puts the pinned
|
|
// version's .felhom.yml record back (v0.263.2), so the NEXT update's undo probes the right version.
|
|
func (m *Manager) restoreDefinition(name, dir string, entry updateJournalEntry) {
|
|
if entry.PrevMeta != "" {
|
|
m.storeAppliedMetaFrom(name, dir, filepath.Join(entry.PrevMeta, ".felhom.yml"))
|
|
}
|
|
prevLive, lerr := os.ReadFile(entry.PrevCompose)
|
|
if lerr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr)
|
|
}
|
|
if len(entry.PrevPin) > 0 {
|
|
applied := prevLive
|
|
if entry.PrevApplied != "" {
|
|
if b, err := os.ReadFile(entry.PrevApplied); err == nil {
|
|
applied = b
|
|
}
|
|
}
|
|
if err := m.SetPin(name, dir, entry.PrevPin, applied); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: putting the pin back failed: %v", name, err)
|
|
}
|
|
}
|
|
if lerr == nil {
|
|
if err := os.WriteFile(ComposePathIn(dir), prevLive, 0o644); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err)
|
|
}
|
|
}
|
|
}
|
|
|
|
func (m *Manager) removePreUpdateCopies(dir string) {
|
|
_ = os.Remove(filepath.Join(dir, preUpdateComposeFile))
|
|
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
|
|
_ = os.RemoveAll(filepath.Join(dir, preUpdateMetaDir))
|
|
}
|
|
|
|
// settleReason names WHY the wait fell back to container state, so the journal and the log do not
|
|
// have to be read together to tell "this app declares no check" from "this app declares one that
|
|
// resolves to no container" (R-630).
|
|
func settleReason(meta Metadata) string {
|
|
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
|
return "no probe container — settled on container state"
|
|
}
|
|
return "no health check declared"
|
|
}
|
|
|
|
// waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the
|
|
// existing probe, or — for an app with none — every container running and none restarting for
|
|
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
|
|
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
|
return m.waitUpdateHealthyMeta(ctx, name, timeout, nil)
|
|
}
|
|
|
|
// waitUpdateHealthyMeta is the health wait with the probe taken from `override` instead of the app's
|
|
// current .felhom.yml — the undo's form (the old version is judged by the old probe). nil = current.
|
|
func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeout time.Duration, override *Metadata) (bool, string) {
|
|
deadline := m.now().Add(timeout)
|
|
var runningSince time.Time
|
|
warnedNoProbe := false
|
|
last := "no observation yet"
|
|
for {
|
|
_ = m.RefreshStatus()
|
|
st, ok := m.GetStack(name)
|
|
// v0.263.0 — THE UNDO'S PROBE MUST BE ALLOWED TO RUN. The periodic health probe judges the app
|
|
// with the CURRENT .felhom.yml and flips a running app to StateUnhealthy when that check fails —
|
|
// which is exactly the undo's situation when the new version brought a probe the old one does
|
|
// not answer. Gating on StateRunning alone meant the old probe was never asked and the undo was
|
|
// reported "did not start" after the full timeout. MEASURED LIVE on 9202 2026-09-23 (docmost,
|
|
// `last: state unhealthy` for 90 s while the old version served). So with an override that
|
|
// declares checks, an Unhealthy app is PROBED — and the override's own check decides. It never
|
|
// settles an Unhealthy app on container state (below: the settle path requires StateRunning).
|
|
// Pinned by TestUndo_OldProbeRunsOnAnAppTheCurrentProbeMarkedUnhealthy.
|
|
undoProbe := ok && override != nil && st.State == StateUnhealthy &&
|
|
override.HealthCheck != nil && len(override.HealthCheck.Checks) > 0
|
|
switch {
|
|
case !ok:
|
|
last = "stack vanished"
|
|
runningSince = time.Time{}
|
|
case st.State == StateRunning || undoProbe:
|
|
// A declared health check is only usable if it resolves to a container. When it does
|
|
// not, the app is judged the same way an app with NO declared check is judged —
|
|
// settling on container state — and the log says so.
|
|
//
|
|
// R-630, and this `else` is the whole defect: the old code set `last = "no probe
|
|
// container"` and looped, so `verifying` spent the FULL `update.health_timeout` and
|
|
// `failAndHold` then STOPPED an app whose containers were all healthy. Measured on
|
|
// paperless-ngx 2026-09-22: `done` was never reachable, `failed` at +313.0 s, front door
|
|
// 404 afterwards. A stack with no probe is not "healthy" and it is not "failing" — it is
|
|
// SETTLED ON CONTAINER STATE (`09` §3), and never a reason to stop a running app.
|
|
usable := false
|
|
meta := st.Meta
|
|
if override != nil {
|
|
meta = *override
|
|
}
|
|
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
|
c, candidates := findProbeContainerMeta(name, &meta, st.Containers)
|
|
if c != "" {
|
|
usable = true
|
|
res := m.probeRun(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.HealthProbe = res
|
|
}
|
|
m.mu.Unlock()
|
|
if res.Healthy {
|
|
return true, "the app's health check passed"
|
|
}
|
|
last = "health check failing"
|
|
} else if !warnedNoProbe {
|
|
warnedNoProbe = true
|
|
m.logger.Printf("[WARN] [stacks] update %s: no probe container for %s — settling on container state instead; candidates: %v",
|
|
name, name, candidates)
|
|
}
|
|
}
|
|
if !usable && st.State != StateRunning {
|
|
last = "unhealthy, and the check resolves to no container"
|
|
} else if !usable {
|
|
if runningSince.IsZero() {
|
|
runningSince = m.now()
|
|
}
|
|
if m.now().Sub(runningSince) >= updateSettleWindow {
|
|
return true, fmt.Sprintf("all containers running, none restarting, for %s (%s)",
|
|
updateSettleWindow, settleReason(meta))
|
|
}
|
|
last = "running, settling"
|
|
}
|
|
default:
|
|
runningSince = time.Time{}
|
|
last = "state " + string(st.State)
|
|
}
|
|
if !m.now().Before(deadline) {
|
|
return false, fmt.Sprintf("not healthy within %s (last: %s)", timeout, last)
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
return false, "cancelled: " + ctx.Err().Error()
|
|
case <-time.After(updatePollEvery):
|
|
}
|
|
}
|
|
}
|
|
|
|
// ── the journal ──────────────────────────────────────────────────────────────────────────────────
|
|
|
|
type updateJournalEntry struct {
|
|
Phase string `json:"phase"`
|
|
StartedAt time.Time `json:"started_at"`
|
|
PrevPin map[string]string `json:"prev_pin,omitempty"`
|
|
// BeforeImages (decision 53): installed_images at the press — the previous image kept if the update ends done.
|
|
BeforeImages map[string]InstalledImage `json:"before_images,omitempty"`
|
|
PrevCompose string `json:"prev_compose,omitempty"`
|
|
PrevApplied string `json:"prev_applied,omitempty"`
|
|
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
|
|
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
|
|
// the right copy. 0 in a journal written by v0.238.1 or older.
|
|
ProvenTier int `json:"proven_tier,omitempty"`
|
|
// v0.263.0 — the undo (undo.go). UndoCopies is journaled BEFORE each copy starts; Copied is true
|
|
// only once every copy finished (it may be true with zero copies: an app with no named volume).
|
|
// PrevMeta is the directory holding the previous .felhom.yml; NewPin the pin the update moved to.
|
|
UndoCopies []undoCopy `json:"undo_copies,omitempty"`
|
|
Copied bool `json:"copied,omitempty"`
|
|
PrevMeta string `json:"prev_meta,omitempty"`
|
|
NewPin map[string]string `json:"new_pin,omitempty"`
|
|
// NewMeta (v0.269.0, R-665/R-664) is the NEW version's own .felhom.yml — the catalog's, or the
|
|
// ladder step's — used for the verify, so a resumed verify judges by the same file.
|
|
NewMeta string `json:"new_meta,omitempty"`
|
|
// v0.273.0 (pgconvert.go): the step's engine conversion, the DB volume it rebuilds, and whether that
|
|
// volume has been touched (emptied) — journaled so a restart during `converting` runs the undo.
|
|
Convert *EngineConversion `json:"convert,omitempty"`
|
|
ConvertVolume string `json:"convert_volume,omitempty"`
|
|
ConvertTouched bool `json:"convert_touched,omitempty"`
|
|
}
|
|
|
|
type updateJournal struct {
|
|
Updates map[string]updateJournalEntry `json:"updates"`
|
|
}
|
|
|
|
func (m *Manager) updateJournalPath() string {
|
|
return filepath.Join(m.cfg.Paths.DataDir, "update-journal.json")
|
|
}
|
|
|
|
func (m *Manager) readUpdateJournal() updateJournal {
|
|
j := updateJournal{Updates: map[string]updateJournalEntry{}}
|
|
data, err := os.ReadFile(m.updateJournalPath())
|
|
if err != nil {
|
|
return j
|
|
}
|
|
if err := json.Unmarshal(data, &j); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update journal at %s is corrupt (%v) — quarantining", m.updateJournalPath(), err)
|
|
_ = os.Rename(m.updateJournalPath(), fmt.Sprintf("%s.corrupt-%d", m.updateJournalPath(), time.Now().Unix()))
|
|
return updateJournal{Updates: map[string]updateJournalEntry{}}
|
|
}
|
|
if j.Updates == nil {
|
|
j.Updates = map[string]updateJournalEntry{}
|
|
}
|
|
return j
|
|
}
|
|
|
|
// writeUpdateJournal is atomic and fsynced (the AppStopGuard shape): the point is surviving a power cut.
|
|
func (m *Manager) writeUpdateJournal(j updateJournal) error {
|
|
p := m.updateJournalPath()
|
|
if len(j.Updates) == 0 {
|
|
if err := os.Remove(p); err != nil && !os.IsNotExist(err) {
|
|
return err
|
|
}
|
|
return nil
|
|
}
|
|
data, err := json.MarshalIndent(j, "", " ")
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
|
|
return err
|
|
}
|
|
tmp := p + ".tmp"
|
|
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if _, err := f.Write(data); err != nil {
|
|
f.Close()
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if err := f.Sync(); err != nil {
|
|
f.Close()
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if err := f.Close(); err != nil {
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
return os.Rename(tmp, p)
|
|
}
|
|
|
|
// enterUpdatePhase journals the phase BEFORE it starts and mirrors it for the UI. False means the
|
|
// journal could not be written — the caller must not perform a mutation it could not record.
|
|
func (m *Manager) enterUpdatePhase(name string, entry *updateJournalEntry, phase string) bool {
|
|
entry.Phase = phase
|
|
m.updateJournalMu.Lock()
|
|
j := m.readUpdateJournal()
|
|
j.Updates[name] = *entry
|
|
err := m.writeUpdateJournal(j)
|
|
m.updateJournalMu.Unlock()
|
|
m.setUpdatePhase(name, phase)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: journal write for phase %s failed: %v", name, phase, err)
|
|
return false
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: phase %s", name, phase)
|
|
return true
|
|
}
|
|
|
|
func (m *Manager) clearJournal(name string) {
|
|
m.updateJournalMu.Lock()
|
|
defer m.updateJournalMu.Unlock()
|
|
j := m.readUpdateJournal()
|
|
delete(j.Updates, name)
|
|
if err := m.writeUpdateJournal(j); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: clearing the journal entry failed: %v", name, err)
|
|
}
|
|
}
|
|
|
|
// RecoverUpdates reads the journal at startup (Scenario G). Call it BEFORE the boot reconciler.
|
|
//
|
|
// - interrupted BEFORE the pin moved (checking, backing-up, safety-dump): nothing moved — the entry is
|
|
// dropped and the app carries the "interrupted" sentence;
|
|
// - interrupted while pinning or pulling: nothing RAN — the pin and definition are put back (as E);
|
|
// - interrupted while starting or verifying: something may have run — the app is marked Updating
|
|
// (so the boot sweep and the dead-app alarm leave it alone) and queued for ResumeInterruptedUpdates,
|
|
// which re-runs `up -d` and the health wait, ending in A or F.
|
|
//
|
|
// It needs no backup wiring, because nothing here holds an app — that is left to the resumed job.
|
|
func (m *Manager) RecoverUpdates() []string {
|
|
m.updateJournalMu.Lock()
|
|
j := m.readUpdateJournal()
|
|
m.updateJournalMu.Unlock()
|
|
if len(j.Updates) == 0 {
|
|
return nil
|
|
}
|
|
var resumed []string
|
|
for name, e := range j.Updates {
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s is in the journal (phase %s) but no longer exists — dropping the entry", name, e.Phase)
|
|
m.clearJournal(name)
|
|
continue
|
|
}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
switch e.Phase {
|
|
case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
|
m.clearJournal(name)
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
|
case UpdatePhasePinning, UpdatePhasePulling:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
|
m.pinBack(name, dir, e)
|
|
m.clearJournal(name)
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
|
case UpdatePhaseCopying:
|
|
// v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the
|
|
// pin goes back, and the previous version is started again.
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while copying its data (started %s) — nothing new ran; removing the partial copy, putting the pin back and starting the previous version", name, e.StartedAt.Format(time.RFC3339))
|
|
m.removeUndoCopies(name, e.UndoCopies)
|
|
m.pinBack(name, dir, e)
|
|
if _, err := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err)
|
|
}
|
|
m.clearJournal(name)
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
|
case UpdatePhaseUndoing, UpdatePhaseConverting:
|
|
// v0.273.0: a restart during `converting` is UNDONE the same way — the database volume may be
|
|
// emptied or half-loaded, and only the copy is known-good (B4). Never "done".
|
|
// v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from
|
|
// the copies (still there: they are removed only after the undo succeeded) and then probes.
|
|
// Never "done": what ran last was a failed new version.
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while %s (started %s) — marking it Updating and RESUMING the undo", name, map[string]string{UpdatePhaseUndoing: "UNDOING", UpdatePhaseConverting: "CONVERTING its database (the volume may be emptied)"}[e.Phase], e.StartedAt.Format(time.RFC3339))
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseUndoing, UpdatePhaseLabel(UpdatePhaseUndoing)
|
|
}
|
|
m.updateResume = append(m.updateResume, name)
|
|
m.mu.Unlock()
|
|
resumed = append(resumed, name)
|
|
case UpdatePhaseStarting, UpdatePhaseVerifying:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
|
|
}
|
|
m.updateResume = append(m.updateResume, name)
|
|
m.mu.Unlock()
|
|
resumed = append(resumed, name)
|
|
default:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s has unknown phase %q — dropping the entry", name, e.Phase)
|
|
m.clearJournal(name)
|
|
}
|
|
}
|
|
return resumed
|
|
}
|
|
|
|
// ResumeInterruptedUpdates continues the updates RecoverUpdates queued, once the backup side is wired
|
|
// (a resumed update that fails must be able to HOLD). Returns how many were resumed.
|
|
func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
|
m.mu.Lock()
|
|
names := m.updateResume
|
|
m.updateResume = nil
|
|
m.mu.Unlock()
|
|
for _, name := range names {
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
|
continue
|
|
}
|
|
m.updateJournalMu.Lock()
|
|
e, ok := m.readUpdateJournal().Updates[name]
|
|
m.updateJournalMu.Unlock()
|
|
if !ok {
|
|
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
|
continue
|
|
}
|
|
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
|
|
rp := UpdateRestorePoint{Tier: e.ProvenTier, ProvenAt: provenAt}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
|
|
env := m.stackEnv(dir)
|
|
if e.Phase == UpdatePhaseUndoing || e.Phase == UpdatePhaseConverting {
|
|
why := "resumed after a restart during the undo"
|
|
if e.Phase == UpdatePhaseConverting {
|
|
why = "the controller restarted during the database conversion — undoing it"
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: resuming the UNDO after a controller restart (was %s)", name, e.Phase)
|
|
m.failAndHold(ctx, name, dir, env, rp, why, &e)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
|
|
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
|
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error(), &e)
|
|
return
|
|
}
|
|
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
|
|
}(name, dir, e, rp)
|
|
}
|
|
return len(names)
|
|
}
|
|
|
|
// probeRun is the network half of the update's health wait — its own seam, so a test can drive the
|
|
// REAL wait loop (the state gate, the container resolution, the settle rule) with only the HTTP/TCP
|
|
// probe faked. nil ⇒ runChecks.
|
|
func (m *Manager) probeRun(t probeTarget) *HealthProbeResult {
|
|
if m.probeRunFn != nil {
|
|
return m.probeRunFn(t)
|
|
}
|
|
return m.runChecks(t)
|
|
}
|