2cd66663f3
gates / gates (push) Successful in 25s
Found live on 9202 (romm): .felhom.yml flows into the stack dir on every catalog sync, so "the old .felhom.yml" saved at update time was already the new one, and the serving old version was judged with the new probe. New record applied-meta/.felhom.yml, written whenever a version is pinned (deploy, adoption, pin advance) and put back by the undo, like applied-compose.yml. The fixture now places the new file at sync time. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
1248 lines
56 KiB
Go
1248 lines
56 KiB
Go
package stacks
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
|
)
|
|
|
|
// ── The guarded update (update arc slice 4, controller v0.237.0) ─────────────────────────────────
|
|
//
|
|
// WHAT IT REPLACED. `UpdateStack` advanced the pin, pulled, ran `up -d` and returned — no copy first,
|
|
// no check of memory, disk, a running backup or a held app, and a success the moment `up` returned.
|
|
// SPIKE-app-update-2026-09-01 §4 measured that as HTTP 200 over an app that was already crash-looping
|
|
// (R-443), and R-439 is that a held app could be updated at all.
|
|
//
|
|
// THE SEQUENCE, and the order is the point:
|
|
//
|
|
// checking → backing-up (only if the proven copy is too old) → safety-dump → pinning → pulling
|
|
// → copying → starting → verifying → done | (undoing → undone | failed)
|
|
//
|
|
// 1. The PRECONDITION is the app's existing verified backup (operator ruling 2026-09-02,
|
|
// 09-update-architecture §3 decision 1): an openable Tier-2 unit with a PROVEN copy date. The
|
|
// same predicate that permits the destructive „Teljes visszaállítás" permits the update — the
|
|
// route back IS that restore, so an update without it has no route back.
|
|
// 2. The safety dump is taken BEFORE the pin moves: it is "the state the customer was in a minute ago",
|
|
// and a minute later the migration may have run.
|
|
// 3. The pin moves BEFORE the pull (v0.235.0's reason: pull and up act on the file on disk).
|
|
// 4. A PULL failure puts the pin BACK — nothing ran, so reverting is safe and honest (Scenario E).
|
|
// 5. A HEALTH failure leaves the pin where it is — the new version's migration may have run, and a
|
|
// pin claiming the old version would be a record of something untrue (Scenario F). The app is
|
|
// HELD STOPPED and the customer is told which backup it can be restored from.
|
|
//
|
|
// SINCE v0.263.0 IT PUTS THE OLD VERSION BACK ITSELF — with the data from before the update (09 §3
|
|
// decision 15, undo.go). Until then it deliberately did not: SPIKE-upgrade-test-2026-09-06 measured
|
|
// that the old image alone refuses data a new one migrated (Docmost, Nextcloud). The undo does not put
|
|
// the old image on migrated data; it puts the pre-migration data back as well. A HOLD now happens only
|
|
// when that undo fails too.
|
|
//
|
|
// CRASH SAFETY IS A JOURNAL, NOT A DEFER. A SIGKILL runs no deferred function (Campaign 8 fault 10),
|
|
// so every phase is written to `update-journal.json` BEFORE it starts, and RecoverUpdates reads it at
|
|
// the next startup (Scenario G) — the AppStopGuard pattern.
|
|
|
|
// Update phases, as recorded in the journal and served as Stack.UpdatePhase.
|
|
const (
|
|
UpdatePhaseChecking = "checking"
|
|
UpdatePhaseBackingUp = "backing-up"
|
|
UpdatePhaseSafetyDump = "safety-dump"
|
|
UpdatePhasePinning = "pinning"
|
|
UpdatePhasePulling = "pulling"
|
|
UpdatePhaseStarting = "starting"
|
|
UpdatePhaseVerifying = "verifying"
|
|
UpdatePhaseDone = "done"
|
|
UpdatePhaseFailed = "failed"
|
|
// UpdatePhaseCopying, UpdatePhaseUndoing and UpdatePhaseUndone live in undo.go.
|
|
)
|
|
|
|
// updatePhaseLabels are the customer labels (slice 4 Part 4, exact). `pinning` has no row in the
|
|
// specification — it is instantaneous and is the first step of fetching the new version, so it shares
|
|
// the pull's label rather than inventing a sentence nobody would read.
|
|
var updatePhaseLabels = map[string]string{
|
|
UpdatePhaseChecking: "Ellenőrzés…",
|
|
UpdatePhaseBackingUp: "Biztonsági mentés készül a frissítés előtt…",
|
|
UpdatePhaseSafetyDump: "Adatbázis pillanatkép…",
|
|
UpdatePhasePinning: "Új verzió letöltése…",
|
|
UpdatePhasePulling: "Új verzió letöltése…",
|
|
UpdatePhaseStarting: "Indítás az új verzióval…",
|
|
UpdatePhaseVerifying: "Működés ellenőrzése…",
|
|
UpdatePhaseDone: "Frissítve",
|
|
UpdatePhaseFailed: "A frissítés nem sikerült",
|
|
// v0.263.0 — born as bundle keys (update.phase.*); TestUndo_PhaseLabelsMatchTheBundle pins them equal.
|
|
UpdatePhaseCopying: "Az adatok másolása a frissítés előtt…",
|
|
UpdatePhaseUndoing: "Visszaállítás az előző változatra…",
|
|
UpdatePhaseUndone: "Visszaállítva az előző változatra",
|
|
}
|
|
|
|
// UpdatePhaseLabel returns the customer label for a phase, "" for an unknown one.
|
|
func UpdatePhaseLabel(phase string) string { return updatePhaseLabels[phase] }
|
|
|
|
// Customer sentences. Named so tests compare against the constant, never a retyped literal (R-364).
|
|
const (
|
|
MsgUpdateNoGuards = "A frissítés nem indítható: a frissítés előtti biztonsági ellenőrzés nem érhető el ezen a szerveren."
|
|
MsgUpdateNotDeployed = "Az alkalmazás nincs telepítve, ezért nem frissíthető."
|
|
MsgUpdateDeployingFmt = "A(z) %s telepítése még folyamatban van — a frissítés utána indítható."
|
|
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
|
|
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
|
|
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
|
|
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani, és most új mentés sem készíthető róla. Ellenőrizd a Mentések oldalon, hogy az alkalmazás meghajtója elérhető-e — utána a frissítés elindítható."
|
|
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
|
|
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdateDumpFailFmt = "A frissítés nem indult el, mert az adatbázis pillanatkép nem készült el: %v. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdatePinFailed = "A frissítés nem indult el: az új verzió leírása nem olvasható be. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdateJournalFailed = "A frissítés nem indult el: a frissítés naplója nem menthető. Az alkalmazás változatlanul fut tovább."
|
|
MsgUpdatePullFailed = "Az új verzió letöltése nem sikerült, ezért a frissítés elmaradt. Az alkalmazás a korábbi verzióval fut tovább."
|
|
MsgUpdateInterrupted = "A frissítés megszakadt, mert a vezérlő újraindult, mielőtt az új verzió elindult volna. Az alkalmazás a korábbi verzióval fut tovább."
|
|
MsgUpdateHoldUnsaved = "A frissítés nem sikerült, az alkalmazás le lett állítva, de a leállítás rögzítése nem sikerült. Ne indítsd újra — vedd fel velünk a kapcsolatot."
|
|
)
|
|
|
|
// updateDiskFloorGiB is the free space the Docker data root must have before a pull. A FIXED FLOOR,
|
|
// stated as such: the new image set's size is not known without a registry query (the catalog
|
|
// records tags, not sizes, and §8.1 of 09 already declines registry calls on the customer box), so
|
|
// the rule is "not less than 2 GB", not "enough for these images".
|
|
const updateDiskFloorGiB = 2.0
|
|
|
|
// updateSettleWindow is the rule for an app with no .felhom.yml health check: every container
|
|
// running, none restarting, for this long.
|
|
const updateSettleWindow = 60 * time.Second
|
|
|
|
// updatePollEvery is how often the health wait re-reads the stack.
|
|
const updatePollEvery = 5 * time.Second
|
|
|
|
// Backup tiers (R-475), mirroring backup.UpdateTier* — stacks cannot import backup, so
|
|
// TestR475_TierConstantsAgree (cmd/controller) pins the two sets equal.
|
|
const (
|
|
UpdateTierLocal = 1 // the app's own recovery unit
|
|
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
|
|
UpdateTierOffsite = 3 // off-site
|
|
)
|
|
|
|
// UpdateRestorePoint is one proven, restorable copy the update may lean on. The backup side returns
|
|
// only proven, restorable copies (never an attempt clock, never an unopenable unit), so there is no
|
|
// "maybe" field here to forget to check.
|
|
type UpdateRestorePoint struct {
|
|
Tier int // UpdateTierSecondDrive / UpdateTierLocal / UpdateTierOffsite
|
|
ProvenAt time.Time // when the data in that copy was last proven written
|
|
}
|
|
|
|
func updateTierName(tier int) string {
|
|
switch tier {
|
|
case UpdateTierSecondDrive:
|
|
return "Tier 2 (second drive)"
|
|
case UpdateTierLocal:
|
|
return "Tier 1 (own recovery unit)"
|
|
case UpdateTierOffsite:
|
|
return "Tier 3 (off-site)"
|
|
}
|
|
return fmt.Sprintf("tier %d", tier)
|
|
}
|
|
|
|
// freshRestorePoint is THE age rule (backup_max_age), applied to whichever tier is being considered
|
|
// — R-475 Scenario M: a stale copy on ANY tier is stale.
|
|
//
|
|
// COMPANION RED-PROOF M (REPORT.md): check the age only for Tier 2 (let any other tier through
|
|
// whatever its age). TestR475_M_TheAgeRuleAppliesToTheChosenTier then fails: a 30-hour-old copy of
|
|
// the app's own unit carries the update with no backup first.
|
|
func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePoint) bool {
|
|
return func(p UpdateRestorePoint) bool {
|
|
return !p.ProvenAt.IsZero() && now.Sub(p.ProvenAt) <= maxAge
|
|
}
|
|
}
|
|
|
|
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
|
|
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
|
|
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
|
|
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
|
|
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
|
|
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
|
|
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
|
|
fresh := freshRestorePoint(now, maxAge)
|
|
return func(p UpdateRestorePoint) bool {
|
|
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
|
|
return false
|
|
}
|
|
return fresh(p)
|
|
}
|
|
}
|
|
|
|
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
|
|
func (m *Manager) currentDeployTime(name string) time.Time {
|
|
st, ok := m.GetStack(name)
|
|
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
|
|
return time.Time{}
|
|
}
|
|
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
|
|
if err != nil {
|
|
return time.Time{}
|
|
}
|
|
return t
|
|
}
|
|
|
|
// ClearUpdateState wipes everything a finished-or-abandoned update left on a stack. R-614: a fresh
|
|
// install of an app that had previously failed an update showed the OLD phase — „Frissítve" on a
|
|
// deploy that had just happened — because a remove cleared the directory and not the in-memory
|
|
// record. The name is the only thing the next install shares with the last one, so the record has to
|
|
// go when the app does.
|
|
func (m *Manager) ClearUpdateState(name string) {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
s, ok := m.stacks[name]
|
|
if !ok {
|
|
return
|
|
}
|
|
s.Updating = false
|
|
s.UpdatePhase = ""
|
|
s.UpdatePhaseLabel = ""
|
|
s.UpdateError = ""
|
|
s.updateHeld = false
|
|
s.HealthProbe = nil
|
|
}
|
|
|
|
func (m *Manager) markUpdateHeld(name string) {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.updateHeld = true
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
func fmtDeployTime(t time.Time) string {
|
|
if t.IsZero() {
|
|
return "unknown"
|
|
}
|
|
return t.UTC().Format(time.RFC3339)
|
|
}
|
|
|
|
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
|
|
if len(pts) == 0 {
|
|
return "none"
|
|
}
|
|
parts := make([]string, 0, len(pts))
|
|
for _, p := range pts {
|
|
parts = append(parts, fmt.Sprintf("%s at %s (%s old)", updateTierName(p.Tier), p.ProvenAt.UTC().Format(time.RFC3339), now.Sub(p.ProvenAt).Round(time.Minute)))
|
|
}
|
|
return strings.Join(parts, "; ")
|
|
}
|
|
|
|
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
|
|
// backup, so cmd/controller/main.go wires an adapter (TestSlice4_UpdateGuardsAreWiredAtStartup).
|
|
type UpdateGuards interface {
|
|
HoldFor(name string) (bool, string)
|
|
Busy(name string) (bool, string)
|
|
// RestorePoints walks the tiers in preference order (2, 1, 3) and returns the first copy accept
|
|
// admits (nil = any), whether one was found, and every copy looked at (R-475).
|
|
RestorePoints(ctx context.Context, name string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint)
|
|
// CanBackUp reports whether "back up first" can run for this app now (R-475 Scenario L).
|
|
CanBackUp(name string) (bool, string)
|
|
BackupNow(ctx context.Context, name string) error
|
|
SafetyDump(ctx context.Context, name string) ([]string, error)
|
|
// HoldAfterFailedUpdate records the hold. undoState (v0.263.0) says what a FAILED undo left the data
|
|
// as (UndoState*), "" when no undo was attempted; the hold sentence says so.
|
|
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint, undoState string) error
|
|
}
|
|
|
|
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
|
|
// an update that cannot see the backup cannot promise a route back.
|
|
func (m *Manager) SetUpdateGuards(g UpdateGuards) {
|
|
m.mu.Lock()
|
|
m.updateGuards = g
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
// SetSelfUpdatingCheck wires the OTHER half of the v0.261.0 lock: is the CONTROLLER swapping itself?
|
|
//
|
|
// A plain callback rather than a member of UpdateGuards, for two reasons. UpdateGuards is the BACKUP
|
|
// side's interface and this has nothing to do with backups; and `stacks` must never import
|
|
// `selfupdate` (selfupdate reaches the agent, and the import would run the wrong way), so the
|
|
// dependency is inverted here and satisfied in main.go with `updater.IsUpdateRunning`.
|
|
//
|
|
// Nil is safe and means the pre-v0.261.0 behaviour: no self-update gate.
|
|
func (m *Manager) SetSelfUpdatingCheck(fn func() bool) {
|
|
m.mu.Lock()
|
|
m.selfUpdating = fn
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
func (m *Manager) selfUpdatingNow() bool {
|
|
m.mu.RLock()
|
|
fn := m.selfUpdating
|
|
m.mu.RUnlock()
|
|
return fn != nil && fn()
|
|
}
|
|
|
|
func (m *Manager) guards() UpdateGuards {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
return m.updateGuards
|
|
}
|
|
|
|
func fillHoldReason(g UpdateGuards, st *Stack) {
|
|
if st == nil {
|
|
return
|
|
}
|
|
held := false
|
|
if g != nil && st.Deployed {
|
|
if h, why := g.HoldFor(st.Name); h {
|
|
st.HoldReason, held = why, true
|
|
}
|
|
}
|
|
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
|
|
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
|
|
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
|
|
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
|
|
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
|
|
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
|
|
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
|
|
}
|
|
}
|
|
|
|
// UpdateRefusal is a refusal taken before anything moved. Reason is a stable key for logs and tests;
|
|
// Message is the customer sentence.
|
|
type UpdateRefusal struct {
|
|
Reason string
|
|
Message string
|
|
// Cause carries the refusal as an ERROR when the producer made one (v0.253.0, R-557). A
|
|
// util.MsgError there knows its bundle key, so `api.Router.errText` renders the refusal in the
|
|
// household's language; Message stays the Hungarian fallback for every refusal built from a
|
|
// literal. Unwrap is what lets errors.Is and util.AsMsg see through this wrapper.
|
|
Cause error
|
|
}
|
|
|
|
func (r *UpdateRefusal) Error() string {
|
|
if r.Cause != nil {
|
|
return r.Cause.Error()
|
|
}
|
|
return r.Message
|
|
}
|
|
|
|
func (r *UpdateRefusal) Unwrap() error { return r.Cause }
|
|
|
|
func (m *Manager) now() time.Time {
|
|
if m.updateNowFn != nil {
|
|
return m.updateNowFn()
|
|
}
|
|
return time.Now()
|
|
}
|
|
|
|
func (m *Manager) refuseUpdate(name, reason, msg, detail string) *UpdateRefusal {
|
|
m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail)
|
|
return &UpdateRefusal{Reason: reason, Message: msg}
|
|
}
|
|
|
|
// refuseUpdateErr is refuseUpdate for a refusal the producer already built as an error — it keeps the
|
|
// error whole, so its kind and its message key both survive to the API.
|
|
func (m *Manager) refuseUpdateErr(name, reason string, cause error, detail string) *UpdateRefusal {
|
|
m.logger.Printf("[ERROR] [stacks] update %s REFUSED (%s): %s", name, reason, detail)
|
|
return &UpdateRefusal{Reason: reason, Message: cause.Error(), Cause: cause}
|
|
}
|
|
|
|
// UpdatePreflight runs every CHEAP refusal (slice 4 Part 1 + the precondition's existence), in order,
|
|
// and returns the first. Nothing is moved and nothing is recorded by it. The router calls it before
|
|
// recording the customer's intent, so an update that was never going to happen records nothing.
|
|
func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack")
|
|
}
|
|
if !st.Deployed {
|
|
return m.refuseUpdate(name, "not_deployed", MsgUpdateNotDeployed, "not deployed")
|
|
}
|
|
g := m.guards()
|
|
if g == nil {
|
|
return m.refuseUpdate(name, "guards_unwired", MsgUpdateNoGuards, "no UpdateGuards wired — fail closed")
|
|
}
|
|
if st.Deploying {
|
|
return m.refuseUpdate(name, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, name), "a deploy is in progress")
|
|
}
|
|
if st.Updating {
|
|
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "an update is already in progress")
|
|
}
|
|
if held, why := g.HoldFor(name); held {
|
|
return m.refuseUpdate(name, "held", why, "the app is held")
|
|
}
|
|
if busy, why := g.Busy(name); busy {
|
|
return m.refuseUpdate(name, "busy", MsgUpdateBusy, why)
|
|
}
|
|
if m.IsMigrating() {
|
|
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
|
|
}
|
|
// v0.261.0 — the other half of the self-update lock. The controller's swap restarts this process;
|
|
// starting an app update into that is how an update loses its own supervisor mid-flight. TRANSIENT:
|
|
// the household is told to try again in a few minutes, and the caller in `09` §6.2 reads the
|
|
// machine-readable `downgrade`-style reason and retries rather than giving up.
|
|
if m.selfUpdatingNow() {
|
|
return m.refuseUpdateErr(name, "self_updating", util.MsgError("err.stacks.update_self_updating"),
|
|
"the controller is swapping itself — refusing to start an app update into a restart")
|
|
}
|
|
// R-524 — THE PIN NEVER MOVES BACKWARDS WITHOUT THE OPERATOR. MEASURED 2026-09-15 (BIGNIGHT
|
|
// Phase 6): privatebin was updated 2.0.5 → 2.0.6, the catalog was reverted to 2.0.5, and the
|
|
// „Frissítés" button behind the badge would have advanced the pin to the OLDER image — on a
|
|
// datadir the newer version may already have migrated, with §4's ruling saying that cannot be
|
|
// undone. A catalog revert is an operator act on our side; it must never become a data event on
|
|
// the customer's side by itself.
|
|
//
|
|
// It refuses ONLY the provable case (stacks.CatalogOrder's Ahead arm: every differing service
|
|
// orderable and newer). Anything unorderable, mixed or equal falls through to the behaviour that
|
|
// shipped in v0.237.0 — this gate can block an update, so it errs towards letting one run.
|
|
//
|
|
// COMPANION RED-PROOF (REPORT.md): make CatalogOrder's Ahead arm return Behind.
|
|
// TestR524_PreflightRefusesDowngrade then fails — the update is allowed to move the pin back.
|
|
if CatalogOrder(*st) == UpdateOrderAhead {
|
|
return m.refuseUpdateErr(name, "downgrade", util.MsgError("err.stacks.update_downgrade"),
|
|
fmt.Sprintf("installed is provably NEWER than the catalog on every differing service (installed=%v catalog=%v)",
|
|
st.AppConfig.InstalledImages, st.CatalogImages))
|
|
}
|
|
// R-475: any tier counts, and an app with no copy at all is backed up first by the job. So the only
|
|
// refusal left here is Scenario L — no copy on any tier AND no way to make one now. (With a copy
|
|
// but no way to back up, the job still applies the age rule and refuses then if the copy is stale.)
|
|
// Tier 3 is looked at only on this branch, so an ordinary update never waits on the network here.
|
|
//
|
|
// COMPANION RED-PROOF (REPORT.md): drop the `!found` condition. TestR475_L then fails — an app
|
|
// with a copy but no way to back up is refused too.
|
|
if canBackUp, why := g.CanBackUp(name); !canBackUp {
|
|
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
|
|
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
|
|
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
|
|
}
|
|
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
|
|
}
|
|
if ref := m.updateMemoryRefusal(name, st); ref != nil {
|
|
return ref
|
|
}
|
|
free, known := m.updateDiskFree()
|
|
switch {
|
|
case !known:
|
|
m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB)
|
|
case free < updateDiskFloorGiB:
|
|
return m.refuseUpdate(name, "disk", fmt.Sprintf(MsgUpdateDiskFmt, free, updateDiskFloorGiB),
|
|
fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// updateMemoryRefusal applies the deploy's memory check to the NEW template's request, releasing the
|
|
// app's CURRENT request first (an update replaces it). An unknown new request proceeds with a WARN.
|
|
func (m *Manager) updateMemoryRefusal(name string, st *Stack) *UpdateRefusal {
|
|
catPath := m.CatalogTemplatePath(name, ".felhom.yml")
|
|
if _, err := os.Stat(catPath); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update %s: the new template's memory request is unknown (%v) — proceeding without the memory check", name, err)
|
|
return nil
|
|
}
|
|
newMeta := LoadMetadata(filepath.Dir(catPath))
|
|
newReq, newLim := ParseMemoryMB(newMeta.Resources.MemRequest), ParseMemoryMB(newMeta.Resources.MemLimit)
|
|
if newReq == 0 {
|
|
m.logger.Printf("[WARN] [stacks] update %s: the new template declares no memory request — proceeding without the memory check", name)
|
|
return nil
|
|
}
|
|
oldReq, oldLim := ParseMemoryMB(st.Meta.Resources.MemRequest), ParseMemoryMB(st.Meta.Resources.MemLimit)
|
|
verdict := m.updateMemoryFn
|
|
if verdict == nil {
|
|
verdict = m.memoryVerdict
|
|
}
|
|
if refusal, _ := verdict(newReq, newLim, oldReq, oldLim); refusal != nil {
|
|
return m.refuseUpdateErr(name, "memory", refusal, fmt.Sprintf("new_req=%dMB replacing %dMB does not fit", newReq, oldReq))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func (m *Manager) updateDiskFree() (float64, bool) {
|
|
if m.updateDiskFreeFn != nil {
|
|
return m.updateDiskFreeFn()
|
|
}
|
|
du := system.GetDiskUsage(system.DockerVolumePath)
|
|
if du == nil {
|
|
return 0, false
|
|
}
|
|
return du.AvailGB, true
|
|
}
|
|
|
|
// StartGuardedUpdate re-checks the cheap refusals, claims the Updating flag atomically and launches the
|
|
// job. It returns as soon as the job has STARTED — the result arrives on GET /api/stacks/{name}.
|
|
func (m *Manager) StartGuardedUpdate(name string) error {
|
|
if ref := m.UpdatePreflight(name); ref != nil {
|
|
return ref
|
|
}
|
|
m.mu.Lock()
|
|
s, ok := m.stacks[name]
|
|
if !ok {
|
|
m.mu.Unlock()
|
|
return &UpdateRefusal{Reason: "not_found", Message: fmt.Sprintf("stack %q not found", name)}
|
|
}
|
|
// A second press between the preflight and here is the race this lock closes.
|
|
if s.Updating || s.Deploying {
|
|
m.mu.Unlock()
|
|
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
|
|
}
|
|
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
|
m.mu.Unlock()
|
|
|
|
m.logger.Printf("[INFO] [stacks] update %s: accepted — guarded update started", name)
|
|
go m.runGuardedUpdate(context.Background(), name)
|
|
return nil
|
|
}
|
|
|
|
// IsUpdating reports whether a guarded update is in progress for the app.
|
|
func (m *Manager) IsUpdating(name string) bool {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
s, ok := m.stacks[name]
|
|
return ok && s.Updating
|
|
}
|
|
|
|
// AnyUpdating reports whether a guarded update is in flight for ANY app.
|
|
//
|
|
// ── WHY IT EXISTS (v0.261.0) ────────────────────────────────────────────────────────────────────
|
|
//
|
|
// The CONTROLLER updates itself too — daily at `self_update.auto_update_time` (04:30 by default) and,
|
|
// once the hub serves a floor above this box, after any report, at any hour. That swap restarts the
|
|
// controller container. Until v0.261.0 its ONLY busy gate was `backupRunning`, so a self-update could
|
|
// land in the middle of a guarded app update.
|
|
//
|
|
// MEASURED, and the gap is narrower than it looks but real: the update's `backing-up` phase takes the
|
|
// backup single-flight (`RunAppBackupNow` → `acquireRunning`), so `backupMgr.IsRunning()` ALREADY
|
|
// covered that one phase. It covers none of the others — `checking`, `safety-dump`, `pinning`,
|
|
// `pulling`, `starting`, `verifying` — and `starting`/`verifying` are exactly where the new version
|
|
// may already have touched the app's data.
|
|
//
|
|
// ⚠ IT MUST ANSWER FALSE FOR A HELD APP. `Stack.Updating` is cleared by `finishUpdate` on done,
|
|
// failed AND held, so a held app does not hold this lock — otherwise one app that cannot come up
|
|
// would block the controller's own updates for ever, which is a worse failure than the one this
|
|
// prevents. TestR608_LockReleasesAfterHold pins that.
|
|
func (m *Manager) AnyUpdating() bool {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
for _, s := range m.stacks {
|
|
if s.Updating {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// UpdatingStacks is the set of apps an update is currently moving — for the dead-app alarm, which
|
|
// must not count an app the update itself is recreating (R-330's class, a third mechanism).
|
|
func (m *Manager) UpdatingStacks() map[string]bool {
|
|
m.mu.RLock()
|
|
defer m.mu.RUnlock()
|
|
var out map[string]bool
|
|
for name, s := range m.stacks {
|
|
if s.Updating {
|
|
if out == nil {
|
|
out = map[string]bool{}
|
|
}
|
|
out[name] = true
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
func (m *Manager) setUpdatePhase(name, phase string) {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure.
|
|
func (m *Manager) finishUpdate(name, phase, msg string) {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Updating = false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
|
s.UpdateError = msg
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) {
|
|
if m.updateComposeFn != nil {
|
|
return m.updateComposeFn(dir, env, args...)
|
|
}
|
|
return m.composeExecCustomEnv(dir, env, args...)
|
|
}
|
|
|
|
func (m *Manager) updateHealth(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
|
if m.updateHealthFn != nil {
|
|
return m.updateHealthFn(ctx, name, timeout)
|
|
}
|
|
return m.waitUpdateHealthy(ctx, name, timeout)
|
|
}
|
|
|
|
func (m *Manager) healthTimeout() time.Duration {
|
|
if m.cfg == nil {
|
|
return 5 * time.Minute
|
|
}
|
|
return m.cfg.Update.HealthTimeoutDuration()
|
|
}
|
|
|
|
func (m *Manager) backupMaxAge() time.Duration {
|
|
if m.cfg == nil {
|
|
return 24 * time.Hour
|
|
}
|
|
return m.cfg.Update.BackupMaxAgeDuration()
|
|
}
|
|
|
|
// pre-update copies, kept in the stack dir so they travel with it. Neither name is one the syncer
|
|
// copies (it copies exactly docker-compose.yml and .felhom.yml).
|
|
const (
|
|
preUpdateComposeFile = "pre-update-compose.yml"
|
|
preUpdateAppliedFile = "pre-update-applied.yml"
|
|
)
|
|
|
|
func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
|
start := m.now()
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
m.finishUpdate(name, UpdatePhaseFailed, fmt.Sprintf("stack %q not found", name))
|
|
return
|
|
}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
g := m.guards()
|
|
entry := updateJournalEntry{StartedAt: start}
|
|
fail := func(msg, detail string) {
|
|
m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail)
|
|
m.clearJournal(name)
|
|
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
|
}
|
|
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) {
|
|
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateJournalFailed)
|
|
return
|
|
}
|
|
if g == nil {
|
|
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
|
|
return
|
|
}
|
|
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
|
|
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
|
|
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
|
|
maxAge := m.backupMaxAge()
|
|
deployedAt := m.currentDeployTime(name)
|
|
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
|
|
if ok {
|
|
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
|
|
} else {
|
|
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
|
fail(MsgUpdateJournalFailed, "journal write failed")
|
|
return
|
|
}
|
|
if err := g.BackupNow(ctx, name); err != nil {
|
|
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
|
|
return
|
|
}
|
|
now := m.now()
|
|
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
|
|
if !ok {
|
|
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
|
|
}
|
|
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
|
|
entry.ProvenTier = rp.Tier
|
|
|
|
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
|
|
fail(MsgUpdateJournalFailed, "journal write failed")
|
|
return
|
|
}
|
|
paths, err := g.SafetyDump(ctx, name)
|
|
if err != nil {
|
|
fail(fmt.Sprintf(MsgUpdateDumpFailFmt, err), "safety dump: "+err.Error())
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
|
|
|
|
// v0.263.0 — the undo's copy is PLANNED here, before anything moves: its size against the disk
|
|
// floor (decision 19's limit). The copy itself is taken after the pull, where the app stops anyway.
|
|
undoVols, perr := m.planUndoCopies(name)
|
|
if perr != nil {
|
|
if se, ok := perr.(*undoSpaceError); ok {
|
|
fail(undoMsg("err.stacks.update_undo_space", se.need, se.free, updateDiskFloorGiB), "undo copy: "+perr.Error())
|
|
} else {
|
|
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy plan: "+perr.Error())
|
|
}
|
|
return
|
|
}
|
|
|
|
// PINNING — the previous definition is copied aside and journaled BEFORE the pin moves, so a crash
|
|
// at any later instant can put it back (Scenario G).
|
|
prevLive, err := os.ReadFile(st.ComposePath)
|
|
if err != nil {
|
|
fail(MsgUpdatePinFailed, "reading the live compose file: "+err.Error())
|
|
return
|
|
}
|
|
if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil {
|
|
fail(MsgUpdateJournalFailed, "saving the pre-update compose copy: "+err.Error())
|
|
return
|
|
}
|
|
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
|
|
// R-639: the OLD .felhom.yml too — its probe is the one the old version answers.
|
|
md, merr := savePreUpdateMeta(dir)
|
|
switch {
|
|
case md == "":
|
|
m.logger.Printf("[WARN] [stacks] update %s: could not keep the previous .felhom.yml (%v) — an undo would check health with the current one", name, merr)
|
|
case merr != nil:
|
|
m.logger.Printf("[WARN] [stacks] update %s: %v", name, merr)
|
|
entry.PrevMeta = md
|
|
default:
|
|
entry.PrevMeta = md
|
|
}
|
|
if applied, aerr := LoadAppliedDefinition(dir); aerr == nil {
|
|
if err := os.WriteFile(filepath.Join(dir, preUpdateAppliedFile), applied, 0o644); err == nil {
|
|
entry.PrevApplied = filepath.Join(dir, preUpdateAppliedFile)
|
|
}
|
|
}
|
|
if cfg := LoadAppConfig(dir); cfg != nil && len(cfg.PinnedImages) > 0 {
|
|
entry.PrevPin = map[string]string{}
|
|
for k, v := range cfg.PinnedImages {
|
|
entry.PrevPin[k] = v
|
|
}
|
|
}
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) {
|
|
m.removePreUpdateCopies(dir)
|
|
fail(MsgUpdateJournalFailed, "journal write failed")
|
|
return
|
|
}
|
|
if err := m.advancePinToCatalog(name, dir); err != nil {
|
|
m.pinBack(name, dir, entry)
|
|
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
|
|
return
|
|
}
|
|
if cfg := LoadAppConfig(dir); cfg != nil {
|
|
entry.NewPin = cfg.PinnedImages
|
|
}
|
|
|
|
env := m.stackEnv(dir)
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
|
|
m.pinBack(name, dir, entry)
|
|
fail(MsgUpdateJournalFailed, "journal write failed")
|
|
return
|
|
}
|
|
if _, err := m.updateCompose(dir, env, "pull"); err != nil {
|
|
// Scenario E: NOTHING RAN. The containers are the old ones and still running, so the honest
|
|
// state is the old pin and the old file — put both back.
|
|
m.pinBack(name, dir, entry)
|
|
m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err)
|
|
fail(MsgUpdatePullFailed, "pull failed: "+err.Error())
|
|
return
|
|
}
|
|
|
|
// COPYING (v0.263.0) — the app is stopped here anyway to be recreated, so the extra downtime is the
|
|
// copy alone. A failed copy moves nothing: the copies go, the pin goes back, the old containers
|
|
// start again.
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) {
|
|
m.pinBack(name, dir, entry)
|
|
fail(MsgUpdateJournalFailed, "journal write failed")
|
|
return
|
|
}
|
|
if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: the undo copy failed: %v — removing it, putting the pin back and starting the previous version", name, err)
|
|
m.removeUndoCopies(name, entry.UndoCopies)
|
|
m.pinBack(name, dir, entry)
|
|
if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr)
|
|
}
|
|
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy: "+err.Error())
|
|
return
|
|
}
|
|
|
|
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
|
|
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up", &entry)
|
|
return
|
|
}
|
|
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
|
// Containers may already have been recreated on the new image — something may have run.
|
|
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error(), &entry)
|
|
return
|
|
}
|
|
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
|
|
}
|
|
|
|
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
|
|
// known, and a failure holds the app.
|
|
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, start time.Time, entry *updateJournalEntry) {
|
|
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
|
|
}
|
|
timeout := m.healthTimeout()
|
|
waitStart := m.now()
|
|
healthy, detail := m.updateHealth(ctx, name, timeout)
|
|
if !healthy {
|
|
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail, entry)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
|
|
m.recordInstalledImages(name, dir, env)
|
|
_ = m.RefreshStatus()
|
|
m.removeUndoCopies(name, entry.UndoCopies)
|
|
m.recordUpdateUndone(name, dir, nil) // a successful update ends the "undone" note
|
|
m.clearJournal(name)
|
|
m.removePreUpdateCopies(dir)
|
|
m.finishUpdate(name, UpdatePhaseDone, "")
|
|
m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second))
|
|
}
|
|
|
|
// holdLogTailLines is how much of each service's log the hold keeps. 400 lines is enough to hold a
|
|
// migration run and a startup failure, and small enough that a hold never fills a disk.
|
|
const holdLogTailLines = "400"
|
|
|
|
// captureHoldLogs writes each service's log into <stackdir>/hold-logs/<ts>/ before the app is
|
|
// stopped. Best-effort by design (R-621): the hold itself must happen either way.
|
|
func (m *Manager) captureHoldLogs(name, dir string, env []string) {
|
|
ts := m.now().UTC().Format("20060102T150405Z")
|
|
outDir := filepath.Join(dir, "hold-logs", ts)
|
|
if err := os.MkdirAll(outDir, 0o755); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: cannot create the hold-log directory %s: %v — the hold still proceeds", name, outDir, err)
|
|
return
|
|
}
|
|
out, err := m.updateCompose(dir, env, "logs", "--no-color", "--tail", holdLogTailLines)
|
|
if err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update %s: `compose logs` before the hold failed: %v — writing what came back anyway", name, err)
|
|
}
|
|
path := filepath.Join(outDir, "compose-logs.txt")
|
|
if werr := os.WriteFile(path, []byte(out), 0o644); werr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: could not write %s: %v", name, path, werr)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: kept %d bytes of the app's own log at %s before stopping it (R-621)", name, len(out), path)
|
|
}
|
|
|
|
// failAndHold is Scenario F. Since v0.263.0 it first UNDOES (decision 15): when the job took its
|
|
// last-second copy, the old version goes back with the data from before the update, and the app is
|
|
// HELD only when that undo fails too. Without a copy (an update journaled by an older controller,
|
|
// resumed after an upgrade) it holds as it always did.
|
|
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string, entry *updateJournalEntry) {
|
|
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s", name, why)
|
|
// R-621: capture the app's own logs BEFORE the `down`, because the `down` destroys them. Two
|
|
// drill nights lost the only evidence of WHY an update failed this way — `adventurelog` ran nine
|
|
// migrations and then never bound its port, and the log that would have said so was gone by the
|
|
// time anyone looked. The capture is bounded and best-effort: a hold must never fail because its
|
|
// evidence could not be written.
|
|
m.captureHoldLogs(name, dir, env)
|
|
undoState := ""
|
|
if entry != nil && entry.Copied {
|
|
if undoState = m.tryUndo(ctx, name, dir, why, entry); undoState == "" {
|
|
return // undone: the previous version runs on the pre-update data
|
|
}
|
|
m.logger.Printf("[ERROR] [stacks] update %s: the UNDO failed too (%s) — HOLDING the app; the undo copies are kept: %v", name, undoState, entry.UndoCopies)
|
|
} else {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: no undo copy for this update — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name)
|
|
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
|
}
|
|
}
|
|
msg := MsgUpdateHoldUnsaved
|
|
if g := m.guards(); g == nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
|
|
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
|
} else if _, why := g.HoldFor(name); why != "" {
|
|
msg = why
|
|
m.markUpdateHeld(name)
|
|
}
|
|
_ = m.RefreshStatus()
|
|
m.clearJournal(name)
|
|
m.removePreUpdateCopies(dir)
|
|
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
|
}
|
|
|
|
// pinBack restores the pin, the stored definition and the live file from the journaled copies, and
|
|
// then removes the copies.
|
|
func (m *Manager) pinBack(name, dir string, entry updateJournalEntry) {
|
|
m.restoreDefinition(name, dir, entry)
|
|
m.removePreUpdateCopies(dir)
|
|
m.logger.Printf("[INFO] [stacks] update %s: pin and definition PUT BACK to the pre-update version (%s)", name, summarisePin(entry.PrevPin))
|
|
}
|
|
|
|
// restoreDefinition is pinBack WITHOUT removing the copies — the undo's form, so a power cut after it
|
|
// can run it again (RecoverUpdates → undoing) and find the copies still there. It also puts the pinned
|
|
// version's .felhom.yml record back (v0.263.2), so the NEXT update's undo probes the right version.
|
|
func (m *Manager) restoreDefinition(name, dir string, entry updateJournalEntry) {
|
|
if entry.PrevMeta != "" {
|
|
m.storeAppliedMetaFrom(name, dir, filepath.Join(entry.PrevMeta, ".felhom.yml"))
|
|
}
|
|
prevLive, lerr := os.ReadFile(entry.PrevCompose)
|
|
if lerr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: cannot read the pre-update compose copy (%v) — the definition could NOT be put back", name, lerr)
|
|
}
|
|
if len(entry.PrevPin) > 0 {
|
|
applied := prevLive
|
|
if entry.PrevApplied != "" {
|
|
if b, err := os.ReadFile(entry.PrevApplied); err == nil {
|
|
applied = b
|
|
}
|
|
}
|
|
if err := m.SetPin(name, dir, entry.PrevPin, applied); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: putting the pin back failed: %v", name, err)
|
|
}
|
|
}
|
|
if lerr == nil {
|
|
if err := os.WriteFile(ComposePathIn(dir), prevLive, 0o644); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: re-rendering the previous definition failed: %v", name, err)
|
|
}
|
|
}
|
|
}
|
|
|
|
func (m *Manager) removePreUpdateCopies(dir string) {
|
|
_ = os.Remove(filepath.Join(dir, preUpdateComposeFile))
|
|
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
|
|
_ = os.RemoveAll(filepath.Join(dir, preUpdateMetaDir))
|
|
}
|
|
|
|
// settleReason names WHY the wait fell back to container state, so the journal and the log do not
|
|
// have to be read together to tell "this app declares no check" from "this app declares one that
|
|
// resolves to no container" (R-630).
|
|
func settleReason(meta Metadata) string {
|
|
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
|
return "no probe container — settled on container state"
|
|
}
|
|
return "no health check declared"
|
|
}
|
|
|
|
// waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the
|
|
// existing probe, or — for an app with none — every container running and none restarting for
|
|
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
|
|
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
|
return m.waitUpdateHealthyMeta(ctx, name, timeout, nil)
|
|
}
|
|
|
|
// waitUpdateHealthyMeta is the health wait with the probe taken from `override` instead of the app's
|
|
// current .felhom.yml — the undo's form (the old version is judged by the old probe). nil = current.
|
|
func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeout time.Duration, override *Metadata) (bool, string) {
|
|
deadline := m.now().Add(timeout)
|
|
var runningSince time.Time
|
|
warnedNoProbe := false
|
|
last := "no observation yet"
|
|
for {
|
|
_ = m.RefreshStatus()
|
|
st, ok := m.GetStack(name)
|
|
// v0.263.0 — THE UNDO'S PROBE MUST BE ALLOWED TO RUN. The periodic health probe judges the app
|
|
// with the CURRENT .felhom.yml and flips a running app to StateUnhealthy when that check fails —
|
|
// which is exactly the undo's situation when the new version brought a probe the old one does
|
|
// not answer. Gating on StateRunning alone meant the old probe was never asked and the undo was
|
|
// reported "did not start" after the full timeout. MEASURED LIVE on 9202 2026-09-23 (docmost,
|
|
// `last: state unhealthy` for 90 s while the old version served). So with an override that
|
|
// declares checks, an Unhealthy app is PROBED — and the override's own check decides. It never
|
|
// settles an Unhealthy app on container state (below: the settle path requires StateRunning).
|
|
// Pinned by TestUndo_OldProbeRunsOnAnAppTheCurrentProbeMarkedUnhealthy.
|
|
undoProbe := ok && override != nil && st.State == StateUnhealthy &&
|
|
override.HealthCheck != nil && len(override.HealthCheck.Checks) > 0
|
|
switch {
|
|
case !ok:
|
|
last = "stack vanished"
|
|
runningSince = time.Time{}
|
|
case st.State == StateRunning || undoProbe:
|
|
// A declared health check is only usable if it resolves to a container. When it does
|
|
// not, the app is judged the same way an app with NO declared check is judged —
|
|
// settling on container state — and the log says so.
|
|
//
|
|
// R-630, and this `else` is the whole defect: the old code set `last = "no probe
|
|
// container"` and looped, so `verifying` spent the FULL `update.health_timeout` and
|
|
// `failAndHold` then STOPPED an app whose containers were all healthy. Measured on
|
|
// paperless-ngx 2026-09-22: `done` was never reachable, `failed` at +313.0 s, front door
|
|
// 404 afterwards. A stack with no probe is not "healthy" and it is not "failing" — it is
|
|
// SETTLED ON CONTAINER STATE (`09` §3), and never a reason to stop a running app.
|
|
usable := false
|
|
meta := st.Meta
|
|
if override != nil {
|
|
meta = *override
|
|
}
|
|
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
|
c, candidates := findProbeContainerMeta(name, &meta, st.Containers)
|
|
if c != "" {
|
|
usable = true
|
|
res := m.probeRun(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.HealthProbe = res
|
|
}
|
|
m.mu.Unlock()
|
|
if res.Healthy {
|
|
return true, "the app's health check passed"
|
|
}
|
|
last = "health check failing"
|
|
} else if !warnedNoProbe {
|
|
warnedNoProbe = true
|
|
m.logger.Printf("[WARN] [stacks] update %s: no probe container for %s — settling on container state instead; candidates: %v",
|
|
name, name, candidates)
|
|
}
|
|
}
|
|
if !usable && st.State != StateRunning {
|
|
last = "unhealthy, and the check resolves to no container"
|
|
} else if !usable {
|
|
if runningSince.IsZero() {
|
|
runningSince = m.now()
|
|
}
|
|
if m.now().Sub(runningSince) >= updateSettleWindow {
|
|
return true, fmt.Sprintf("all containers running, none restarting, for %s (%s)",
|
|
updateSettleWindow, settleReason(meta))
|
|
}
|
|
last = "running, settling"
|
|
}
|
|
default:
|
|
runningSince = time.Time{}
|
|
last = "state " + string(st.State)
|
|
}
|
|
if !m.now().Before(deadline) {
|
|
return false, fmt.Sprintf("not healthy within %s (last: %s)", timeout, last)
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
return false, "cancelled: " + ctx.Err().Error()
|
|
case <-time.After(updatePollEvery):
|
|
}
|
|
}
|
|
}
|
|
|
|
// ── the journal ──────────────────────────────────────────────────────────────────────────────────
|
|
|
|
type updateJournalEntry struct {
|
|
Phase string `json:"phase"`
|
|
StartedAt time.Time `json:"started_at"`
|
|
PrevPin map[string]string `json:"prev_pin,omitempty"`
|
|
PrevCompose string `json:"prev_compose,omitempty"`
|
|
PrevApplied string `json:"prev_applied,omitempty"`
|
|
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
|
|
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
|
|
// the right copy. 0 in a journal written by v0.238.1 or older.
|
|
ProvenTier int `json:"proven_tier,omitempty"`
|
|
// v0.263.0 — the undo (undo.go). UndoCopies is journaled BEFORE each copy starts; Copied is true
|
|
// only once every copy finished (it may be true with zero copies: an app with no named volume).
|
|
// PrevMeta is the directory holding the previous .felhom.yml; NewPin the pin the update moved to.
|
|
UndoCopies []undoCopy `json:"undo_copies,omitempty"`
|
|
Copied bool `json:"copied,omitempty"`
|
|
PrevMeta string `json:"prev_meta,omitempty"`
|
|
NewPin map[string]string `json:"new_pin,omitempty"`
|
|
}
|
|
|
|
type updateJournal struct {
|
|
Updates map[string]updateJournalEntry `json:"updates"`
|
|
}
|
|
|
|
func (m *Manager) updateJournalPath() string {
|
|
return filepath.Join(m.cfg.Paths.DataDir, "update-journal.json")
|
|
}
|
|
|
|
func (m *Manager) readUpdateJournal() updateJournal {
|
|
j := updateJournal{Updates: map[string]updateJournalEntry{}}
|
|
data, err := os.ReadFile(m.updateJournalPath())
|
|
if err != nil {
|
|
return j
|
|
}
|
|
if err := json.Unmarshal(data, &j); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] update journal at %s is corrupt (%v) — quarantining", m.updateJournalPath(), err)
|
|
_ = os.Rename(m.updateJournalPath(), fmt.Sprintf("%s.corrupt-%d", m.updateJournalPath(), time.Now().Unix()))
|
|
return updateJournal{Updates: map[string]updateJournalEntry{}}
|
|
}
|
|
if j.Updates == nil {
|
|
j.Updates = map[string]updateJournalEntry{}
|
|
}
|
|
return j
|
|
}
|
|
|
|
// writeUpdateJournal is atomic and fsynced (the AppStopGuard shape): the point is surviving a power cut.
|
|
func (m *Manager) writeUpdateJournal(j updateJournal) error {
|
|
p := m.updateJournalPath()
|
|
if len(j.Updates) == 0 {
|
|
if err := os.Remove(p); err != nil && !os.IsNotExist(err) {
|
|
return err
|
|
}
|
|
return nil
|
|
}
|
|
data, err := json.MarshalIndent(j, "", " ")
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
|
|
return err
|
|
}
|
|
tmp := p + ".tmp"
|
|
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if _, err := f.Write(data); err != nil {
|
|
f.Close()
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if err := f.Sync(); err != nil {
|
|
f.Close()
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if err := f.Close(); err != nil {
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
return os.Rename(tmp, p)
|
|
}
|
|
|
|
// enterUpdatePhase journals the phase BEFORE it starts and mirrors it for the UI. False means the
|
|
// journal could not be written — the caller must not perform a mutation it could not record.
|
|
func (m *Manager) enterUpdatePhase(name string, entry *updateJournalEntry, phase string) bool {
|
|
entry.Phase = phase
|
|
m.updateJournalMu.Lock()
|
|
j := m.readUpdateJournal()
|
|
j.Updates[name] = *entry
|
|
err := m.writeUpdateJournal(j)
|
|
m.updateJournalMu.Unlock()
|
|
m.setUpdatePhase(name, phase)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: journal write for phase %s failed: %v", name, phase, err)
|
|
return false
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: phase %s", name, phase)
|
|
return true
|
|
}
|
|
|
|
func (m *Manager) clearJournal(name string) {
|
|
m.updateJournalMu.Lock()
|
|
defer m.updateJournalMu.Unlock()
|
|
j := m.readUpdateJournal()
|
|
delete(j.Updates, name)
|
|
if err := m.writeUpdateJournal(j); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update %s: clearing the journal entry failed: %v", name, err)
|
|
}
|
|
}
|
|
|
|
// RecoverUpdates reads the journal at startup (Scenario G). Call it BEFORE the boot reconciler.
|
|
//
|
|
// - interrupted BEFORE the pin moved (checking, backing-up, safety-dump): nothing moved — the entry is
|
|
// dropped and the app carries the "interrupted" sentence;
|
|
// - interrupted while pinning or pulling: nothing RAN — the pin and definition are put back (as E);
|
|
// - interrupted while starting or verifying: something may have run — the app is marked Updating
|
|
// (so the boot sweep and the dead-app alarm leave it alone) and queued for ResumeInterruptedUpdates,
|
|
// which re-runs `up -d` and the health wait, ending in A or F.
|
|
//
|
|
// It needs no backup wiring, because nothing here holds an app — that is left to the resumed job.
|
|
func (m *Manager) RecoverUpdates() []string {
|
|
m.updateJournalMu.Lock()
|
|
j := m.readUpdateJournal()
|
|
m.updateJournalMu.Unlock()
|
|
if len(j.Updates) == 0 {
|
|
return nil
|
|
}
|
|
var resumed []string
|
|
for name, e := range j.Updates {
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s is in the journal (phase %s) but no longer exists — dropping the entry", name, e.Phase)
|
|
m.clearJournal(name)
|
|
continue
|
|
}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
switch e.Phase {
|
|
case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
|
m.clearJournal(name)
|
|
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
|
case UpdatePhasePinning, UpdatePhasePulling:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
|
m.pinBack(name, dir, e)
|
|
m.clearJournal(name)
|
|
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
|
case UpdatePhaseCopying:
|
|
// v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the
|
|
// pin goes back, and the previous version is started again.
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while copying its data (started %s) — nothing new ran; removing the partial copy, putting the pin back and starting the previous version", name, e.StartedAt.Format(time.RFC3339))
|
|
m.removeUndoCopies(name, e.UndoCopies)
|
|
m.pinBack(name, dir, e)
|
|
if _, err := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err)
|
|
}
|
|
m.clearJournal(name)
|
|
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
|
case UpdatePhaseUndoing:
|
|
// v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from
|
|
// the copies (still there: they are removed only after the undo succeeded) and then probes.
|
|
// Never "done": what ran last was a failed new version.
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted while UNDOING (started %s) — marking it Updating and RESUMING the undo", name, e.StartedAt.Format(time.RFC3339))
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseUndoing, UpdatePhaseLabel(UpdatePhaseUndoing)
|
|
}
|
|
m.updateResume = append(m.updateResume, name)
|
|
m.mu.Unlock()
|
|
resumed = append(resumed, name)
|
|
case UpdatePhaseStarting, UpdatePhaseVerifying:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
|
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
|
|
}
|
|
m.updateResume = append(m.updateResume, name)
|
|
m.mu.Unlock()
|
|
resumed = append(resumed, name)
|
|
default:
|
|
m.logger.Printf("[WARN] [stacks] update recovery: %s has unknown phase %q — dropping the entry", name, e.Phase)
|
|
m.clearJournal(name)
|
|
}
|
|
}
|
|
return resumed
|
|
}
|
|
|
|
// ResumeInterruptedUpdates continues the updates RecoverUpdates queued, once the backup side is wired
|
|
// (a resumed update that fails must be able to HOLD). Returns how many were resumed.
|
|
func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
|
m.mu.Lock()
|
|
names := m.updateResume
|
|
m.updateResume = nil
|
|
m.mu.Unlock()
|
|
for _, name := range names {
|
|
st, ok := m.GetStack(name)
|
|
if !ok {
|
|
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
|
continue
|
|
}
|
|
m.updateJournalMu.Lock()
|
|
e, ok := m.readUpdateJournal().Updates[name]
|
|
m.updateJournalMu.Unlock()
|
|
if !ok {
|
|
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
|
continue
|
|
}
|
|
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
|
|
rp := UpdateRestorePoint{Tier: e.ProvenTier, ProvenAt: provenAt}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
|
|
env := m.stackEnv(dir)
|
|
if e.Phase == UpdatePhaseUndoing {
|
|
m.logger.Printf("[INFO] [stacks] update %s: resuming the UNDO after a controller restart", name)
|
|
m.failAndHold(ctx, name, dir, env, rp, "resumed after a restart during the undo", &e)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
|
|
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
|
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error(), &e)
|
|
return
|
|
}
|
|
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
|
|
}(name, dir, e, rp)
|
|
}
|
|
return len(names)
|
|
}
|
|
|
|
// probeRun is the network half of the update's health wait — its own seam, so a test can drive the
|
|
// REAL wait loop (the state gate, the container resolution, the settle rule) with only the HTTP/TCP
|
|
// probe faked. nil ⇒ runChecks.
|
|
func (m *Manager) probeRun(t probeTarget) *HealthProbeResult {
|
|
if m.probeRunFn != nil {
|
|
return m.probeRunFn(t)
|
|
}
|
|
return m.runChecks(t)
|
|
}
|