controller v0.264.0: the household is told when an update is undone or held, in its language
gates / gates (push) Successful in 25s

app_update_undone / app_update_held events (09 decision 15), on by
default and seeded once on existing boxes; R-606 update sentences as
key+args rendered per reader; R-646 startup applied-meta backfill for
apps current with the catalog; R-620 a disabled notifier WARNs once per
event type. Needs hub v0.120.0.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-23 13:51:21 +02:00
parent c3a2aba0d2
commit bc278944a3
27 changed files with 1086 additions and 86 deletions
+78 -39
View File
@@ -9,6 +9,7 @@ import (
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
)
@@ -352,26 +353,26 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack")
}
if !st.Deployed {
return m.refuseUpdate(name, "not_deployed", MsgUpdateNotDeployed, "not deployed")
return m.refuseUpdateErr(name, "not_deployed", util.MsgError("update.refusal.not_deployed"), "not deployed")
}
g := m.guards()
if g == nil {
return m.refuseUpdate(name, "guards_unwired", MsgUpdateNoGuards, "no UpdateGuards wired — fail closed")
return m.refuseUpdateErr(name, "guards_unwired", util.MsgError("update.error.no_guards"), "no UpdateGuards wired — fail closed")
}
if st.Deploying {
return m.refuseUpdate(name, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, name), "a deploy is in progress")
return m.refuseUpdateErr(name, "deploying", util.MsgError("update.refusal.deploying", name), "a deploy is in progress")
}
if st.Updating {
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "an update is already in progress")
return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "an update is already in progress")
}
if held, why := g.HoldFor(name); held {
return m.refuseUpdate(name, "held", why, "the app is held")
}
if busy, why := g.Busy(name); busy {
return m.refuseUpdate(name, "busy", MsgUpdateBusy, why)
return m.refuseUpdateErr(name, "busy", util.MsgError("update.refusal.busy"), why)
}
if m.IsMigrating() {
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
return m.refuseUpdateErr(name, "migrating", util.MsgError("update.refusal.migrating"), "a data migration is running")
}
// v0.261.0 — the other half of the self-update lock. The controller's swap restarts this process;
// starting an app update into that is how an update loses its own supervisor mid-flight. TRANSIENT:
@@ -408,7 +409,7 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
// with a copy but no way to back up is refused too.
if canBackUp, why := g.CanBackUp(name); !canBackUp {
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
return m.refuseUpdateErr(name, "no_backup", util.MsgError("update.refusal.no_backup", name),
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
}
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
@@ -421,7 +422,7 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
case !known:
m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB)
case free < updateDiskFloorGiB:
return m.refuseUpdate(name, "disk", fmt.Sprintf(MsgUpdateDiskFmt, free, updateDiskFloorGiB),
return m.refuseUpdateErr(name, "disk", util.MsgError("update.refusal.disk", free, updateDiskFloorGiB),
fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB))
}
return nil
@@ -478,7 +479,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
// A second press between the preflight and here is the race this lock closes.
if s.Updating || s.Deploying {
m.mu.Unlock()
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "lost the race for the Updating flag")
}
s.Updating, s.UpdateError, s.updateHeld = true, "", false
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
@@ -552,17 +553,49 @@ func (m *Manager) setUpdatePhase(name, phase string) {
m.mu.Unlock()
}
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure.
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure — a
// finished one (the hold's own sentence, or none). A sentence the job owns goes through finishUpdateKey.
func (m *Manager) finishUpdate(name, phase, msg string) {
m.finishUpdateKey(name, phase, "", msg)
}
// finishUpdateKey is finishUpdate with the sentence as a bundle KEY (v0.264.0, R-606): UpdateError
// keeps the Hungarian (byte-identical to the MsgUpdate* literal it replaced), and the key + args ride
// beside it for the page. key "" = `plain` is a finished sentence and is stored as it is.
func (m *Manager) finishUpdateKey(name, phase, key, plain string, args ...interface{}) {
msg := plain
if key != "" {
msg = util.Text(i18n.Default, key, args...)
}
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Updating = false
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
s.UpdateError = msg
s.UpdateError, s.UpdateErrorKey, s.UpdateErrorArgs = msg, key, args
}
m.mu.Unlock()
}
// UpdatePhaseLabelIn is a phase's label in lang (v0.264.0, R-606). Hungarian is the updatePhaseLabels
// map itself; another language reads `update.phase.<phase>` and falls back to the Hungarian.
func UpdatePhaseLabelIn(lang, phase string) string {
if lang == i18n.Default || phase == "" {
return UpdatePhaseLabel(phase)
}
if b, err := i18n.Shared(); err == nil && b.Has(lang, "update.phase."+phase) {
return b.Msg(lang, "update.phase."+phase)
}
return UpdatePhaseLabel(phase)
}
// UpdateErrorIn is the stack's update sentence in lang (v0.264.0, R-606).
func (s Stack) UpdateErrorIn(lang string) string {
if s.UpdateErrorKey == "" || lang == i18n.Default {
return s.UpdateError
}
return util.Text(lang, s.UpdateErrorKey, s.UpdateErrorArgs...)
}
func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) {
if m.updateComposeFn != nil {
return m.updateComposeFn(dir, env, args...)
@@ -608,18 +641,18 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
dir := filepath.Dir(st.ComposePath)
g := m.guards()
entry := updateJournalEntry{StartedAt: start}
fail := func(msg, detail string) {
fail := func(key, detail string, args ...interface{}) {
m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail)
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, msg)
m.finishUpdateKey(name, UpdatePhaseFailed, key, "", args...)
}
if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) {
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateJournalFailed)
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.journal_failed", "")
return
}
if g == nil {
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
fail("update.error.no_guards", "no UpdateGuards wired")
return
}
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
@@ -633,17 +666,17 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
} else {
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
fail(MsgUpdateJournalFailed, "journal write failed")
fail("update.error.journal_failed", "journal write failed")
return
}
if err := g.BackupNow(ctx, name); err != nil {
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
fail("update.error.backup_failed", "pre-update backup: "+err.Error(), err.Error())
return
}
now := m.now()
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
if !ok {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
fail("update.error.backup_no_unit", fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
return
}
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
@@ -653,12 +686,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
fail(MsgUpdateJournalFailed, "journal write failed")
fail("update.error.journal_failed", "journal write failed")
return
}
paths, err := g.SafetyDump(ctx, name)
if err != nil {
fail(fmt.Sprintf(MsgUpdateDumpFailFmt, err), "safety dump: "+err.Error())
fail("update.error.dump_failed", "safety dump: "+err.Error(), err.Error())
return
}
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
@@ -668,9 +701,9 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
undoVols, perr := m.planUndoCopies(name)
if perr != nil {
if se, ok := perr.(*undoSpaceError); ok {
fail(undoMsg("err.stacks.update_undo_space", se.need, se.free, updateDiskFloorGiB), "undo copy: "+perr.Error())
fail("err.stacks.update_undo_space", "undo copy: "+perr.Error(), se.need, se.free, updateDiskFloorGiB)
} else {
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy plan: "+perr.Error())
fail("err.stacks.update_undo_copy_failed", "undo copy plan: "+perr.Error())
}
return
}
@@ -679,11 +712,11 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
// at any later instant can put it back (Scenario G).
prevLive, err := os.ReadFile(st.ComposePath)
if err != nil {
fail(MsgUpdatePinFailed, "reading the live compose file: "+err.Error())
fail("update.error.pin_failed", "reading the live compose file: "+err.Error())
return
}
if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil {
fail(MsgUpdateJournalFailed, "saving the pre-update compose copy: "+err.Error())
fail("update.error.journal_failed", "saving the pre-update compose copy: "+err.Error())
return
}
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
@@ -711,12 +744,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
}
if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) {
m.removePreUpdateCopies(dir)
fail(MsgUpdateJournalFailed, "journal write failed")
fail("update.error.journal_failed", "journal write failed")
return
}
if err := m.advancePinToCatalog(name, dir); err != nil {
m.pinBack(name, dir, entry)
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
fail("update.error.pin_failed", "advancing the pin: "+err.Error())
return
}
if cfg := LoadAppConfig(dir); cfg != nil {
@@ -726,7 +759,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
env := m.stackEnv(dir)
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
m.pinBack(name, dir, entry)
fail(MsgUpdateJournalFailed, "journal write failed")
fail("update.error.journal_failed", "journal write failed")
return
}
if _, err := m.updateCompose(dir, env, "pull"); err != nil {
@@ -734,7 +767,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
// state is the old pin and the old file — put both back.
m.pinBack(name, dir, entry)
m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err)
fail(MsgUpdatePullFailed, "pull failed: "+err.Error())
fail("update.error.pull_failed", "pull failed: "+err.Error())
return
}
@@ -743,7 +776,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
// start again.
if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) {
m.pinBack(name, dir, entry)
fail(MsgUpdateJournalFailed, "journal write failed")
fail("update.error.journal_failed", "journal write failed")
return
}
if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil {
@@ -753,7 +786,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil {
m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr)
}
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy: "+err.Error())
fail("err.stacks.update_undo_copy_failed", "undo copy: "+err.Error())
return
}
@@ -842,19 +875,25 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
}
}
msg := MsgUpdateHoldUnsaved
holdWhy := ""
if g := m.guards(); g == nil {
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil {
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
} else if _, w := g.HoldFor(name); w != "" {
holdWhy = w
m.markUpdateHeld(name)
}
_ = m.RefreshStatus()
m.clearJournal(name)
m.removePreUpdateCopies(dir)
m.finishUpdate(name, UpdatePhaseFailed, msg)
if holdWhy == "" {
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.hold_unsaved", "")
} else {
// The hold's own sentence (the page renders it per reader through RestoreHoldForLang).
m.finishUpdate(name, UpdatePhaseFailed, holdWhy)
}
m.emitUpdateEvent(UpdateEventHeld, name, entry, rp, holdWhy != "")
}
// pinBack restores the pin, the stored definition and the live file from the journaled copies, and
@@ -1147,12 +1186,12 @@ func (m *Manager) RecoverUpdates() []string {
case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump:
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
case UpdatePhasePinning, UpdatePhasePulling:
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.pinBack(name, dir, e)
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
case UpdatePhaseCopying:
// v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the
// pin goes back, and the previous version is started again.
@@ -1163,7 +1202,7 @@ func (m *Manager) RecoverUpdates() []string {
m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err)
}
m.clearJournal(name)
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
case UpdatePhaseUndoing:
// v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from
// the copies (still there: they are removed only after the undo succeeded) and then probes.
@@ -1205,14 +1244,14 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
for _, name := range names {
st, ok := m.GetStack(name)
if !ok {
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
continue
}
m.updateJournalMu.Lock()
e, ok := m.readUpdateJournal().Updates[name]
m.updateJournalMu.Unlock()
if !ok {
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
continue
}
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)