controller v0.264.0: the household is told when an update is undone or held, in its language
gates / gates (push) Successful in 25s
gates / gates (push) Successful in 25s
app_update_undone / app_update_held events (09 decision 15), on by default and seeded once on existing boxes; R-606 update sentences as key+args rendered per reader; R-646 startup applied-meta backfill for apps current with the catalog; R-620 a disabled notifier WARNs once per event type. Needs hub v0.120.0. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -9,6 +9,7 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
)
|
||||
@@ -352,26 +353,26 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
||||
return m.refuseUpdate(name, "not_found", fmt.Sprintf("stack %q not found", name), "no such stack")
|
||||
}
|
||||
if !st.Deployed {
|
||||
return m.refuseUpdate(name, "not_deployed", MsgUpdateNotDeployed, "not deployed")
|
||||
return m.refuseUpdateErr(name, "not_deployed", util.MsgError("update.refusal.not_deployed"), "not deployed")
|
||||
}
|
||||
g := m.guards()
|
||||
if g == nil {
|
||||
return m.refuseUpdate(name, "guards_unwired", MsgUpdateNoGuards, "no UpdateGuards wired — fail closed")
|
||||
return m.refuseUpdateErr(name, "guards_unwired", util.MsgError("update.error.no_guards"), "no UpdateGuards wired — fail closed")
|
||||
}
|
||||
if st.Deploying {
|
||||
return m.refuseUpdate(name, "deploying", fmt.Sprintf(MsgUpdateDeployingFmt, name), "a deploy is in progress")
|
||||
return m.refuseUpdateErr(name, "deploying", util.MsgError("update.refusal.deploying", name), "a deploy is in progress")
|
||||
}
|
||||
if st.Updating {
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "an update is already in progress")
|
||||
return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "an update is already in progress")
|
||||
}
|
||||
if held, why := g.HoldFor(name); held {
|
||||
return m.refuseUpdate(name, "held", why, "the app is held")
|
||||
}
|
||||
if busy, why := g.Busy(name); busy {
|
||||
return m.refuseUpdate(name, "busy", MsgUpdateBusy, why)
|
||||
return m.refuseUpdateErr(name, "busy", util.MsgError("update.refusal.busy"), why)
|
||||
}
|
||||
if m.IsMigrating() {
|
||||
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
|
||||
return m.refuseUpdateErr(name, "migrating", util.MsgError("update.refusal.migrating"), "a data migration is running")
|
||||
}
|
||||
// v0.261.0 — the other half of the self-update lock. The controller's swap restarts this process;
|
||||
// starting an app update into that is how an update loses its own supervisor mid-flight. TRANSIENT:
|
||||
@@ -408,7 +409,7 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
||||
// with a copy but no way to back up is refused too.
|
||||
if canBackUp, why := g.CanBackUp(name); !canBackUp {
|
||||
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
|
||||
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
|
||||
return m.refuseUpdateErr(name, "no_backup", util.MsgError("update.refusal.no_backup", name),
|
||||
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
|
||||
}
|
||||
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
|
||||
@@ -421,7 +422,7 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
||||
case !known:
|
||||
m.logger.Printf("[WARN] [stacks] update %s: free space on the Docker data root is unreadable — proceeding without the %.0f GB floor", name, updateDiskFloorGiB)
|
||||
case free < updateDiskFloorGiB:
|
||||
return m.refuseUpdate(name, "disk", fmt.Sprintf(MsgUpdateDiskFmt, free, updateDiskFloorGiB),
|
||||
return m.refuseUpdateErr(name, "disk", util.MsgError("update.refusal.disk", free, updateDiskFloorGiB),
|
||||
fmt.Sprintf("%.2f GiB free on the Docker data root, floor %.0f GiB (fixed floor — image size unknown)", free, updateDiskFloorGiB))
|
||||
}
|
||||
return nil
|
||||
@@ -478,7 +479,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
|
||||
// A second press between the preflight and here is the race this lock closes.
|
||||
if s.Updating || s.Deploying {
|
||||
m.mu.Unlock()
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
|
||||
return m.refuseUpdateErr(name, "updating", util.MsgError("update.refusal.already", name), "lost the race for the Updating flag")
|
||||
}
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
||||
@@ -552,17 +553,49 @@ func (m *Manager) setUpdatePhase(name, phase string) {
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure.
|
||||
// finishUpdate is the ONE place Updating goes false. msg is the customer sentence on failure — a
|
||||
// finished one (the hold's own sentence, or none). A sentence the job owns goes through finishUpdateKey.
|
||||
func (m *Manager) finishUpdate(name, phase, msg string) {
|
||||
m.finishUpdateKey(name, phase, "", msg)
|
||||
}
|
||||
|
||||
// finishUpdateKey is finishUpdate with the sentence as a bundle KEY (v0.264.0, R-606): UpdateError
|
||||
// keeps the Hungarian (byte-identical to the MsgUpdate* literal it replaced), and the key + args ride
|
||||
// beside it for the page. key "" = `plain` is a finished sentence and is stored as it is.
|
||||
func (m *Manager) finishUpdateKey(name, phase, key, plain string, args ...interface{}) {
|
||||
msg := plain
|
||||
if key != "" {
|
||||
msg = util.Text(i18n.Default, key, args...)
|
||||
}
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating = false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = phase, UpdatePhaseLabel(phase)
|
||||
s.UpdateError = msg
|
||||
s.UpdateError, s.UpdateErrorKey, s.UpdateErrorArgs = msg, key, args
|
||||
}
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
// UpdatePhaseLabelIn is a phase's label in lang (v0.264.0, R-606). Hungarian is the updatePhaseLabels
|
||||
// map itself; another language reads `update.phase.<phase>` and falls back to the Hungarian.
|
||||
func UpdatePhaseLabelIn(lang, phase string) string {
|
||||
if lang == i18n.Default || phase == "" {
|
||||
return UpdatePhaseLabel(phase)
|
||||
}
|
||||
if b, err := i18n.Shared(); err == nil && b.Has(lang, "update.phase."+phase) {
|
||||
return b.Msg(lang, "update.phase."+phase)
|
||||
}
|
||||
return UpdatePhaseLabel(phase)
|
||||
}
|
||||
|
||||
// UpdateErrorIn is the stack's update sentence in lang (v0.264.0, R-606).
|
||||
func (s Stack) UpdateErrorIn(lang string) string {
|
||||
if s.UpdateErrorKey == "" || lang == i18n.Default {
|
||||
return s.UpdateError
|
||||
}
|
||||
return util.Text(lang, s.UpdateErrorKey, s.UpdateErrorArgs...)
|
||||
}
|
||||
|
||||
func (m *Manager) updateCompose(dir string, env []string, args ...string) (string, error) {
|
||||
if m.updateComposeFn != nil {
|
||||
return m.updateComposeFn(dir, env, args...)
|
||||
@@ -608,18 +641,18 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
g := m.guards()
|
||||
entry := updateJournalEntry{StartedAt: start}
|
||||
fail := func(msg, detail string) {
|
||||
fail := func(key, detail string, args ...interface{}) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED in phase %s after %s — nothing was moved: %s", name, entry.Phase, m.now().Sub(start).Round(time.Millisecond), detail)
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, key, "", args...)
|
||||
}
|
||||
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseChecking) {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateJournalFailed)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.journal_failed", "")
|
||||
return
|
||||
}
|
||||
if g == nil {
|
||||
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
|
||||
fail("update.error.no_guards", "no UpdateGuards wired")
|
||||
return
|
||||
}
|
||||
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
|
||||
@@ -633,17 +666,17 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
fail("update.error.journal_failed", "journal write failed")
|
||||
return
|
||||
}
|
||||
if err := g.BackupNow(ctx, name); err != nil {
|
||||
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
|
||||
fail("update.error.backup_failed", "pre-update backup: "+err.Error(), err.Error())
|
||||
return
|
||||
}
|
||||
now := m.now()
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
|
||||
if !ok {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
||||
fail("update.error.backup_no_unit", fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
|
||||
@@ -653,12 +686,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
|
||||
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
fail("update.error.journal_failed", "journal write failed")
|
||||
return
|
||||
}
|
||||
paths, err := g.SafetyDump(ctx, name)
|
||||
if err != nil {
|
||||
fail(fmt.Sprintf(MsgUpdateDumpFailFmt, err), "safety dump: "+err.Error())
|
||||
fail("update.error.dump_failed", "safety dump: "+err.Error(), err.Error())
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: safety dump done (%d file(s)) %v", name, len(paths), paths)
|
||||
@@ -668,9 +701,9 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
undoVols, perr := m.planUndoCopies(name)
|
||||
if perr != nil {
|
||||
if se, ok := perr.(*undoSpaceError); ok {
|
||||
fail(undoMsg("err.stacks.update_undo_space", se.need, se.free, updateDiskFloorGiB), "undo copy: "+perr.Error())
|
||||
fail("err.stacks.update_undo_space", "undo copy: "+perr.Error(), se.need, se.free, updateDiskFloorGiB)
|
||||
} else {
|
||||
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy plan: "+perr.Error())
|
||||
fail("err.stacks.update_undo_copy_failed", "undo copy plan: "+perr.Error())
|
||||
}
|
||||
return
|
||||
}
|
||||
@@ -679,11 +712,11 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
// at any later instant can put it back (Scenario G).
|
||||
prevLive, err := os.ReadFile(st.ComposePath)
|
||||
if err != nil {
|
||||
fail(MsgUpdatePinFailed, "reading the live compose file: "+err.Error())
|
||||
fail("update.error.pin_failed", "reading the live compose file: "+err.Error())
|
||||
return
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, preUpdateComposeFile), prevLive, 0o644); err != nil {
|
||||
fail(MsgUpdateJournalFailed, "saving the pre-update compose copy: "+err.Error())
|
||||
fail("update.error.journal_failed", "saving the pre-update compose copy: "+err.Error())
|
||||
return
|
||||
}
|
||||
entry.PrevCompose = filepath.Join(dir, preUpdateComposeFile)
|
||||
@@ -711,12 +744,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
}
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhasePinning) {
|
||||
m.removePreUpdateCopies(dir)
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
fail("update.error.journal_failed", "journal write failed")
|
||||
return
|
||||
}
|
||||
if err := m.advancePinToCatalog(name, dir); err != nil {
|
||||
m.pinBack(name, dir, entry)
|
||||
fail(MsgUpdatePinFailed, "advancing the pin: "+err.Error())
|
||||
fail("update.error.pin_failed", "advancing the pin: "+err.Error())
|
||||
return
|
||||
}
|
||||
if cfg := LoadAppConfig(dir); cfg != nil {
|
||||
@@ -726,7 +759,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
env := m.stackEnv(dir)
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhasePulling) {
|
||||
m.pinBack(name, dir, entry)
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
fail("update.error.journal_failed", "journal write failed")
|
||||
return
|
||||
}
|
||||
if _, err := m.updateCompose(dir, env, "pull"); err != nil {
|
||||
@@ -734,7 +767,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
// state is the old pin and the old file — put both back.
|
||||
m.pinBack(name, dir, entry)
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: pull failed — pin and definition PUT BACK; the app was not touched. Docker said: %v", name, err)
|
||||
fail(MsgUpdatePullFailed, "pull failed: "+err.Error())
|
||||
fail("update.error.pull_failed", "pull failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
@@ -743,7 +776,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
// start again.
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseCopying) {
|
||||
m.pinBack(name, dir, entry)
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
fail("update.error.journal_failed", "journal write failed")
|
||||
return
|
||||
}
|
||||
if err := m.makeUndoCopies(name, dir, env, undoVols, &entry); err != nil {
|
||||
@@ -753,7 +786,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
if _, uerr := m.updateCompose(dir, m.stackEnv(dir), "up", "-d", "--remove-orphans"); uerr != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: restarting the previous version after the failed copy also failed: %v", name, uerr)
|
||||
}
|
||||
fail(undoMsg("err.stacks.update_undo_copy_failed"), "undo copy: "+err.Error())
|
||||
fail("err.stacks.update_undo_copy_failed", "undo copy: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
@@ -842,19 +875,25 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
msg := MsgUpdateHoldUnsaved
|
||||
holdWhy := ""
|
||||
if g := m.guards(); g == nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
|
||||
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp, undoState); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
} else if _, w := g.HoldFor(name); w != "" {
|
||||
holdWhy = w
|
||||
m.markUpdateHeld(name)
|
||||
}
|
||||
_ = m.RefreshStatus()
|
||||
m.clearJournal(name)
|
||||
m.removePreUpdateCopies(dir)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, msg)
|
||||
if holdWhy == "" {
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.hold_unsaved", "")
|
||||
} else {
|
||||
// The hold's own sentence (the page renders it per reader through RestoreHoldForLang).
|
||||
m.finishUpdate(name, UpdatePhaseFailed, holdWhy)
|
||||
}
|
||||
m.emitUpdateEvent(UpdateEventHeld, name, entry, rp, holdWhy != "")
|
||||
}
|
||||
|
||||
// pinBack restores the pin, the stored definition and the live file from the journaled copies, and
|
||||
@@ -1147,12 +1186,12 @@ func (m *Manager) RecoverUpdates() []string {
|
||||
case UpdatePhaseChecking, UpdatePhaseBackingUp, UpdatePhaseSafetyDump:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had moved; dropping it", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
||||
case UpdatePhasePinning, UpdatePhasePulling:
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — nothing had run; putting the pin back", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.pinBack(name, dir, e)
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
||||
case UpdatePhaseCopying:
|
||||
// v0.263.0: the app was STOPPED for the copy and nothing new ran. The partial copies go, the
|
||||
// pin goes back, and the previous version is started again.
|
||||
@@ -1163,7 +1202,7 @@ func (m *Manager) RecoverUpdates() []string {
|
||||
m.logger.Printf("[ERROR] [stacks] update recovery: %s: starting the previous version failed: %v", name, err)
|
||||
}
|
||||
m.clearJournal(name)
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
||||
case UpdatePhaseUndoing:
|
||||
// v0.263.0: a power cut DURING the undo. Resumed like `starting` — the undo runs again from
|
||||
// the copies (still there: they are removed only after the undo succeeded) and then probes.
|
||||
@@ -1205,14 +1244,14 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
||||
for _, name := range names {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
||||
continue
|
||||
}
|
||||
m.updateJournalMu.Lock()
|
||||
e, ok := m.readUpdateJournal().Updates[name]
|
||||
m.updateJournalMu.Unlock()
|
||||
if !ok {
|
||||
m.finishUpdate(name, UpdatePhaseFailed, MsgUpdateInterrupted)
|
||||
m.finishUpdateKey(name, UpdatePhaseFailed, "update.error.interrupted", "")
|
||||
continue
|
||||
}
|
||||
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
|
||||
|
||||
Reference in New Issue
Block a user