v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.
The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.
Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.
Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.
Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -2,6 +2,7 @@ package stacks
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
@@ -130,17 +131,28 @@ type HealthCheckDetail struct {
|
||||
|
||||
// Stack represents a docker compose stack on disk.
|
||||
type Stack struct {
|
||||
Name string `json:"name"`
|
||||
Meta Metadata `json:"meta"`
|
||||
ComposePath string `json:"compose_path"`
|
||||
State ContainerState `json:"state"`
|
||||
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
|
||||
Protected bool `json:"protected"`
|
||||
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
|
||||
Containers []ContainerInfo `json:"containers"`
|
||||
AppConfig *AppConfig `json:"app_config,omitempty"`
|
||||
Deploying bool `json:"deploying"` // compose up in progress
|
||||
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
|
||||
Name string `json:"name"`
|
||||
Meta Metadata `json:"meta"`
|
||||
ComposePath string `json:"compose_path"`
|
||||
State ContainerState `json:"state"`
|
||||
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
|
||||
Protected bool `json:"protected"`
|
||||
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
|
||||
Containers []ContainerInfo `json:"containers"`
|
||||
AppConfig *AppConfig `json:"app_config,omitempty"`
|
||||
Deploying bool `json:"deploying"` // compose up in progress
|
||||
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
|
||||
// Updating / UpdatePhase / UpdatePhaseLabel / UpdateError (update arc slice 4, v0.237.0) are the
|
||||
// guarded update's in-memory progress, the same shape as Deploying/DeployError: the API answers
|
||||
// 202 at once and the page polls GET /api/stacks/{name}. See update.go.
|
||||
Updating bool `json:"updating"`
|
||||
UpdatePhase string `json:"update_phase,omitempty"`
|
||||
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
|
||||
UpdateError string `json:"update_error,omitempty"`
|
||||
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
|
||||
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
|
||||
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
|
||||
HoldReason string `json:"hold_reason,omitempty"`
|
||||
HealthProbe *HealthProbeResult `json:"health_probe,omitempty"` // controller-side probe result
|
||||
LastUpdated time.Time `json:"last_updated"`
|
||||
// RestartingSince (C9-F2) is when this stack was FIRST observed in StateRestarting during the
|
||||
@@ -198,6 +210,16 @@ type Manager struct {
|
||||
restartPolicyCache map[string]string
|
||||
// execFn replaces execCommand's process boundary in tests; nil in production.
|
||||
execFn func(name string, args ...string) (string, error)
|
||||
|
||||
// --- guarded update (slice 4, update.go) ---
|
||||
updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED
|
||||
updateComposeFn func(dir string, env []string, args ...string) (string, error)
|
||||
updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string)
|
||||
updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string)
|
||||
updateDiskFreeFn func() (freeGiB float64, ok bool)
|
||||
updateNowFn func() time.Time
|
||||
updateJournalMu sync.Mutex
|
||||
updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist
|
||||
// inspectRestartPolicyFn is the docker-inspect seam for the above; nil in production
|
||||
// (dockerRestartPolicy). Tests inject a scripted lookup and never touch docker.
|
||||
inspectRestartPolicyFn func(containerName string) (string, error)
|
||||
@@ -960,12 +982,15 @@ func aggregateState(containers []ContainerInfo, policyOf restartPolicyLookup) Co
|
||||
|
||||
func (m *Manager) GetStacks() []Stack {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
|
||||
result := make([]Stack, 0, len(m.stacks))
|
||||
for _, s := range m.stacks {
|
||||
result = append(result, deepCopyStack(s))
|
||||
}
|
||||
g := m.updateGuards
|
||||
m.mu.RUnlock()
|
||||
for i := range result {
|
||||
fillHoldReason(g, &result[i])
|
||||
}
|
||||
|
||||
// Sort alphabetically by display name for consistent UI ordering
|
||||
sort.Slice(result, func(i, j int) bool {
|
||||
@@ -984,6 +1009,7 @@ func (m *Manager) GetStack(name string) (*Stack, bool) {
|
||||
return nil, false
|
||||
}
|
||||
cp := deepCopyStack(s)
|
||||
fillHoldReason(m.updateGuards, &cp)
|
||||
return &cp, true
|
||||
}
|
||||
|
||||
@@ -1224,54 +1250,11 @@ func (m *Manager) RestartStack(name string) error {
|
||||
return m.RefreshStatus()
|
||||
}
|
||||
|
||||
func (m *Manager) UpdateStack(name string) error {
|
||||
stack, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
return fmt.Errorf("stack %q not found", name)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Updating stack: %s", name)
|
||||
start := time.Now()
|
||||
dir := filepath.Dir(stack.ComposePath)
|
||||
|
||||
// v0.235.0 — ADVANCE THE PIN FIRST, AND RE-RENDER BEFORE THE PULL.
|
||||
//
|
||||
// This is the ONE act entitled to move a version; the freeze exists so that nothing else can.
|
||||
// The ordering is load-bearing, not stylistic: `compose pull` and `up -d` act on the file on
|
||||
// disk, so the catalog's current definition has to BE that file before either runs. Setting the
|
||||
// pin afterwards would pull the frozen version and change nothing, while reporting success — and
|
||||
// a button that lies is worse than a button that refuses.
|
||||
//
|
||||
// A FAILED PIN WRITE REFUSES THE UPDATE, deliberately the opposite of recordInstalledImages.
|
||||
// That field is an observation and a failed write is a bookkeeping gap; this one is INTENT, and
|
||||
// an update whose intent could not be recorded leaves the box running a version it has no record
|
||||
// of choosing — the exact ambiguity R-166 closed for desired_state, one field over.
|
||||
if err := m.advancePinToCatalog(name, dir); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] Stack %s update refused: %v", name, err)
|
||||
return fmt.Errorf("updating stack %s: %w", name, err)
|
||||
}
|
||||
|
||||
env := m.stackEnv(dir)
|
||||
|
||||
if m.isDebug() {
|
||||
m.checkLocalImages(name, dir)
|
||||
}
|
||||
|
||||
if _, err := m.composeExecCustomEnv(dir, env, "pull"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] Stack %s update (pull) failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
|
||||
return fmt.Errorf("pulling images for %s: %w", name, err)
|
||||
}
|
||||
|
||||
if _, err := m.composeExecCustomEnv(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] Stack %s update (up) failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
|
||||
return fmt.Errorf("recreating %s: %w", name, err)
|
||||
}
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Stack %s updated successfully (took %.1fs)", name, time.Since(start).Seconds())
|
||||
m.recordInstalledImages(name, dir, env)
|
||||
m.logPostStartStatus(name, dir, env)
|
||||
return m.RefreshStatus()
|
||||
}
|
||||
// UpdateStack was REMOVED in v0.237.0 (update arc slice 4). It advanced the pin, pulled, ran `up -d`
|
||||
// and reported success on the compose exit code — no copy first, no refusals, and HTTP 200 over a
|
||||
// crash loop (R-443). Its only caller was the API, which now runs StartGuardedUpdate (update.go).
|
||||
// Deleting it rather than leaving it is deliberate: an unguarded update path that still compiles is
|
||||
// one caller away from being the next R-439.
|
||||
|
||||
func (m *Manager) GetLogs(name string, lines int) (string, error) {
|
||||
stack, ok := m.GetStack(name)
|
||||
|
||||
Reference in New Issue
Block a user