dbcb306fcf
gates / gates (push) Successful in 8s
The box stops inferring the customer's intent from a container count and reads
what they actually asked for.
Part 1 — desired state. AppConfig gains a tri-state `desired_state`
(""/running/stopped), written ONLY by the customer's own action: the API action
switch, DeployStack, UpdateOptionalConfig's redeploy branch, and the .fab
import. Intent is written BEFORE the act and a failed write REFUSES the act.
StartStack/StopStack are deliberately not writers — 14 callers, only 2 are the
customer. bootrecon.isBootOrphan now reads intent instead of len(Containers)>0,
which closes R-157 mechanism B (a power cut or interrupted deploy left an app
with zero containers, read as a deliberate stop, and stranded silently).
ABSENT MEANS UNKNOWN, NEVER "running": every pre-v0.189.0 app.yaml reads absent,
so the legacy fallback is byte-identical to the old rule. A running-only startup
backfill converges the unambiguous cases; `stopped` is never inferred.
Part 2 — backup.AppStopGuard, a persisted marker over every stop→work→start
window (volume dump, offbox reconstitute, .fab export). Its own file, never
quiesce's. Written before the stop, cleared only after a restart that succeeded,
kept when one fails. Recover() completes before the boot reconciler is launched
and returns its outcome, which main.go reports on the existing backup_failed
event once the notifier exists. A defer is not the mechanism — a SIGKILL runs
none (Campaign 8 fault 10).
Also: SaveAppConfig rebuilt AppConfig field-by-field (the R-100 shape) and would
have dropped desired_state on every save across nine call sites. Replaced with
copy-and-overlay. Measured: app.yaml does not round-trip unknown YAML keys.
No hub change, no agent coupling, no user-visible string. 27/27 packages green;
7 red-proofs observed FAIL then restored.
252 lines
11 KiB
Go
252 lines
11 KiB
Go
// Package bootrecon implements R-52: the bounded, start-ONCE recovery of apps that were left
|
||
// behind by an interrupted boot.
|
||
//
|
||
// The live failure it closes (AUDIT-vacation-remote-ops-2026-07-20, finding F5): a pre-transport
|
||
// shutdown left `immich-server` and `calibre-web` Exited; ten sibling containers came back and
|
||
// those two did not, and they were still down ~18 hours later. The controller REPORTED them (the
|
||
// 30 s deadapp-check) but never started them — deployed-but-stopped was an alarm with no recovery.
|
||
//
|
||
// Two deliberate boundaries, both load-bearing:
|
||
//
|
||
// - **Bounded, never a loop.** At most `attempts` tries, `retryDelay` apart, then it stops and the
|
||
// alarm owns the problem. A restart loop would paper over a genuinely broken app forever and
|
||
// hammer docker while doing it.
|
||
// - **A user's Stop survives a reboot.** This is still the whole safety argument; only the way it
|
||
// is established changed. Until v0.189.0 it was inferred — the UI's Stop is `docker compose
|
||
// down`, which REMOVES containers, so "zero containers" was read as "the customer stopped it"
|
||
// and left alone. Since v0.189.0 (R-166) the customer's intent is RECORDED in app.yaml and read
|
||
// directly, because the inference could not distinguish a deliberate Stop from a power cut or an
|
||
// interrupted backup, and silently stranded both. An app.yaml with no recorded intent — every
|
||
// app on every box predating the field — keeps the old inference exactly. See isBootOrphan,
|
||
// TestReconcile_UserStoppedAppIsNeverStarted and
|
||
// TestReconcile_LegacyNoDesiredState_BehavesExactlyAsBefore.
|
||
//
|
||
// It runs inside the notifier's boot grace (cmd/controller/main.go `deadAppBootGrace`), so a
|
||
// successful recovery never fires an alert and a failed one alerts honestly once the grace expires.
|
||
package bootrecon
|
||
|
||
import (
|
||
"context"
|
||
"log"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||
)
|
||
|
||
// StackProvider is the slice of *stacks.Manager this package needs. Declared consumer-side so the
|
||
// tests can count StartStack calls without a docker anywhere near them.
|
||
type StackProvider interface {
|
||
GetStacks() []stacks.Stack
|
||
StartStack(name string) error
|
||
RefreshStatus() error
|
||
}
|
||
|
||
const (
|
||
// DefaultAttempts is the total number of start attempts per boot (not per app per retry-forever).
|
||
DefaultAttempts = 2
|
||
// DefaultRetryDelay spaces the attempts. 2 × 30 s fits comfortably inside the 90 s boot grace,
|
||
// so a recovery that works is silent and one that does not is honest.
|
||
DefaultRetryDelay = 30 * time.Second
|
||
)
|
||
|
||
// Reconciler performs the start-once sweep. Zero value is not usable — use New.
|
||
type Reconciler struct {
|
||
stacks StackProvider
|
||
logger *log.Logger
|
||
attempts int
|
||
retryDelay time.Duration
|
||
|
||
// sleep is the inter-attempt wait; injectable so tests never spend 30 real seconds.
|
||
sleep func(context.Context, time.Duration)
|
||
}
|
||
|
||
// Result is the outcome, returned for logging/testing (the hub learns about failures only through
|
||
// the existing app_start_failed alarm — this package deliberately pushes no events of its own).
|
||
type Result struct {
|
||
Candidates []string // boot-orphaned apps found
|
||
Recovered []string // running again by the end
|
||
StillDown []string // still down after the last attempt — the alarm's problem now
|
||
Attempts int // attempts actually made (0 when there was nothing to do)
|
||
}
|
||
|
||
// New builds a Reconciler with the shipped defaults.
|
||
func New(p StackProvider, logger *log.Logger) *Reconciler {
|
||
return &Reconciler{
|
||
stacks: p,
|
||
logger: logger,
|
||
attempts: DefaultAttempts,
|
||
retryDelay: DefaultRetryDelay,
|
||
sleep: sleepCtx,
|
||
}
|
||
}
|
||
|
||
func sleepCtx(ctx context.Context, d time.Duration) {
|
||
t := time.NewTimer(d)
|
||
defer t.Stop()
|
||
select {
|
||
case <-ctx.Done():
|
||
case <-t.C:
|
||
}
|
||
}
|
||
|
||
// isBootOrphan reports whether a stack is an app the boot left behind.
|
||
//
|
||
// The gate, term by term:
|
||
// - Deployed — an app that is installed. NOTE: `Deployed` means INSTALLED, not "wanted running";
|
||
// the two were conflated until v0.189.0 and that conflation is what the desired-state term below
|
||
// repairs.
|
||
// - not Protected — traefik/cloudflared/felhom-controller have their own supervision; this must
|
||
// never race the base-stack self-heal.
|
||
// - not Deploying — mid-deploy is not a fault.
|
||
// - desired state — see below. REPLACES the old container-count term.
|
||
// - IsDownState — stopped/exited/degraded (R-51 included: a boot that half-started a stack is the
|
||
// same interrupted-boot shape).
|
||
//
|
||
// ── WHY INTENT REPLACED THE CONTAINER COUNT (R-166, closing R-157 mechanism B) ────────────────────
|
||
//
|
||
// This gate used to end in `len(s.Containers) > 0`, and its comment called that "the D-case guard":
|
||
// a UI Stop is `compose down`, which REMOVES containers, so zero containers was read as "the
|
||
// customer stopped this" and left alone. The safety goal was right and still holds. The SIGNAL was
|
||
// wrong, because zero containers has at least three causes and the count cannot tell them apart:
|
||
//
|
||
// a deliberate Stop → must stay down
|
||
// a power cut mid-compose, or an interrupted deploy → must come back
|
||
// a backup that stopped the app and died before restarting it → must come back
|
||
//
|
||
// Two of those three were silently unrecoverable: the app simply stayed gone until a human noticed.
|
||
// The count was never capable of separating them, so the fix is not a better inference — it is to
|
||
// stop inferring and read what the customer actually asked for, which app.yaml now records.
|
||
//
|
||
// ── WHAT ABSENT STILL MEANS, AND WHY THE OLD BEHAVIOUR IS KEPT ────────────────────────────────────
|
||
//
|
||
// DesiredStateUnknown falls back to the ORIGINAL container-count rule, byte-for-byte. This is the
|
||
// single most important line in the change. Every app.yaml on every existing box predates the field,
|
||
// so absent is what the whole fleet reads on upgrade; treating absent as "running" would start, on
|
||
// the first boot after the upgrade, every app its owner had deliberately stopped. The fallback is
|
||
// what makes this feature inert for an app nobody has pressed a button on since — see
|
||
// TestReconcile_LegacyNoDesiredState_BehavesExactlyAsBefore and its red-proof.
|
||
//
|
||
// The full decision table (§8.1):
|
||
//
|
||
// desired containers state → result
|
||
// stopped any any → never an orphan (the customer said so)
|
||
// running 0 — → ORPHAN ← the R-157 case, invisible before v0.189.0
|
||
// running >0 IsDownState → ORPHAN (unchanged)
|
||
// running >0 up → not an orphan
|
||
// absent 0 — → not an orphan (exactly the pre-v0.189.0 behaviour)
|
||
// absent >0 IsDownState → ORPHAN (exactly the pre-v0.189.0 behaviour)
|
||
func isBootOrphan(s stacks.Stack) bool {
|
||
if !s.Deployed || s.Protected || s.Deploying {
|
||
return false
|
||
}
|
||
switch stacks.DesiredStateOf(s) {
|
||
case stacks.DesiredStateStopped:
|
||
// The customer pressed Stop. No observation may overturn that — not a missing container, not
|
||
// a down state, not a reboot. Nothing else in this package starts an app.
|
||
return false
|
||
case stacks.DesiredStateRunning:
|
||
// Wanted running. ANY way of not being up is a fault to repair, including having no
|
||
// containers at all — which is the case the old count term structurally could not see.
|
||
return len(s.Containers) == 0 || stacks.IsDownState(s.State)
|
||
default:
|
||
// DesiredStateUnknown — legacy. Keep the pre-R-166 rule exactly.
|
||
return len(s.Containers) > 0 && stacks.IsDownState(s.State)
|
||
}
|
||
}
|
||
|
||
// Run performs the sweep once and returns what happened. It is safe to call with no boot orphans
|
||
// (the quiet path logs one DEBUG-free INFO-free line — see below) and it never returns an error:
|
||
// a failure to start is an app-level fact the alarm reports, not a controller startup failure.
|
||
func (r *Reconciler) Run(ctx context.Context) Result {
|
||
var res Result
|
||
|
||
pending := map[string]bool{}
|
||
for _, s := range r.stacks.GetStacks() {
|
||
if isBootOrphan(s) {
|
||
pending[s.Name] = true
|
||
res.Candidates = append(res.Candidates, s.Name)
|
||
}
|
||
}
|
||
sortStrings(res.Candidates)
|
||
|
||
if len(pending) == 0 {
|
||
// The healthy path must be observable — "no alarms" and "never ran" have to be
|
||
// distinguishable in a log (the v0.91.2 lesson).
|
||
r.logger.Printf("[INFO] [bootrecon] Boot reconciliation: no boot-orphaned apps (nothing to start)")
|
||
return res
|
||
}
|
||
|
||
r.logger.Printf("[INFO] [bootrecon] Boot reconciliation: %d boot-orphaned app(s) found: %v — up to %d attempt(s)",
|
||
len(res.Candidates), res.Candidates, r.attempts)
|
||
|
||
for attempt := 1; attempt <= r.attempts && len(pending) > 0; attempt++ {
|
||
res.Attempts = attempt
|
||
for _, name := range sortedKeys(pending) {
|
||
if ctx.Err() != nil {
|
||
break
|
||
}
|
||
start := time.Now()
|
||
if err := r.stacks.StartStack(name); err != nil {
|
||
r.logger.Printf("[WARN] [bootrecon] Boot reconciliation attempt %d/%d: start %q failed after %.1fs: %v",
|
||
attempt, r.attempts, name, time.Since(start).Seconds(), err)
|
||
continue
|
||
}
|
||
r.logger.Printf("[INFO] [bootrecon] Boot reconciliation attempt %d/%d: started %q (took %.1fs)",
|
||
attempt, r.attempts, name, time.Since(start).Seconds())
|
||
}
|
||
if ctx.Err() != nil {
|
||
break
|
||
}
|
||
|
||
// Re-read reality rather than trusting a nil error: `compose up -d` exits 0 on a crash-loop
|
||
// (a session-critical invariant of this repo), so only a fresh docker ps can say whether the
|
||
// app is actually up.
|
||
if err := r.stacks.RefreshStatus(); err != nil {
|
||
r.logger.Printf("[WARN] [bootrecon] Boot reconciliation: status refresh failed: %v", err)
|
||
}
|
||
for _, s := range r.stacks.GetStacks() {
|
||
if pending[s.Name] && !stacks.IsDownState(s.State) {
|
||
delete(pending, s.Name)
|
||
res.Recovered = append(res.Recovered, s.Name)
|
||
}
|
||
}
|
||
|
||
if len(pending) > 0 && attempt < r.attempts {
|
||
r.sleep(ctx, r.retryDelay)
|
||
}
|
||
}
|
||
|
||
res.StillDown = sortedKeys(pending)
|
||
sortStrings(res.Recovered)
|
||
|
||
if len(res.StillDown) == 0 {
|
||
r.logger.Printf("[INFO] [bootrecon] Boot reconciliation complete: %d app(s) recovered in %d attempt(s): %v",
|
||
len(res.Recovered), res.Attempts, res.Recovered)
|
||
} else {
|
||
// Deliberately no hub event here: the app_start_failed alarm fires on its own once the boot
|
||
// grace expires, and two events for one dead app is how an operator inbox becomes noise.
|
||
r.logger.Printf("[WARN] [bootrecon] Boot reconciliation gave up after %d attempt(s): recovered=%v still down=%v (the dead-app alarm now owns these)",
|
||
res.Attempts, res.Recovered, res.StillDown)
|
||
}
|
||
return res
|
||
}
|
||
|
||
// --- tiny local helpers (no dependency on sort ordering semantics elsewhere) ---
|
||
|
||
func sortedKeys(m map[string]bool) []string {
|
||
out := make([]string, 0, len(m))
|
||
for k := range m {
|
||
out = append(out, k)
|
||
}
|
||
sortStrings(out)
|
||
return out
|
||
}
|
||
|
||
func sortStrings(s []string) {
|
||
for i := 1; i < len(s); i++ {
|
||
for j := i; j > 0 && s[j] < s[j-1]; j-- {
|
||
s[j], s[j-1] = s[j-1], s[j]
|
||
}
|
||
}
|
||
}
|