// Package crashboot decides how long the controller's app mails wait after the controller starts // (R-856, `09` §3 decision 143, extending decision 129). // // After a NORMAL start the dead-app check waits NormalGrace (90 s) — apps legitimately take 30–60 s // to come up. After a CRASH boot of the host (a kernel crash, a power cut or a hard reset — the crash // guard cannot tell them apart, `11` §5.9) the apps come up slower and the hub already tells the // household „restarted after an unexpected stop"; the 2026-10-04 crash-guard test on demo-hp then // mailed `app_start_failed` and `app_stopped_unhealthy` 3.5 and 9 minutes after the boot, for apps // that were still coming up. So after a crash boot the wait is CrashGrace (about 15 minutes). // // THE FACT comes from the host's crash guard (`felhom-crash-guard`, state.json `last_boot_unclean` + // `last_boot_at`), through the agent's local API. The guest cannot see the host's state file, so it // is a Probe seam. // // UNKNOWN IS A NORMAL BOOT. An agent that predates the route (404), an unreachable agent, a box with // no crash guard, an unparseable time: all keep today's 90 s. A longer silence must never come from // a guess — it would hide a real outage for 15 minutes on every box whose fact we cannot read. package crashboot import ( "context" "log" "sync" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/agentapi" ) const ( // NormalGrace is the dead-app boot grace after an ordinary start (the pre-R-856 deadAppBootGrace). NormalGrace = 90 * time.Second // CrashGrace is the wait after a crash boot — „about 15 minutes" (decision 143). The 2026-10-04 // mails came at 3.5 and 9 minutes; 15 covers both with room. CrashGrace = 15 * time.Minute // BootWindow bounds how recent the host's unclean boot must be for THIS controller start to be the // one that followed it. A controller restarted days after a crash (a self-update, a kill) is a // normal start; a guest starts its controller well within this window after a host boot. BootWindow = 30 * time.Minute probeTimeout = 5 * time.Second ) // Fact is what the host's crash guard says about its most recent boot. type Fact struct { Known bool // false: no crash guard, no state, or an agent that cannot say Unclean bool // the most recent host boot followed an unclean stop BootAt time.Time // when that boot happened } // Probe asks the host for the fact. An error means unknown. type Probe func(ctx context.Context) (Fact, error) // Grace answers "must the app mails still wait?" for one controller run. type Grace struct { start time.Time probe Probe logger *log.Logger mu sync.Mutex resolved bool crashBoot bool } // New builds the grace for a controller that started at start. probe may be nil (no agent): unknown. func New(start time.Time, probe Probe, logger *log.Logger) *Grace { return &Grace{start: start, probe: probe, logger: logger} } // Within reports whether now is still inside the boot grace, i.e. the app mails must wait. // // The fact is asked for while the normal grace runs (the agent may come up a few seconds after the // controller) and once more at its end; whatever is known then decides, and unknown resolves to a // normal boot. Once resolved it is never asked again. func (g *Grace) Within(now time.Time) bool { elapsed := now.Sub(g.start) return elapsed < g.duration(elapsed) } // Duration is the grace this run uses (resolving it if it can). For logs and tests. func (g *Grace) Duration(now time.Time) time.Duration { return g.duration(now.Sub(g.start)) } func (g *Grace) duration(elapsed time.Duration) time.Duration { g.mu.Lock() defer g.mu.Unlock() if !g.resolved { g.tryResolve(elapsed >= NormalGrace) } if g.crashBoot { return CrashGrace } return NormalGrace } // tryResolve asks the probe once. final: the normal grace is over, so an unknown answer is final too. func (g *Grace) tryResolve(final bool) { if g.probe == nil { g.resolve(false, "no agent to ask — a normal boot") return } ctx, cancel := context.WithTimeout(context.Background(), probeTimeout) f, err := g.probe(ctx) cancel() switch { case err != nil || !f.Known: if final { why := "the host's crash guard has no record" if err != nil { why = "the host's crash guard could not be read (" + err.Error() + ")" } g.resolve(false, why+" — treated as a normal boot") } case !f.Unclean: g.resolve(false, "the host's last boot was clean") case f.BootAt.IsZero() || g.start.Sub(f.BootAt) > BootWindow: g.resolve(false, "the host's last unclean boot ("+f.BootAt.UTC().Format(time.RFC3339)+") is not the one this start followed") default: g.crashBoot = true g.resolve(true, "the host's last boot ("+f.BootAt.UTC().Format(time.RFC3339)+") followed an UNCLEAN stop") } } func (g *Grace) resolve(crash bool, why string) { g.resolved = true if g.logger == nil { return } d := NormalGrace if crash { d = CrashGrace } g.logger.Printf("[INFO] [deadapp] boot grace %s: %s (R-856)", d, why) } // AgentProbe adapts the agent's GET /host/crash-guard (agentapi.Client.CrashGuard) to a Probe. A // state the host has not written (Present=false) is unknown; an unparseable boot time is a known // unclean boot with no time, which New's window check reads as NOT this start's — a normal boot. func AgentProbe(get func(ctx context.Context) (agentapi.CrashGuardState, error)) Probe { return func(ctx context.Context) (Fact, error) { st, err := get(ctx) if err != nil { return Fact{}, err } if !st.Present { return Fact{}, nil } f := Fact{Known: true, Unclean: st.LastBootUnclean} if t, perr := time.Parse(time.RFC3339, st.LastBootAt); perr == nil { f.BootAt = t } return f, nil } }