c393d8529a
The dead-app check (source of app_start_failed and app_stopped_unhealthy) now gates on a crash-aware
boot grace (internal/crashboot): 15 min when the host crash guard's last boot was UNCLEAN and within
30 min of the controller start, otherwise 90 s. The fact is read from the agent's local API
(GET /host/crash-guard, agentapi.Client.CrashGuard). UNKNOWN - no agent, an older agent's 404, no
crash-guard state - is a normal boot. The decision is logged once ("boot grace ...: ... (R-856)").
NEEDS AN AGENT CHANGE to take effect: GET /host/crash-guard serving the guard's state.json fields
(present, last_boot_at, last_boot_unclean, tripped). Until then every box keeps 90 s.
Tests: TestR856_CrashBootHoldsTheMailsForTheLongGrace, TestR856_NormalBootKeeps90s,
TestR856_FactReadLateInTheNormalGraceStillCounts, TestR856_AgentProbeReadsTheCrashGuardState,
TestR856_CrashGuardDecodesAndAnOlderAgentIs404, TestR856_DeadAppCheckWaitsOnTheCrashAwareGrace,
TestR856_NormalGraceIsTheDeadAppBootGrace.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
154 lines
5.6 KiB
Go
154 lines
5.6 KiB
Go
// Package crashboot decides how long the controller's app mails wait after the controller starts
|
||
// (R-856, `09` §3 decision 143, extending decision 129).
|
||
//
|
||
// After a NORMAL start the dead-app check waits NormalGrace (90 s) — apps legitimately take 30–60 s
|
||
// to come up. After a CRASH boot of the host (a kernel crash, a power cut or a hard reset — the crash
|
||
// guard cannot tell them apart, `11` §5.9) the apps come up slower and the hub already tells the
|
||
// household „restarted after an unexpected stop"; the 2026-10-04 crash-guard test on demo-hp then
|
||
// mailed `app_start_failed` and `app_stopped_unhealthy` 3.5 and 9 minutes after the boot, for apps
|
||
// that were still coming up. So after a crash boot the wait is CrashGrace (about 15 minutes).
|
||
//
|
||
// THE FACT comes from the host's crash guard (`felhom-crash-guard`, state.json `last_boot_unclean` +
|
||
// `last_boot_at`), through the agent's local API. The guest cannot see the host's state file, so it
|
||
// is a Probe seam.
|
||
//
|
||
// UNKNOWN IS A NORMAL BOOT. An agent that predates the route (404), an unreachable agent, a box with
|
||
// no crash guard, an unparseable time: all keep today's 90 s. A longer silence must never come from
|
||
// a guess — it would hide a real outage for 15 minutes on every box whose fact we cannot read.
|
||
package crashboot
|
||
|
||
import (
|
||
"context"
|
||
"log"
|
||
"sync"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/agentapi"
|
||
)
|
||
|
||
const (
|
||
// NormalGrace is the dead-app boot grace after an ordinary start (the pre-R-856 deadAppBootGrace).
|
||
NormalGrace = 90 * time.Second
|
||
// CrashGrace is the wait after a crash boot — „about 15 minutes" (decision 143). The 2026-10-04
|
||
// mails came at 3.5 and 9 minutes; 15 covers both with room.
|
||
CrashGrace = 15 * time.Minute
|
||
// BootWindow bounds how recent the host's unclean boot must be for THIS controller start to be the
|
||
// one that followed it. A controller restarted days after a crash (a self-update, a kill) is a
|
||
// normal start; a guest starts its controller well within this window after a host boot.
|
||
BootWindow = 30 * time.Minute
|
||
probeTimeout = 5 * time.Second
|
||
)
|
||
|
||
// Fact is what the host's crash guard says about its most recent boot.
|
||
type Fact struct {
|
||
Known bool // false: no crash guard, no state, or an agent that cannot say
|
||
Unclean bool // the most recent host boot followed an unclean stop
|
||
BootAt time.Time // when that boot happened
|
||
}
|
||
|
||
// Probe asks the host for the fact. An error means unknown.
|
||
type Probe func(ctx context.Context) (Fact, error)
|
||
|
||
// Grace answers "must the app mails still wait?" for one controller run.
|
||
type Grace struct {
|
||
start time.Time
|
||
probe Probe
|
||
logger *log.Logger
|
||
|
||
mu sync.Mutex
|
||
resolved bool
|
||
crashBoot bool
|
||
}
|
||
|
||
// New builds the grace for a controller that started at start. probe may be nil (no agent): unknown.
|
||
func New(start time.Time, probe Probe, logger *log.Logger) *Grace {
|
||
return &Grace{start: start, probe: probe, logger: logger}
|
||
}
|
||
|
||
// Within reports whether now is still inside the boot grace, i.e. the app mails must wait.
|
||
//
|
||
// The fact is asked for while the normal grace runs (the agent may come up a few seconds after the
|
||
// controller) and once more at its end; whatever is known then decides, and unknown resolves to a
|
||
// normal boot. Once resolved it is never asked again.
|
||
func (g *Grace) Within(now time.Time) bool {
|
||
elapsed := now.Sub(g.start)
|
||
return elapsed < g.duration(elapsed)
|
||
}
|
||
|
||
// Duration is the grace this run uses (resolving it if it can). For logs and tests.
|
||
func (g *Grace) Duration(now time.Time) time.Duration {
|
||
return g.duration(now.Sub(g.start))
|
||
}
|
||
|
||
func (g *Grace) duration(elapsed time.Duration) time.Duration {
|
||
g.mu.Lock()
|
||
defer g.mu.Unlock()
|
||
if !g.resolved {
|
||
g.tryResolve(elapsed >= NormalGrace)
|
||
}
|
||
if g.crashBoot {
|
||
return CrashGrace
|
||
}
|
||
return NormalGrace
|
||
}
|
||
|
||
// tryResolve asks the probe once. final: the normal grace is over, so an unknown answer is final too.
|
||
func (g *Grace) tryResolve(final bool) {
|
||
if g.probe == nil {
|
||
g.resolve(false, "no agent to ask — a normal boot")
|
||
return
|
||
}
|
||
ctx, cancel := context.WithTimeout(context.Background(), probeTimeout)
|
||
f, err := g.probe(ctx)
|
||
cancel()
|
||
switch {
|
||
case err != nil || !f.Known:
|
||
if final {
|
||
why := "the host's crash guard has no record"
|
||
if err != nil {
|
||
why = "the host's crash guard could not be read (" + err.Error() + ")"
|
||
}
|
||
g.resolve(false, why+" — treated as a normal boot")
|
||
}
|
||
case !f.Unclean:
|
||
g.resolve(false, "the host's last boot was clean")
|
||
case f.BootAt.IsZero() || g.start.Sub(f.BootAt) > BootWindow:
|
||
g.resolve(false, "the host's last unclean boot ("+f.BootAt.UTC().Format(time.RFC3339)+") is not the one this start followed")
|
||
default:
|
||
g.crashBoot = true
|
||
g.resolve(true, "the host's last boot ("+f.BootAt.UTC().Format(time.RFC3339)+") followed an UNCLEAN stop")
|
||
}
|
||
}
|
||
|
||
func (g *Grace) resolve(crash bool, why string) {
|
||
g.resolved = true
|
||
if g.logger == nil {
|
||
return
|
||
}
|
||
d := NormalGrace
|
||
if crash {
|
||
d = CrashGrace
|
||
}
|
||
g.logger.Printf("[INFO] [deadapp] boot grace %s: %s (R-856)", d, why)
|
||
}
|
||
|
||
// AgentProbe adapts the agent's GET /host/crash-guard (agentapi.Client.CrashGuard) to a Probe. A
|
||
// state the host has not written (Present=false) is unknown; an unparseable boot time is a known
|
||
// unclean boot with no time, which New's window check reads as NOT this start's — a normal boot.
|
||
func AgentProbe(get func(ctx context.Context) (agentapi.CrashGuardState, error)) Probe {
|
||
return func(ctx context.Context) (Fact, error) {
|
||
st, err := get(ctx)
|
||
if err != nil {
|
||
return Fact{}, err
|
||
}
|
||
if !st.Present {
|
||
return Fact{}, nil
|
||
}
|
||
f := Fact{Known: true, Unclean: st.LastBootUnclean}
|
||
if t, perr := time.Parse(time.RFC3339, st.LastBootAt); perr == nil {
|
||
f.BootAt = t
|
||
}
|
||
return f, nil
|
||
}
|
||
}
|