R-856: after a crash boot of the host, app mails wait ~15 minutes; a normal boot keeps 90 s (09 decision 143)

The dead-app check (source of app_start_failed and app_stopped_unhealthy) now gates on a crash-aware
boot grace (internal/crashboot): 15 min when the host crash guard's last boot was UNCLEAN and within
30 min of the controller start, otherwise 90 s. The fact is read from the agent's local API
(GET /host/crash-guard, agentapi.Client.CrashGuard). UNKNOWN - no agent, an older agent's 404, no
crash-guard state - is a normal boot. The decision is logged once ("boot grace ...: ... (R-856)").

NEEDS AN AGENT CHANGE to take effect: GET /host/crash-guard serving the guard's state.json fields
(present, last_boot_at, last_boot_unclean, tripped). Until then every box keeps 90 s.

Tests: TestR856_CrashBootHoldsTheMailsForTheLongGrace, TestR856_NormalBootKeeps90s,
TestR856_FactReadLateInTheNormalGraceStillCounts, TestR856_AgentProbeReadsTheCrashGuardState,
TestR856_CrashGuardDecodesAndAnOlderAgentIs404, TestR856_DeadAppCheckWaitsOnTheCrashAwareGrace,
TestR856_NormalGraceIsTheDeadAppBootGrace.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-06 11:38:19 +02:00
parent 2d63714eca
commit c393d8529a
7 changed files with 411 additions and 2 deletions
+17 -2
View File
@@ -35,6 +35,7 @@ import (
"gitea.dooplex.hu/admin/felhom-controller/internal/channelhealth"
cf "gitea.dooplex.hu/admin/felhom-controller/internal/cloudflare"
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
"gitea.dooplex.hu/admin/felhom-controller/internal/crashboot"
"gitea.dooplex.hu/admin/felhom-controller/internal/crypto"
"gitea.dooplex.hu/admin/felhom-controller/internal/fillwatch"
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
@@ -936,9 +937,23 @@ func main() {
// deadAppHeartbeatEvery-th scan emits one INFO carrying the scan count and what it found, so an
// operator can always answer "is it running, and what does it see?" from a default box — and a
// STALLED detector is visible as the heartbeat stopping.
//
// R-856 (`09` §3 decision 143, extending 129): after a CRASH boot of the host the grace is about 15
// minutes, not 90 s — the 2026-10-04 crash-guard test mailed app_start_failed / app_stopped_unhealthy
// 3.5 and 9 minutes after the boot for apps still coming up. The fact is the host crash guard's,
// read through the agent (GET /host/crash-guard); unknown (no agent, an older agent's 404) keeps 90 s.
var crashProbe crashboot.Probe
if cfg.LocalAPI.Endpoint != "" && cfg.LocalAPI.Token != "" {
if ac, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint); err != nil {
logger.Printf("[WARN] [deadapp] agent client init failed (%v) — the boot grace cannot learn of a crash boot; 90 s", err)
} else {
crashProbe = crashboot.AgentProbe(ac.CrashGuard)
}
}
bootGrace := crashboot.New(startTime, crashProbe, logger)
sched.Every("deadapp-check", 30*time.Second, func(ctx context.Context) error {
if time.Since(startTime) < deadAppBootGrace {
return nil // still inside the startup settle window
if bootGrace.Within(time.Now()) {
return nil // still inside the startup settle window (90 s, or ~15 min after a crash boot)
}
dead, states := scanDeployedAppRunStates(stackMgr, quiesceLoop, appStopGuard, backupMgr)
alertMgr.SetDeadAppAlerts(dead)
@@ -0,0 +1,64 @@
package main
import (
"go/ast"
"go/parser"
"go/token"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/crashboot"
)
// R-856 (`09` §3 decision 143): the dead-app check — the source of app_start_failed and
// app_stopped_unhealthy — waits on the crash-aware boot grace, not on the fixed 90 s constant.
//
// Red-proof: put `time.Since(startTime) < deadAppBootGrace` back as the check's gate — this test fails.
func TestR856_DeadAppCheckWaitsOnTheCrashAwareGrace(t *testing.T) {
f, err := parser.ParseFile(token.NewFileSet(), "main.go", nil, 0)
if err != nil {
t.Fatal(err)
}
var body *ast.BlockStmt
ast.Inspect(f, func(n ast.Node) bool {
call, ok := n.(*ast.CallExpr)
if !ok || len(call.Args) < 3 {
return true
}
if sel, ok := call.Fun.(*ast.SelectorExpr); !ok || sel.Sel.Name != "Every" {
return true
}
if lit, ok := call.Args[0].(*ast.BasicLit); ok && lit.Value == `"deadapp-check"` {
if fl, ok := call.Args[2].(*ast.FuncLit); ok {
body = fl.Body
}
}
return true
})
if body == nil {
t.Fatal("the deadapp-check schedule is gone from main.go")
}
within, constant := false, false
ast.Inspect(body, func(n ast.Node) bool {
switch x := n.(type) {
case *ast.SelectorExpr:
if id, ok := x.X.(*ast.Ident); ok && id.Name == "bootGrace" && x.Sel.Name == "Within" {
within = true
}
case *ast.Ident:
if x.Name == "deadAppBootGrace" {
constant = true
}
}
return true
})
if !within || constant {
t.Errorf("the dead-app check must gate on bootGrace.Within (found=%v) and not on deadAppBootGrace (found=%v)", within, constant)
}
}
// The normal grace is today's 90 s: a normal boot must keep it exactly (decision 143's other half).
func TestR856_NormalGraceIsTheDeadAppBootGrace(t *testing.T) {
if crashboot.NormalGrace != deadAppBootGrace {
t.Errorf("crashboot.NormalGrace = %s, deadAppBootGrace = %s — a normal boot must keep today's grace", crashboot.NormalGrace, deadAppBootGrace)
}
}
@@ -0,0 +1,33 @@
package agentapi
import (
"context"
"encoding/json"
"fmt"
)
// CrashGuardState mirrors the agent's GET /host/crash-guard (R-856, `09` §3 decision 143): what the
// host's crash guard (`felhom-crash-guard`, `11` §5.9) recorded about the most recent HOST boot. The
// agent reads /var/lib/felhom-crash-guard/state.json (0644) and passes these fields through.
//
// Present=false: the host has no crash guard or no state yet. An agent that predates the route answers
// 404 (a *StatusError) — both mean UNKNOWN, and the controller then keeps its normal boot grace.
type CrashGuardState struct {
Present bool `json:"present"`
LastBootAt string `json:"last_boot_at,omitempty"` // RFC3339 UTC ("2006-01-02T15:04:05Z")
LastBootUnclean bool `json:"last_boot_unclean"`
Tripped bool `json:"tripped"`
}
// CrashGuard calls GET /host/crash-guard.
func (c *Client) CrashGuard(ctx context.Context) (CrashGuardState, error) {
var out CrashGuardState
body, err := c.get(ctx, "/host/crash-guard")
if err != nil {
return out, err
}
if err := json.Unmarshal(body, &out); err != nil {
return out, fmt.Errorf("agentapi: decode /host/crash-guard: %w", err)
}
return out, nil
}
@@ -0,0 +1,37 @@
package agentapi
import (
"context"
"errors"
"net/http"
"net/http/httptest"
"strings"
"testing"
)
// R-856: GET /host/crash-guard decodes the crash guard's fields, and an agent that predates the route
// surfaces as a typed 404 (the caller reads it as unknown → the normal boot grace).
func TestR856_CrashGuardDecodesAndAnOlderAgentIs404(t *testing.T) {
mux := http.NewServeMux()
mux.HandleFunc("GET /host/crash-guard", func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{"ok":true,"data":{"present":true,"last_boot_at":"2026-10-04T12:00:00Z","last_boot_unclean":true,"tripped":false}}`))
})
s := httptest.NewTLSServer(mux)
defer s.Close()
c := clientFor(t, s, strings.TrimPrefix(s.URL, "https://"))
st, err := c.CrashGuard(context.Background())
if err != nil {
t.Fatal(err)
}
if !st.Present || !st.LastBootUnclean || st.LastBootAt != "2026-10-04T12:00:00Z" || st.Tripped {
t.Fatalf("state = %+v", st)
}
old := httptest.NewTLSServer(http.NewServeMux())
defer old.Close()
_, err = clientFor(t, old, strings.TrimPrefix(old.URL, "https://")).CrashGuard(context.Background())
var se *StatusError
if !errors.As(err, &se) || se.Code != http.StatusNotFound {
t.Fatalf("an older agent must answer a typed 404, got %v", err)
}
}
+153
View File
@@ -0,0 +1,153 @@
// Package crashboot decides how long the controller's app mails wait after the controller starts
// (R-856, `09` §3 decision 143, extending decision 129).
//
// After a NORMAL start the dead-app check waits NormalGrace (90 s) — apps legitimately take 30–60 s
// to come up. After a CRASH boot of the host (a kernel crash, a power cut or a hard reset — the crash
// guard cannot tell them apart, `11` §5.9) the apps come up slower and the hub already tells the
// household „restarted after an unexpected stop"; the 2026-10-04 crash-guard test on demo-hp then
// mailed `app_start_failed` and `app_stopped_unhealthy` 3.5 and 9 minutes after the boot, for apps
// that were still coming up. So after a crash boot the wait is CrashGrace (about 15 minutes).
//
// THE FACT comes from the host's crash guard (`felhom-crash-guard`, state.json `last_boot_unclean` +
// `last_boot_at`), through the agent's local API. The guest cannot see the host's state file, so it
// is a Probe seam.
//
// UNKNOWN IS A NORMAL BOOT. An agent that predates the route (404), an unreachable agent, a box with
// no crash guard, an unparseable time: all keep today's 90 s. A longer silence must never come from
// a guess — it would hide a real outage for 15 minutes on every box whose fact we cannot read.
package crashboot
import (
"context"
"log"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/agentapi"
)
const (
// NormalGrace is the dead-app boot grace after an ordinary start (the pre-R-856 deadAppBootGrace).
NormalGrace = 90 * time.Second
// CrashGrace is the wait after a crash boot — „about 15 minutes" (decision 143). The 2026-10-04
// mails came at 3.5 and 9 minutes; 15 covers both with room.
CrashGrace = 15 * time.Minute
// BootWindow bounds how recent the host's unclean boot must be for THIS controller start to be the
// one that followed it. A controller restarted days after a crash (a self-update, a kill) is a
// normal start; a guest starts its controller well within this window after a host boot.
BootWindow = 30 * time.Minute
probeTimeout = 5 * time.Second
)
// Fact is what the host's crash guard says about its most recent boot.
type Fact struct {
Known bool // false: no crash guard, no state, or an agent that cannot say
Unclean bool // the most recent host boot followed an unclean stop
BootAt time.Time // when that boot happened
}
// Probe asks the host for the fact. An error means unknown.
type Probe func(ctx context.Context) (Fact, error)
// Grace answers "must the app mails still wait?" for one controller run.
type Grace struct {
start time.Time
probe Probe
logger *log.Logger
mu sync.Mutex
resolved bool
crashBoot bool
}
// New builds the grace for a controller that started at start. probe may be nil (no agent): unknown.
func New(start time.Time, probe Probe, logger *log.Logger) *Grace {
return &Grace{start: start, probe: probe, logger: logger}
}
// Within reports whether now is still inside the boot grace, i.e. the app mails must wait.
//
// The fact is asked for while the normal grace runs (the agent may come up a few seconds after the
// controller) and once more at its end; whatever is known then decides, and unknown resolves to a
// normal boot. Once resolved it is never asked again.
func (g *Grace) Within(now time.Time) bool {
elapsed := now.Sub(g.start)
return elapsed < g.duration(elapsed)
}
// Duration is the grace this run uses (resolving it if it can). For logs and tests.
func (g *Grace) Duration(now time.Time) time.Duration {
return g.duration(now.Sub(g.start))
}
func (g *Grace) duration(elapsed time.Duration) time.Duration {
g.mu.Lock()
defer g.mu.Unlock()
if !g.resolved {
g.tryResolve(elapsed >= NormalGrace)
}
if g.crashBoot {
return CrashGrace
}
return NormalGrace
}
// tryResolve asks the probe once. final: the normal grace is over, so an unknown answer is final too.
func (g *Grace) tryResolve(final bool) {
if g.probe == nil {
g.resolve(false, "no agent to ask — a normal boot")
return
}
ctx, cancel := context.WithTimeout(context.Background(), probeTimeout)
f, err := g.probe(ctx)
cancel()
switch {
case err != nil || !f.Known:
if final {
why := "the host's crash guard has no record"
if err != nil {
why = "the host's crash guard could not be read (" + err.Error() + ")"
}
g.resolve(false, why+" — treated as a normal boot")
}
case !f.Unclean:
g.resolve(false, "the host's last boot was clean")
case f.BootAt.IsZero() || g.start.Sub(f.BootAt) > BootWindow:
g.resolve(false, "the host's last unclean boot ("+f.BootAt.UTC().Format(time.RFC3339)+") is not the one this start followed")
default:
g.crashBoot = true
g.resolve(true, "the host's last boot ("+f.BootAt.UTC().Format(time.RFC3339)+") followed an UNCLEAN stop")
}
}
func (g *Grace) resolve(crash bool, why string) {
g.resolved = true
if g.logger == nil {
return
}
d := NormalGrace
if crash {
d = CrashGrace
}
g.logger.Printf("[INFO] [deadapp] boot grace %s: %s (R-856)", d, why)
}
// AgentProbe adapts the agent's GET /host/crash-guard (agentapi.Client.CrashGuard) to a Probe. A
// state the host has not written (Present=false) is unknown; an unparseable boot time is a known
// unclean boot with no time, which New's window check reads as NOT this start's — a normal boot.
func AgentProbe(get func(ctx context.Context) (agentapi.CrashGuardState, error)) Probe {
return func(ctx context.Context) (Fact, error) {
st, err := get(ctx)
if err != nil {
return Fact{}, err
}
if !st.Present {
return Fact{}, nil
}
f := Fact{Known: true, Unclean: st.LastBootUnclean}
if t, perr := time.Parse(time.RFC3339, st.LastBootAt); perr == nil {
f.BootAt = t
}
return f, nil
}
}
@@ -0,0 +1,106 @@
package crashboot
import (
"bytes"
"context"
"errors"
"log"
"strings"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/agentapi"
)
// R-856 (`09` §3 decision 143): after a CRASH boot the app mails wait about 15 minutes; a normal boot
// keeps 90 s; unknown is a normal boot.
//
// COMPANION RED-PROOF: in tryResolve's default branch, leave g.crashBoot false — the crash-boot test
// then reads Within(start+10m) = false (mails would go out at 10 minutes), and fails.
var start = time.Date(2026, 10, 4, 12, 0, 0, 0, time.UTC)
func fixed(f Fact, err error) Probe {
return func(context.Context) (Fact, error) { return f, err }
}
// The 2026-10-04 shape: the host crashed, came back, and the controller started ~1 minute later. The
// mails came at 3.5 and 9 minutes after the boot — both must now be inside the grace.
func TestR856_CrashBootHoldsTheMailsForTheLongGrace(t *testing.T) {
var logs bytes.Buffer
g := New(start, fixed(Fact{Known: true, Unclean: true, BootAt: start.Add(-time.Minute)}, nil), log.New(&logs, "", 0))
for _, at := range []time.Duration{30 * time.Second, 100 * time.Second, 210 * time.Second, 9 * time.Minute, 14 * time.Minute} {
if !g.Within(start.Add(at)) {
t.Errorf("crash boot: at +%s the mails must still wait (grace %s)", at, g.Duration(start.Add(at)))
}
}
if g.Within(start.Add(CrashGrace + time.Second)) {
t.Error("crash boot: after the long grace the mails must flow again")
}
if !strings.Contains(logs.String(), "boot grace 15m0s") || !strings.Contains(logs.String(), "UNCLEAN") {
t.Errorf("the decision must be logged once, positively; log:\n%s", logs.String())
}
}
func TestR856_NormalBootKeeps90s(t *testing.T) {
for name, p := range map[string]Probe{
"clean boot": fixed(Fact{Known: true, Unclean: false, BootAt: start.Add(-time.Minute)}, nil),
"old unclean boot": fixed(Fact{Known: true, Unclean: true, BootAt: start.Add(-48 * time.Hour)}, nil),
"unclean, no time": fixed(Fact{Known: true, Unclean: true}, nil),
"agent predates it": fixed(Fact{}, errors.New("agentapi: GET /host/crash-guard: HTTP 404")),
"no crash guard": fixed(Fact{Known: false}, nil),
"no agent (nil)": nil,
} {
g := New(start, p, nil)
if !g.Within(start.Add(30 * time.Second)) {
t.Errorf("%s: inside the normal grace the mails wait", name)
}
if g.Within(start.Add(NormalGrace + time.Second)) {
t.Errorf("%s: a normal or unknown boot must keep today's 90 s grace, got %s", name, g.Duration(start.Add(NormalGrace+time.Second)))
}
}
}
// The agent comes up a few seconds after the controller: an error early in the normal grace is not
// final; the fact read later in it still decides.
func TestR856_FactReadLateInTheNormalGraceStillCounts(t *testing.T) {
calls := 0
g := New(start, func(context.Context) (Fact, error) {
calls++
if calls == 1 {
return Fact{}, errors.New("connection refused")
}
return Fact{Known: true, Unclean: true, BootAt: start.Add(-2 * time.Minute)}, nil
}, nil)
g.Within(start.Add(30 * time.Second))
if !g.Within(start.Add(5 * time.Minute)) {
t.Fatal("a crash boot learned on the second ask must still hold the mails")
}
g.Within(start.Add(6 * time.Minute))
if calls != 2 {
t.Errorf("once resolved the fact is never asked again, asked %d times", calls)
}
}
// The agent's answer, end to end through the adapter: the state.json shape the crash guard writes.
func TestR856_AgentProbeReadsTheCrashGuardState(t *testing.T) {
boot := start.Add(-time.Minute).Format("2006-01-02T15:04:05Z")
g := New(start, AgentProbe(func(context.Context) (agentapi.CrashGuardState, error) {
return agentapi.CrashGuardState{Present: true, LastBootAt: boot, LastBootUnclean: true}, nil
}), nil)
if !g.Within(start.Add(10 * time.Minute)) {
t.Error("an unclean boot read through the agent must hold the mails")
}
g = New(start, AgentProbe(func(context.Context) (agentapi.CrashGuardState, error) {
return agentapi.CrashGuardState{Present: false, LastBootUnclean: true}, nil
}), nil)
if g.Within(start.Add(NormalGrace + time.Second)) {
t.Error("a host with no crash-guard state is unknown — a normal boot")
}
g = New(start, AgentProbe(func(context.Context) (agentapi.CrashGuardState, error) {
return agentapi.CrashGuardState{}, &agentapi.StatusError{Path: "/host/crash-guard", Code: 404}
}), nil)
if g.Within(start.Add(NormalGrace + time.Second)) {
t.Error("an agent that predates the route (404) is unknown — a normal boot")
}
}