Files
felhom.eu/hub/internal/monitor/controller_supervisor.go
T
admin 37ae31fd44
gates / gates (push) Successful in 21s
hub v0.117.0: the status follows the configured threshold; slow crash loop and interrupted restore events
R-549 (operator ruling A): controllerStatus hardcoded 30m/1h while both
staleness checkers and hostStatus read alerting.stale_threshold. Moving the
threshold to 45m would have painted a customer amber 15 minutes before the
alarm could fire - the second definition rollup.go's header forbids. It now
reads the same value, down at 2x. Both 'checker initialized' log lines print
the threshold, which no line did before.

R-539 (ruling 3 of 2026-09-16): controller_slow_crashloop (warning,
operator-only), minted when the agent's slow_crashloop_since moves, with the
fast sibling's first-sight rule.

R-550: restore_interrupted (warning, for the household) allowlisted with a
Hungarian customer message.

Red-proofs, each seen failing then passing: the status test with the old
hardcoded numbers; the checker test with the movement branch removed; the
operator-only test with the registration removed; the household-message test
with the Hungarian entry removed (asserted on the SUBJECT - the body
legitimately repeats the raw message, which my first version of the test
mistook for a fallback).

go build/vet/test ./... green, 18 packages.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-17 10:20:32 +02:00

156 lines
6.6 KiB
Go

package monitor
import (
"encoding/json"
"fmt"
"log"
"sync"
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
)
// R-523 (felhom-agent v0.131.0). The agent's in-guest controller supervisor restarts a controller
// container that is not running and, after 3 restarts in 15 minutes, gives up for 30 minutes. The
// agent has no event channel of its own — its heartbeat is the channel — so the per-guest record
// rides the host report as `controller_supervisor`, and this checker turns movement in it into events:
//
// - `controller_restarted_by_agent` (info) when a guest's last_restart_at MOVES to a new value;
// - `controller_crashloop` (error, operator-only) when a guest's crashloop_since MOVES;
// - `controller_slow_crashloop` (warning, operator-only) when a guest's slow_crashloop_since MOVES
// (R-539, operator ruling 3 of 2026-09-16, agent v0.132.0): five restarts inside 24 hours, the
// loop the 15-minute brake cannot see because each restart falls outside its window.
//
// TIMESTAMPS, NOT COUNTERS. The agent's record is in-memory, so an agent restart zeroes
// restarts_total; keying on a counter would read that as nothing (fine) but a later 1 would not
// exceed the remembered 3 (a lost event). A timestamp that changes is a new act whatever the counter
// says.
//
// First observation (hub restart, new host): SEED SILENTLY, except a guest that is IN a crash-loop at
// first sight emits once — a hub restarted during an outage must not stay silent about it (the
// HostCapabilityChecker F2 rule). Both types are pinned in allowedEventTypes and operatorOnlyEvents
// by controller_supervisor_event_test.go in the api package.
type ControllerSupervisorChecker struct {
store *store.Store
logger *log.Logger
onEvent EventNotifyFunc
mu sync.Mutex
seen map[string]supSeen // hostID/vmid → last timestamps seen
}
type supSeen struct {
lastRestartAt string
crashloopSince string
slowCrashloopSince string
}
// ControllerSupervisorGuest mirrors the agent's hub.ControllerSupervisorGuest wire shape.
type ControllerSupervisorGuest struct {
VMID int `json:"vmid"`
RestartsTotal int `json:"restarts_total"`
LastRestartAt string `json:"last_restart_at"`
LastReason string `json:"last_reason"`
Crashloop bool `json:"crashloop"`
CrashloopSince string `json:"crashloop_since"`
Parked bool `json:"parked"`
// R-539 (agent v0.132.0). Absent on older agents → zero values → never an event.
Restarts24h int `json:"restarts_24h"`
SlowCrashloop bool `json:"slow_crashloop"`
SlowCrashloopSince string `json:"slow_crashloop_since"`
}
// ParseControllerSupervisor extracts the stanza from a host report body. A missing or malformed
// stanza (a pre-v0.131.0 agent) yields nil — never an event.
func ParseControllerSupervisor(reportJSON string) []ControllerSupervisorGuest {
var body struct {
CS *struct {
Guests []ControllerSupervisorGuest `json:"guests"`
} `json:"controller_supervisor"`
}
if err := json.Unmarshal([]byte(reportJSON), &body); err != nil || body.CS == nil {
return nil
}
return body.CS.Guests
}
func NewControllerSupervisorChecker(s *store.Store, onEvent EventNotifyFunc, logger *log.Logger) *ControllerSupervisorChecker {
return &ControllerSupervisorChecker{store: s, logger: logger, onEvent: onEvent, seen: map[string]supSeen{}}
}
// Check reads every host's latest report and emits on movement. Same 60 s sweep as its siblings.
func (c *ControllerSupervisorChecker) Check() {
rows, err := c.store.GetLatestHostReports()
if err != nil {
c.logger.Printf("[WARN] Controller supervisor check failed: %v", err)
return
}
for _, row := range rows {
if c.store.IsCustomerBlocked(row.CustomerID) {
continue
}
c.observe(row.HostID, row.CustomerID, ParseControllerSupervisor(row.ReportJSON))
}
}
func (c *ControllerSupervisorChecker) observe(hostID, customerID string, guests []ControllerSupervisorGuest) {
c.mu.Lock()
defer c.mu.Unlock()
for _, g := range guests {
key := fmt.Sprintf("%s/%d", hostID, g.VMID)
prev, known := c.seen[key]
c.seen[key] = supSeen{lastRestartAt: g.LastRestartAt, crashloopSince: g.CrashloopSince, slowCrashloopSince: g.SlowCrashloopSince}
if !known {
if g.Crashloop && g.CrashloopSince != "" {
c.emit(customerID, hostID, g, "controller_crashloop")
}
if g.SlowCrashloop && g.SlowCrashloopSince != "" {
c.emit(customerID, hostID, g, "controller_slow_crashloop")
}
continue
}
if g.LastRestartAt != "" && g.LastRestartAt != prev.lastRestartAt {
c.emit(customerID, hostID, g, "controller_restarted_by_agent")
}
if g.CrashloopSince != "" && g.CrashloopSince != prev.crashloopSince {
c.emit(customerID, hostID, g, "controller_crashloop")
}
if g.SlowCrashloopSince != "" && g.SlowCrashloopSince != prev.slowCrashloopSince {
c.emit(customerID, hostID, g, "controller_slow_crashloop")
}
}
}
func (c *ControllerSupervisorChecker) emit(customerID, hostID string, g ControllerSupervisorGuest, eventType string) {
var severity, message string
switch eventType {
case "controller_restarted_by_agent":
severity = "info"
message = fmt.Sprintf("Host %s guest %d: the agent restarted the controller (%s) — restart #%d since the agent started",
hostID, g.VMID, g.LastReason, g.RestartsTotal)
case "controller_crashloop":
severity = "error"
message = fmt.Sprintf("Host %s guest %d: the controller will not stay up — the agent stopped restarting it after repeated attempts (since %s) and will try again in 30 minutes. Last reason: %s",
hostID, g.VMID, g.CrashloopSince, g.LastReason)
case "controller_slow_crashloop":
severity = "warning"
message = fmt.Sprintf("Host %s guest %d: the controller keeps dying — the agent restarted it %d times in 24 hours (each one too far apart for the 15-minute brake). It is still being restarted; this is the warning that it will not stay up. Last reason: %s",
hostID, g.VMID, g.Restarts24h, g.LastReason)
default:
return
}
details, _ := json.Marshal(map[string]any{
"host_id": hostID, "vmid": g.VMID, "restarts_total": g.RestartsTotal,
"last_restart_at": g.LastRestartAt, "last_reason": g.LastReason,
"crashloop_since": g.CrashloopSince, "parked": g.Parked,
"restarts_24h": g.Restarts24h, "slow_crashloop_since": g.SlowCrashloopSince,
})
c.logger.Printf("[INFO] Controller supervisor: %s %s/%d (%s)", eventType, hostID, g.VMID, g.LastReason)
if _, err := c.store.SaveEvent(customerID, eventType, severity, message, string(details), "hub"); err != nil {
c.logger.Printf("[WARN] save %s for %s: %v", eventType, hostID, err)
return
}
if c.onEvent != nil {
c.onEvent(customerID, eventType, severity, message, string(details), "hub")
}
}