37ae31fd44
gates / gates (push) Successful in 21s
R-549 (operator ruling A): controllerStatus hardcoded 30m/1h while both staleness checkers and hostStatus read alerting.stale_threshold. Moving the threshold to 45m would have painted a customer amber 15 minutes before the alarm could fire - the second definition rollup.go's header forbids. It now reads the same value, down at 2x. Both 'checker initialized' log lines print the threshold, which no line did before. R-539 (ruling 3 of 2026-09-16): controller_slow_crashloop (warning, operator-only), minted when the agent's slow_crashloop_since moves, with the fast sibling's first-sight rule. R-550: restore_interrupted (warning, for the household) allowlisted with a Hungarian customer message. Red-proofs, each seen failing then passing: the status test with the old hardcoded numbers; the checker test with the movement branch removed; the operator-only test with the registration removed; the household-message test with the Hungarian entry removed (asserted on the SUBJECT - the body legitimately repeats the raw message, which my first version of the test mistook for a fallback). go build/vet/test ./... green, 18 packages. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
156 lines
6.6 KiB
Go
156 lines
6.6 KiB
Go
package monitor
|
|
|
|
import (
|
|
"encoding/json"
|
|
"fmt"
|
|
"log"
|
|
"sync"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
|
)
|
|
|
|
// R-523 (felhom-agent v0.131.0). The agent's in-guest controller supervisor restarts a controller
|
|
// container that is not running and, after 3 restarts in 15 minutes, gives up for 30 minutes. The
|
|
// agent has no event channel of its own — its heartbeat is the channel — so the per-guest record
|
|
// rides the host report as `controller_supervisor`, and this checker turns movement in it into events:
|
|
//
|
|
// - `controller_restarted_by_agent` (info) when a guest's last_restart_at MOVES to a new value;
|
|
// - `controller_crashloop` (error, operator-only) when a guest's crashloop_since MOVES;
|
|
// - `controller_slow_crashloop` (warning, operator-only) when a guest's slow_crashloop_since MOVES
|
|
// (R-539, operator ruling 3 of 2026-09-16, agent v0.132.0): five restarts inside 24 hours, the
|
|
// loop the 15-minute brake cannot see because each restart falls outside its window.
|
|
//
|
|
// TIMESTAMPS, NOT COUNTERS. The agent's record is in-memory, so an agent restart zeroes
|
|
// restarts_total; keying on a counter would read that as nothing (fine) but a later 1 would not
|
|
// exceed the remembered 3 (a lost event). A timestamp that changes is a new act whatever the counter
|
|
// says.
|
|
//
|
|
// First observation (hub restart, new host): SEED SILENTLY, except a guest that is IN a crash-loop at
|
|
// first sight emits once — a hub restarted during an outage must not stay silent about it (the
|
|
// HostCapabilityChecker F2 rule). Both types are pinned in allowedEventTypes and operatorOnlyEvents
|
|
// by controller_supervisor_event_test.go in the api package.
|
|
type ControllerSupervisorChecker struct {
|
|
store *store.Store
|
|
logger *log.Logger
|
|
onEvent EventNotifyFunc
|
|
|
|
mu sync.Mutex
|
|
seen map[string]supSeen // hostID/vmid → last timestamps seen
|
|
}
|
|
|
|
type supSeen struct {
|
|
lastRestartAt string
|
|
crashloopSince string
|
|
slowCrashloopSince string
|
|
}
|
|
|
|
// ControllerSupervisorGuest mirrors the agent's hub.ControllerSupervisorGuest wire shape.
|
|
type ControllerSupervisorGuest struct {
|
|
VMID int `json:"vmid"`
|
|
RestartsTotal int `json:"restarts_total"`
|
|
LastRestartAt string `json:"last_restart_at"`
|
|
LastReason string `json:"last_reason"`
|
|
Crashloop bool `json:"crashloop"`
|
|
CrashloopSince string `json:"crashloop_since"`
|
|
Parked bool `json:"parked"`
|
|
// R-539 (agent v0.132.0). Absent on older agents → zero values → never an event.
|
|
Restarts24h int `json:"restarts_24h"`
|
|
SlowCrashloop bool `json:"slow_crashloop"`
|
|
SlowCrashloopSince string `json:"slow_crashloop_since"`
|
|
}
|
|
|
|
// ParseControllerSupervisor extracts the stanza from a host report body. A missing or malformed
|
|
// stanza (a pre-v0.131.0 agent) yields nil — never an event.
|
|
func ParseControllerSupervisor(reportJSON string) []ControllerSupervisorGuest {
|
|
var body struct {
|
|
CS *struct {
|
|
Guests []ControllerSupervisorGuest `json:"guests"`
|
|
} `json:"controller_supervisor"`
|
|
}
|
|
if err := json.Unmarshal([]byte(reportJSON), &body); err != nil || body.CS == nil {
|
|
return nil
|
|
}
|
|
return body.CS.Guests
|
|
}
|
|
|
|
func NewControllerSupervisorChecker(s *store.Store, onEvent EventNotifyFunc, logger *log.Logger) *ControllerSupervisorChecker {
|
|
return &ControllerSupervisorChecker{store: s, logger: logger, onEvent: onEvent, seen: map[string]supSeen{}}
|
|
}
|
|
|
|
// Check reads every host's latest report and emits on movement. Same 60 s sweep as its siblings.
|
|
func (c *ControllerSupervisorChecker) Check() {
|
|
rows, err := c.store.GetLatestHostReports()
|
|
if err != nil {
|
|
c.logger.Printf("[WARN] Controller supervisor check failed: %v", err)
|
|
return
|
|
}
|
|
for _, row := range rows {
|
|
if c.store.IsCustomerBlocked(row.CustomerID) {
|
|
continue
|
|
}
|
|
c.observe(row.HostID, row.CustomerID, ParseControllerSupervisor(row.ReportJSON))
|
|
}
|
|
}
|
|
|
|
func (c *ControllerSupervisorChecker) observe(hostID, customerID string, guests []ControllerSupervisorGuest) {
|
|
c.mu.Lock()
|
|
defer c.mu.Unlock()
|
|
for _, g := range guests {
|
|
key := fmt.Sprintf("%s/%d", hostID, g.VMID)
|
|
prev, known := c.seen[key]
|
|
c.seen[key] = supSeen{lastRestartAt: g.LastRestartAt, crashloopSince: g.CrashloopSince, slowCrashloopSince: g.SlowCrashloopSince}
|
|
if !known {
|
|
if g.Crashloop && g.CrashloopSince != "" {
|
|
c.emit(customerID, hostID, g, "controller_crashloop")
|
|
}
|
|
if g.SlowCrashloop && g.SlowCrashloopSince != "" {
|
|
c.emit(customerID, hostID, g, "controller_slow_crashloop")
|
|
}
|
|
continue
|
|
}
|
|
if g.LastRestartAt != "" && g.LastRestartAt != prev.lastRestartAt {
|
|
c.emit(customerID, hostID, g, "controller_restarted_by_agent")
|
|
}
|
|
if g.CrashloopSince != "" && g.CrashloopSince != prev.crashloopSince {
|
|
c.emit(customerID, hostID, g, "controller_crashloop")
|
|
}
|
|
if g.SlowCrashloopSince != "" && g.SlowCrashloopSince != prev.slowCrashloopSince {
|
|
c.emit(customerID, hostID, g, "controller_slow_crashloop")
|
|
}
|
|
}
|
|
}
|
|
|
|
func (c *ControllerSupervisorChecker) emit(customerID, hostID string, g ControllerSupervisorGuest, eventType string) {
|
|
var severity, message string
|
|
switch eventType {
|
|
case "controller_restarted_by_agent":
|
|
severity = "info"
|
|
message = fmt.Sprintf("Host %s guest %d: the agent restarted the controller (%s) — restart #%d since the agent started",
|
|
hostID, g.VMID, g.LastReason, g.RestartsTotal)
|
|
case "controller_crashloop":
|
|
severity = "error"
|
|
message = fmt.Sprintf("Host %s guest %d: the controller will not stay up — the agent stopped restarting it after repeated attempts (since %s) and will try again in 30 minutes. Last reason: %s",
|
|
hostID, g.VMID, g.CrashloopSince, g.LastReason)
|
|
case "controller_slow_crashloop":
|
|
severity = "warning"
|
|
message = fmt.Sprintf("Host %s guest %d: the controller keeps dying — the agent restarted it %d times in 24 hours (each one too far apart for the 15-minute brake). It is still being restarted; this is the warning that it will not stay up. Last reason: %s",
|
|
hostID, g.VMID, g.Restarts24h, g.LastReason)
|
|
default:
|
|
return
|
|
}
|
|
details, _ := json.Marshal(map[string]any{
|
|
"host_id": hostID, "vmid": g.VMID, "restarts_total": g.RestartsTotal,
|
|
"last_restart_at": g.LastRestartAt, "last_reason": g.LastReason,
|
|
"crashloop_since": g.CrashloopSince, "parked": g.Parked,
|
|
"restarts_24h": g.Restarts24h, "slow_crashloop_since": g.SlowCrashloopSince,
|
|
})
|
|
c.logger.Printf("[INFO] Controller supervisor: %s %s/%d (%s)", eventType, hostID, g.VMID, g.LastReason)
|
|
if _, err := c.store.SaveEvent(customerID, eventType, severity, message, string(details), "hub"); err != nil {
|
|
c.logger.Printf("[WARN] save %s for %s: %v", eventType, hostID, err)
|
|
return
|
|
}
|
|
if c.onEvent != nil {
|
|
c.onEvent(customerID, eventType, severity, message, string(details), "hub")
|
|
}
|
|
}
|