From 37ae31fd44754e15f763ba630a29c0664c6c3978 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Thu, 17 Sep 2026 10:20:32 +0200 Subject: [PATCH] hub v0.117.0: the status follows the configured threshold; slow crash loop and interrupted restore events R-549 (operator ruling A): controllerStatus hardcoded 30m/1h while both staleness checkers and hostStatus read alerting.stale_threshold. Moving the threshold to 45m would have painted a customer amber 15 minutes before the alarm could fire - the second definition rollup.go's header forbids. It now reads the same value, down at 2x. Both 'checker initialized' log lines print the threshold, which no line did before. R-539 (ruling 3 of 2026-09-16): controller_slow_crashloop (warning, operator-only), minted when the agent's slow_crashloop_since moves, with the fast sibling's first-sight rule. R-550: restore_interrupted (warning, for the household) allowlisted with a Hungarian customer message. Red-proofs, each seen failing then passing: the status test with the old hardcoded numbers; the checker test with the movement branch removed; the operator-only test with the registration removed; the household-message test with the Hungarian entry removed (asserted on the SUBJECT - the body legitimately repeats the raw message, which my first version of the test mistook for a fallback). go build/vet/test ./... green, 18 packages. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- .claude/rules/hub.md | 5 +- .../architecture/05-hub-architecture.md | 4 +- hub/CHANGELOG.md | 23 ++++++++ hub/internal/api/chaosnight_events_test.go | 49 +++++++++++++++++ hub/internal/api/handler.go | 5 ++ hub/internal/monitor/controller_supervisor.go | 27 ++++++++-- .../monitor/controller_supervisor_test.go | 53 +++++++++++++++++++ hub/internal/monitor/host_staleness.go | 2 +- hub/internal/monitor/staleness.go | 2 +- hub/internal/notify/dispatcher.go | 2 + hub/internal/notify/templates.go | 2 + hub/internal/web/configs.go | 4 +- hub/internal/web/rollup.go | 14 +++-- hub/internal/web/rollup_test.go | 31 +++++++++++ hub/internal/web/server.go | 2 +- 15 files changed, 210 insertions(+), 15 deletions(-) create mode 100644 hub/internal/api/chaosnight_events_test.go diff --git a/.claude/rules/hub.md b/.claude/rules/hub.md index 6407ba80..27707b06 100644 --- a/.claude/rules/hub.md +++ b/.claude/rules/hub.md @@ -51,7 +51,10 @@ cannot be read off the code: writing the filesystem. - New event types must enter `allowedEventTypes` **and** `customerMessages` together, or `POST /event` 400s. -- Status logic: OK (report < 30m), WARN (30m–1h or `health=warn`), DOWN (> 1h or `health=fail`). +- Status logic: OK (report younger than `alerting.stale_threshold`), WARN (past the threshold or + `health=warn`), DOWN (past 2× the threshold or `health=fail`). The threshold is **configuration** + (`manifests/hub.yaml`; 45 m by operator ruling 2026-09-17, R-549), and the display + (`controllerStatus`, `hostStatus`) and both checkers read the same value — never hardcode it. Host-liveness thresholds are **shared** between UI and checker — never invent a second definition. - SQLite timestamps vary in format — always `parseSQLiteTime()`. - **Logging**: DEBUG = flow detail, INFO = state change + duration; operator English; keys never diff --git a/documentation/architecture/05-hub-architecture.md b/documentation/architecture/05-hub-architecture.md index ff4d50f0..93cd1814 100644 --- a/documentation/architecture/05-hub-architecture.md +++ b/documentation/architecture/05-hub-architecture.md @@ -85,8 +85,8 @@ These two streams are the bottom-up mirror of §1 — they keep the hub current ## 4. Liveness / dead-man's-switch -Evolves the existing staleness checker (60s **cadence**, 30m/1h **thresholds** — OK <30m, down at -2× = >1h; today: controller-report recency → `node_stale`/`down`/`recovered`): +Evolves the existing staleness checker (60s **cadence**, a **configured** threshold — 30 m until +2026-09-17, **45 m** since operator ruling A on R-549; OK under it, down at 2× = 90 m; today: controller-report recency → `node_stale`/`down`/`recovered`): - **Primary = host-report recency → `host_stale` / `host_down`.** The agent heartbeat is the box's liveness signal; a silent agent = the box is gone (the critical alert). diff --git a/hub/CHANGELOG.md b/hub/CHANGELOG.md index 34371866..7958a8d9 100644 --- a/hub/CHANGELOG.md +++ b/hub/CHANGELOG.md @@ -1,3 +1,26 @@ +## v0.117.0 — the quiet-box alarm waits three report cycles, and two new events (2026-09-17, R-549 / R-550 / R-539) + +- **The dashboard's customer status now follows the configured staleness threshold** (R-549, operator + ruling A). `controllerStatus` hardcoded 30 m / 1 h while both staleness checkers and the host status + read `alerting.stale_threshold`. With the ruling moving the threshold to **45 m** (manifest, same + release) the customer row would have turned amber at 30 m and red at 60 m while the alarms wait + until 45 m and 90 m — the second definition `rollup.go`'s own header forbids. It now reads the same + value, down at 2×. Red-proof: `TestControllerStatus_FollowsConfiguredThreshold` fails with the old + hardcoded numbers ("report 40m old: status warn, want ok"). +- **The threshold is now visible at startup.** Both "checker initialized" lines print it + (`node_stale after 45m0s, node_down after 1h30m0s`); before this, no log line showed which value a + running hub was using. +- **`controller_slow_crashloop`** (warning, operator-only; R-539, operator ruling 3 of 2026-09-16). + The controller-supervisor checker reads the agent's new `restarts_24h` / `slow_crashloop` / + `slow_crashloop_since` (agent v0.132.0) and emits when `slow_crashloop_since` MOVES — the same + timestamps-not-counters rule as its fast sibling, and the same first-sight rule (a hub restarted + during a slow loop says so once). Older agents send none of the fields → never an event. +- **`restore_interrupted`** (warning, for the household; R-550). Pushed by controller v0.246.0 when it + starts and finds a restore that was running when the box stopped. Allowlisted, **not** operator-only, + with a Hungarian `customerMessages` entry — pinned by a test that reads the customer mail's subject. + +**Requires:** nothing new. Old agents and controllers send none of the new fields or events. + ## v0.116.0 — every new customer starts WITH the off-site copy (2026-09-16, operator ruling; R-536 event pair) - **Off-site backup is ON by default for a new customer** — shared (a sub-account on the pool box), diff --git a/hub/internal/api/chaosnight_events_test.go b/hub/internal/api/chaosnight_events_test.go new file mode 100644 index 00000000..ea0da4a8 --- /dev/null +++ b/hub/internal/api/chaosnight_events_test.go @@ -0,0 +1,49 @@ +package api + +import ( + "strings" + "testing" + + "gitea.dooplex.hu/admin/felhom-hub/internal/notify" +) + +// R-539 — controller_slow_crashloop joins its fast sibling: allowlisted AND operator-only. +// RED-PROOF: remove it from operatorOnlyEvents → "must be operator-only". +func TestControllerSlowCrashloopIsAllowlistedAndOperatorOnly(t *testing.T) { + et := "controller_slow_crashloop" + if !allowedEventTypes[et] { + t.Fatalf("%s must be in allowedEventTypes (R-77)", et) + } + if !notify.IsOperatorOnly(et) { + t.Fatalf("%s must be operator-only — host ids and vmids are not a household's business", et) + } +} + +// R-550 — restore_interrupted is pushed by controller v0.246.0 and IS for the household: they started +// the restore and can run it again. So it must be allowlisted (or POST /event 400s — R-77), must NOT be +// operator-only, and must carry a Hungarian customer message (a missing entry falls back to the raw +// English message — the v0.78.0 defect). +// RED-PROOF: drop the customerMessages entry → the subject/body carry the raw message. +func TestRestoreInterruptedReachesTheHouseholdInHungarian(t *testing.T) { + et := "restore_interrupted" + if !allowedEventTypes[et] { + t.Fatalf("%s must be in allowedEventTypes — the controller's push would 400", et) + } + if notify.IsOperatorOnly(et) { + t.Fatalf("%s must reach the household, not only the operator", et) + } + const raw = "RAW-ENGLISH-SENTINEL restore interrupted" + // The SUBJECT is built from the Hungarian message alone; the body legitimately repeats the raw + // message as a detail line ("Üzenet: …"), so the body cannot be the witness — the subject is. + subject, body := notify.FormatCustomerEmail("c1", et, "warning", raw, "{}") + if strings.Contains(subject, "RAW-ENGLISH-SENTINEL") { + t.Fatalf("customer mail subject for %s is the raw message — customerMessages entry missing: %q", et, subject) + } + // ASCII fragment of the Hungarian text („megszakadt"), with a negative control. + if !strings.Contains(body, "megszakadt") { + t.Fatalf("customer mail for %s lacks the Hungarian sentence (fragment \"megszakadt\")", et) + } + if strings.Contains(body, "zzzz-not-present") { + t.Fatalf("negative control matched — the fragment search is broken") + } +} diff --git a/hub/internal/api/handler.go b/hub/internal/api/handler.go index 6bd8340f..1924d163 100644 --- a/hub/internal/api/handler.go +++ b/hub/internal/api/handler.go @@ -2121,6 +2121,11 @@ var allowedEventTypes = map[string]bool{ // report stanza (monitor/controller_supervisor.go). Both operator-only (notify.operatorOnlyEvents). "controller_restarted_by_agent": true, "controller_crashloop": true, + // R-539 (hub v0.117.0, agent v0.132.0) — hub-generated from slow_crashloop_since; operator-only. + "controller_slow_crashloop": true, + // R-550 (hub v0.117.0, controller v0.246.0) — pushed by the controller at startup when it finds a + // restore that was in flight when the box stopped. For the household: they can run it again. + "restore_interrupted": true, // R-509 / R-511 / R-518 / R-514 (hub v0.114.0). selfbind_link_sent and pbsdr_adopted are // hub-internal audit rows (stored, not dispatched). backup_tier_skipped and app_oom are pushed by // controller v0.243.0; both operator-only (notify.operatorOnlyEvents). diff --git a/hub/internal/monitor/controller_supervisor.go b/hub/internal/monitor/controller_supervisor.go index ec83754d..e28a22a8 100644 --- a/hub/internal/monitor/controller_supervisor.go +++ b/hub/internal/monitor/controller_supervisor.go @@ -15,7 +15,10 @@ import ( // rides the host report as `controller_supervisor`, and this checker turns movement in it into events: // // - `controller_restarted_by_agent` (info) when a guest's last_restart_at MOVES to a new value; -// - `controller_crashloop` (error, operator-only) when a guest's crashloop_since MOVES. +// - `controller_crashloop` (error, operator-only) when a guest's crashloop_since MOVES; +// - `controller_slow_crashloop` (warning, operator-only) when a guest's slow_crashloop_since MOVES +// (R-539, operator ruling 3 of 2026-09-16, agent v0.132.0): five restarts inside 24 hours, the +// loop the 15-minute brake cannot see because each restart falls outside its window. // // TIMESTAMPS, NOT COUNTERS. The agent's record is in-memory, so an agent restart zeroes // restarts_total; keying on a counter would read that as nothing (fine) but a later 1 would not @@ -36,8 +39,9 @@ type ControllerSupervisorChecker struct { } type supSeen struct { - lastRestartAt string - crashloopSince string + lastRestartAt string + crashloopSince string + slowCrashloopSince string } // ControllerSupervisorGuest mirrors the agent's hub.ControllerSupervisorGuest wire shape. @@ -49,6 +53,10 @@ type ControllerSupervisorGuest struct { Crashloop bool `json:"crashloop"` CrashloopSince string `json:"crashloop_since"` Parked bool `json:"parked"` + // R-539 (agent v0.132.0). Absent on older agents → zero values → never an event. + Restarts24h int `json:"restarts_24h"` + SlowCrashloop bool `json:"slow_crashloop"` + SlowCrashloopSince string `json:"slow_crashloop_since"` } // ParseControllerSupervisor extracts the stanza from a host report body. A missing or malformed @@ -90,11 +98,14 @@ func (c *ControllerSupervisorChecker) observe(hostID, customerID string, guests for _, g := range guests { key := fmt.Sprintf("%s/%d", hostID, g.VMID) prev, known := c.seen[key] - c.seen[key] = supSeen{lastRestartAt: g.LastRestartAt, crashloopSince: g.CrashloopSince} + c.seen[key] = supSeen{lastRestartAt: g.LastRestartAt, crashloopSince: g.CrashloopSince, slowCrashloopSince: g.SlowCrashloopSince} if !known { if g.Crashloop && g.CrashloopSince != "" { c.emit(customerID, hostID, g, "controller_crashloop") } + if g.SlowCrashloop && g.SlowCrashloopSince != "" { + c.emit(customerID, hostID, g, "controller_slow_crashloop") + } continue } if g.LastRestartAt != "" && g.LastRestartAt != prev.lastRestartAt { @@ -103,6 +114,9 @@ func (c *ControllerSupervisorChecker) observe(hostID, customerID string, guests if g.CrashloopSince != "" && g.CrashloopSince != prev.crashloopSince { c.emit(customerID, hostID, g, "controller_crashloop") } + if g.SlowCrashloopSince != "" && g.SlowCrashloopSince != prev.slowCrashloopSince { + c.emit(customerID, hostID, g, "controller_slow_crashloop") + } } } @@ -117,6 +131,10 @@ func (c *ControllerSupervisorChecker) emit(customerID, hostID string, g Controll severity = "error" message = fmt.Sprintf("Host %s guest %d: the controller will not stay up — the agent stopped restarting it after repeated attempts (since %s) and will try again in 30 minutes. Last reason: %s", hostID, g.VMID, g.CrashloopSince, g.LastReason) + case "controller_slow_crashloop": + severity = "warning" + message = fmt.Sprintf("Host %s guest %d: the controller keeps dying — the agent restarted it %d times in 24 hours (each one too far apart for the 15-minute brake). It is still being restarted; this is the warning that it will not stay up. Last reason: %s", + hostID, g.VMID, g.Restarts24h, g.LastReason) default: return } @@ -124,6 +142,7 @@ func (c *ControllerSupervisorChecker) emit(customerID, hostID string, g Controll "host_id": hostID, "vmid": g.VMID, "restarts_total": g.RestartsTotal, "last_restart_at": g.LastRestartAt, "last_reason": g.LastReason, "crashloop_since": g.CrashloopSince, "parked": g.Parked, + "restarts_24h": g.Restarts24h, "slow_crashloop_since": g.SlowCrashloopSince, }) c.logger.Printf("[INFO] Controller supervisor: %s %s/%d (%s)", eventType, hostID, g.VMID, g.LastReason) if _, err := c.store.SaveEvent(customerID, eventType, severity, message, string(details), "hub"); err != nil { diff --git a/hub/internal/monitor/controller_supervisor_test.go b/hub/internal/monitor/controller_supervisor_test.go index ab96e4a4..2585be40 100644 --- a/hub/internal/monitor/controller_supervisor_test.go +++ b/hub/internal/monitor/controller_supervisor_test.go @@ -93,3 +93,56 @@ func TestControllerSupervisorChecker_CrashloopAtFirstSightEmits(t *testing.T) { t.Fatalf("crash-loop at first sight: want exactly one controller_crashloop, got %v", evs) } } + +// R-539 (operator ruling 3, 2026-09-16; agent v0.132.0). A SLOW crash loop — restarts spread wider than +// the 15-minute brake — is reported by the agent as slow_crashloop_since. The consequence: when that +// timestamp MOVES, the dispatcher receives controller_slow_crashloop at WARNING; a guest that never +// sets it never produces one. +// +// RED-PROOF: delete the SlowCrashloopSince movement branch in observe() → "slow crash loop produced no +// controller_slow_crashloop". +func TestControllerSupervisorChecker_SlowCrashloop(t *testing.T) { + st := newCapStore(t) + rec := &evRec{} + c := NewControllerSupervisorChecker(st, rec.fn, log.New(io.Discard, "", 0)) + + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":4,"last_restart_at":"2026-09-17T08:00:00Z","last_reason":"exited","crashloop":false,"parked":false,"restarts_24h":4}]`), store.HostReportDenorm{}) + c.Check() + if evs := rec.take(); len(evs) != 0 { + t.Fatalf("seed must be silent, got %v", evs) + } + // Fifth restart inside 24 h: last_restart_at moves AND slow_crashloop_since appears. + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":5,"last_restart_at":"2026-09-17T08:20:00Z","last_reason":"exited","crashloop":false,"parked":false,"restarts_24h":5,"slow_crashloop":true,"slow_crashloop_since":"2026-09-17T08:20:00Z"}]`), store.HostReportDenorm{}) + c.Check() + evs := rec.take() + found := false + for _, e := range evs { + if e == "controller_slow_crashloop/warning" { + found = true + } + } + if !found { + t.Fatalf("slow crash loop produced no controller_slow_crashloop/warning; got %v", evs) + } + // Same timestamp on the next sweep → no second event. + c.Check() + for _, e := range rec.take() { + if e == "controller_slow_crashloop/warning" { + t.Fatalf("an unmoved slow_crashloop_since must not re-emit") + } + } +} + +// A hub restarted while a guest is in a slow crash loop must not stay silent (the F2 rule the fast +// crash loop already follows): first sight with slow_crashloop=true emits once. +func TestControllerSupervisorChecker_SlowCrashloopAtFirstSight(t *testing.T) { + st := newCapStore(t) + rec := &evRec{} + c := NewControllerSupervisorChecker(st, rec.fn, log.New(io.Discard, "", 0)) + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":5,"last_restart_at":"2026-09-17T08:20:00Z","last_reason":"exited","crashloop":false,"parked":false,"restarts_24h":5,"slow_crashloop":true,"slow_crashloop_since":"2026-09-17T08:20:00Z"}]`), store.HostReportDenorm{}) + c.Check() + evs := rec.take() + if len(evs) != 1 || evs[0] != "controller_slow_crashloop/warning" { + t.Fatalf("first sight in a slow crash loop: got %v, want exactly [controller_slow_crashloop/warning]", evs) + } +} diff --git a/hub/internal/monitor/host_staleness.go b/hub/internal/monitor/host_staleness.go index 8de236b6..b26598e9 100644 --- a/hub/internal/monitor/host_staleness.go +++ b/hub/internal/monitor/host_staleness.go @@ -69,7 +69,7 @@ func NewHostStalenessChecker(s *store.Store, threshold time.Duration, onEvent Ev okCount++ } } - logger.Printf("[INFO] Host staleness checker initialized: %d ok, %d stale, %d down (stale/down left unseeded → first Check emits)", okCount, staleCount, downCount) + logger.Printf("[INFO] Host staleness checker initialized: %d ok, %d stale, %d down (stale/down left unseeded → first Check emits; host_stale after %s, host_down after %s)", okCount, staleCount, downCount, threshold, 2*threshold) return sc } diff --git a/hub/internal/monitor/staleness.go b/hub/internal/monitor/staleness.go index a0037442..29034161 100644 --- a/hub/internal/monitor/staleness.go +++ b/hub/internal/monitor/staleness.go @@ -88,7 +88,7 @@ func NewStalenessChecker(s *store.Store, threshold time.Duration, onEvent EventN okCount++ } } - logger.Printf("[INFO] Staleness checker initialized: %d ok, %d stale, %d down", okCount, staleCount, downCount) + logger.Printf("[INFO] Staleness checker initialized: %d ok, %d stale, %d down (node_stale after %s, node_down after %s)", okCount, staleCount, downCount, threshold, 2*threshold) return sc } diff --git a/hub/internal/notify/dispatcher.go b/hub/internal/notify/dispatcher.go index a3c15912..d4f768d6 100644 --- a/hub/internal/notify/dispatcher.go +++ b/hub/internal/notify/dispatcher.go @@ -594,6 +594,8 @@ var operatorOnlyEvents = map[string]bool{ // mints the types. "controller_restarted_by_agent": true, "controller_crashloop": true, + // R-539 (v0.117.0) — the slow sibling: same audience, same reason. + "controller_slow_crashloop": true, // R-518 / R-514 (v0.114.0, controller v0.243.0). A whole-guest tier skipped for absent storage is a // provisioning fact the household cannot act on; an OOM-killed app process carries raw container // names — the household's side is the dashboard tag. Registered in the same commit. diff --git a/hub/internal/notify/templates.go b/hub/internal/notify/templates.go index aaff3040..4457cff8 100644 --- a/hub/internal/notify/templates.go +++ b/hub/internal/notify/templates.go @@ -139,6 +139,8 @@ var customerMessages = map[string]string{ "app_deployed": "Alkalmazás telepítve.", "app_removed": "Alkalmazás eltávolítva.", "app_start_failed": "Egy telepített alkalmazás nem fut — ellenőrizze a rendszermonitort.", + // R-550 (v0.117.0): the box stopped while a restore was running. + "restore_interrupted": "Egy visszaállítás megszakadt, mert a doboz újraindult. Indítsd el újra a vezérlőpult Visszaállítás oldalán.", // R-536 (controller v0.244.0): the pair that makes „telepítve" mean it. The started event is the // acceptance `app_deployed` used to assert beside the 202; the failed one is what an interrupted // install used to be — silence. A new type must enter allowedEventTypes AND this map together. diff --git a/hub/internal/web/configs.go b/hub/internal/web/configs.go index a7bf3df8..b45ea336 100644 --- a/hub/internal/web/configs.go +++ b/hub/internal/web/configs.go @@ -102,7 +102,7 @@ func (s *Server) handleConfigList(w http.ResponseWriter, r *http.Request) { for _, c := range customers { // Controller-derived status + the v0.53.0 dead-host roll-up (rollup.go). - status, hostCause := s.foldHostStatus(c.CustomerID, controllerStatus(&c), true) + status, hostCause := s.foldHostStatus(c.CustomerID, controllerStatus(&c, s.staleThreshold), true) if entry, ok := merged[c.CustomerID]; ok { // Config exists — enrich with report data @@ -212,7 +212,7 @@ func (s *Server) handleCustomerUnified(w http.ResponseWriter, r *http.Request, c // renders regardless so the header says WHICH host is the problem. overallStatus := "pending" if customer != nil { - overallStatus = controllerStatus(customer) + overallStatus = controllerStatus(customer, s.staleThreshold) } var hostCause string overallStatus, hostCause = s.foldHostStatus(customerID, overallStatus, customer != nil) diff --git a/hub/internal/web/rollup.go b/hub/internal/web/rollup.go index b82e554a..eb3e56c9 100644 --- a/hub/internal/web/rollup.go +++ b/hub/internal/web/rollup.go @@ -20,13 +20,21 @@ import ( // controllerStatus is the controller-report-derived customer status — the pre-roll-up chain the // dashboard, the /configs list and the customer detail all inlined verbatim; this is now the // ONE copy. Behavior-preserving: the branch order (incl. fail-after-warn) is the historical one. -func controllerStatus(c *store.CustomerSummary) string { +func controllerStatus(c *store.CustomerSummary, threshold time.Duration) string { + // R-549 (operator ruling A, 2026-09-17): the threshold is CONFIGURATION (alerting.stale_threshold) + // and the checkers read it; this display must read the same value, "down" at 2x like them. It + // hardcoded 30 m / 1 h until hub v0.117.0 — a second definition the header above forbids, and one + // that would have painted a customer amber 15 minutes before the alarm could fire. + // Pinned by TestControllerStatus_FollowsConfiguredThreshold. + if threshold <= 0 { + threshold = 30 * time.Minute + } switch { case c.HealthStatus == "disabled": return "disabled" - case c.TimeSinceReport > time.Hour: + case c.TimeSinceReport > 2*threshold: return "down" - case c.TimeSinceReport > 30*time.Minute || c.HealthStatus == "warn": + case c.TimeSinceReport > threshold || c.HealthStatus == "warn": return "warn" case c.HealthStatus == "fail": return "down" diff --git a/hub/internal/web/rollup_test.go b/hub/internal/web/rollup_test.go index 65789ac9..e3953936 100644 --- a/hub/internal/web/rollup_test.go +++ b/hub/internal/web/rollup_test.go @@ -192,3 +192,34 @@ func TestRollup_Boundaries(t *testing.T) { } }) } + +// R-549 (operator ruling A, 2026-09-17): the "box went quiet" threshold moved from 30 m to 45 m in +// configuration. The customer status on the dashboard must move WITH it — the file header forbids a +// second definition, and before this change controllerStatus hardcoded 30 m / 1 h while the checkers +// read the config. The consequence asserted: with the configured threshold at 45 m, a report 40 m old +// is NOT amber (the alarm has not fired either), 50 m is amber, 95 m is down (2 × threshold, the same +// multiplier the checkers use). +// +// RED-PROOF: controllerStatus ignoring its threshold (the hardcoded 30 m / 1 h) → the 40 m case +// reads "warn" and the 70 m case reads "down". +func TestControllerStatus_FollowsConfiguredThreshold(t *testing.T) { + th := 45 * time.Minute + for _, tc := range []struct { + age time.Duration + want string + }{ + {40 * time.Minute, "ok"}, + {50 * time.Minute, "warn"}, + {70 * time.Minute, "warn"}, + {95 * time.Minute, "down"}, + } { + c := &store.CustomerSummary{TimeSinceReport: tc.age, HealthStatus: "ok"} + if got := controllerStatus(c, th); got != tc.want { + t.Errorf("threshold %s, report %s old: status %q, want %q — the dashboard and the alarm disagree", th, tc.age, got, tc.want) + } + } + // A zero threshold (a Server built without one) keeps the historical default, never "everything down". + if got := controllerStatus(&store.CustomerSummary{TimeSinceReport: 10 * time.Minute, HealthStatus: "ok"}, 0); got != "ok" { + t.Errorf("zero threshold: 10m-old report reads %q, want ok", got) + } +} diff --git a/hub/internal/web/server.go b/hub/internal/web/server.go index d36d2fc4..c0b9404e 100644 --- a/hub/internal/web/server.go +++ b/hub/internal/web/server.go @@ -833,7 +833,7 @@ func (s *Server) handleDashboard(w http.ResponseWriter, r *http.Request) { // Controller-derived status + the v0.53.0 dead-host roll-up (rollup.go): a customer // may never look better than its worst expected host. - dc.OverallStatus, dc.HostCause = s.foldHostStatus(c.CustomerID, controllerStatus(&c), true) + dc.OverallStatus, dc.HostCause = s.foldHostStatus(c.CustomerID, controllerStatus(&c, s.staleThreshold), true) // Backup age if c.BackupLastSnapshot != nil {