hub: F2 — alert on host already degraded/stale at (re)start (seed only healthy) v0.21.0
Constructors seed only healthy hosts; an already-degraded/stale host is left unseeded so the first Check() emits once (cooldown dedups on hub restart). Born-degraded red-proof + staleness test updated. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Pg8ANF97SEeKYSN5Jxw3qJ
This commit is contained in:
@@ -1,5 +1,21 @@
|
|||||||
# Felhom Hub — Changelog
|
# Felhom Hub — Changelog
|
||||||
|
|
||||||
|
## v0.21.0 — F2: alert on a host already degraded/stale at hub (re)start (2026-06-29)
|
||||||
|
|
||||||
|
Mirror of the controller's F2 fix, for the hub checkers: a host that was already **degraded**
|
||||||
|
(capability) or **stale/down** (staleness) when the hub (re)started was seeded silently and **never
|
||||||
|
alerted**. Now the constructors seed only HEALTHY hosts; an unhealthy host is left **unseeded** so the
|
||||||
|
first `Check()` emits once. The dispatcher's 1 h operator cooldown dedups the re-emit across a hub
|
||||||
|
restart (so a hub bounce doesn't re-page for an already-known issue within the window).
|
||||||
|
|
||||||
|
- `internal/monitor/host_capability.go` + `host_staleness.go`: constructor seeds only the healthy
|
||||||
|
state; `Check()` first-obs (`oldState==""`) emits `emitTransition(…, "unknown", newState, …)` when
|
||||||
|
the observed state isn't healthy, else seeds silently.
|
||||||
|
- Tests: born-degraded **red-proof** (host degraded at construction → unseeded → one
|
||||||
|
`agent_capability_degraded` on first Check, no duplicate on the next); the staleness test updated to
|
||||||
|
the F2 behavior (born-stale → unseeded → one `host_stale` on first Check). Version `0.20.0 → 0.21.0`.
|
||||||
|
|
||||||
|
|
||||||
## v0.20.0 — Accept controller `agent_channel_*` events (channel-health relay) (2026-06-29)
|
## v0.20.0 — Accept controller `agent_channel_*` events (channel-health relay) (2026-06-29)
|
||||||
|
|
||||||
Companion to felhom-controller v0.90.0's controller→agent channel health-check. The controller pushes
|
Companion to felhom-controller v0.90.0's controller→agent channel health-check. The controller pushes
|
||||||
|
|||||||
@@ -53,14 +53,17 @@ func NewHostCapabilityChecker(s *store.Store, onEvent EventNotifyFunc, logger *l
|
|||||||
}
|
}
|
||||||
cc.customerOf[row.HostID] = row.CustomerID
|
cc.customerOf[row.HostID] = row.CustomerID
|
||||||
st, _, _ := capabilityState(row.Capabilities)
|
st, _, _ := capabilityState(row.Capabilities)
|
||||||
cc.states[row.HostID] = st
|
// F2: seed only HEALTHY hosts. A host already degraded at (re)start is left UNSEEDED so the
|
||||||
|
// first Check() observes oldState=="" and emits once — otherwise a box already broken when the
|
||||||
|
// hub restarts would stay silent forever. The dispatcher's 1h cooldown dedups the re-emit.
|
||||||
if st == "degraded" {
|
if st == "degraded" {
|
||||||
degCount++
|
degCount++
|
||||||
} else {
|
continue
|
||||||
|
}
|
||||||
|
cc.states[row.HostID] = st
|
||||||
okCount++
|
okCount++
|
||||||
}
|
}
|
||||||
}
|
logger.Printf("[INFO] Host capability checker initialized: %d ok, %d degraded (degraded left unseeded → first Check emits)", okCount, degCount)
|
||||||
logger.Printf("[INFO] Host capability checker initialized: %d ok, %d degraded", okCount, degCount)
|
|
||||||
return cc
|
return cc
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -87,7 +90,14 @@ func (cc *HostCapabilityChecker) Check() {
|
|||||||
newState, names, features := capabilityState(row.Capabilities)
|
newState, names, features := capabilityState(row.Capabilities)
|
||||||
oldState := cc.states[row.HostID]
|
oldState := cc.states[row.HostID]
|
||||||
if oldState == "" {
|
if oldState == "" {
|
||||||
cc.states[row.HostID] = newState // first observation — no event
|
// F2: first observation (incl. a hub restart). A healthy first-obs seeds silently; a host
|
||||||
|
// already DEGRADED at first sight emits once (else a host that was already broken when the
|
||||||
|
// hub (re)started would never alert). The dispatcher's 1h operator cooldown dedups a re-emit
|
||||||
|
// across a hub restart.
|
||||||
|
cc.states[row.HostID] = newState
|
||||||
|
if newState == "degraded" {
|
||||||
|
cc.emitTransition(row.HostID, row.CustomerID, "unknown", newState, names, features)
|
||||||
|
}
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if oldState == newState {
|
if oldState == newState {
|
||||||
|
|||||||
@@ -105,3 +105,28 @@ func TestHostCapabilityChecker_OldAgentNoCaps(t *testing.T) {
|
|||||||
t.Fatalf("old agent (no caps) state = %s, want ok", cc.GetState("h1"))
|
t.Fatalf("old agent (no caps) state = %s, want ok", cc.GetState("h1"))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// F2 RED-PROOF: a host already DEGRADED at hub (re)start is left UNSEEDED by the constructor and the
|
||||||
|
// FIRST Check emits once — so a box that was already broken when the hub restarted alerts (vs the
|
||||||
|
// pre-fix path which seeded it silently and never alerted). Companion: pre-fix the constructor seeded
|
||||||
|
// "degraded" and Check saw oldState==newState → no event.
|
||||||
|
func TestHostCapabilityChecker_F2_BornDegradedAlerts(t *testing.T) {
|
||||||
|
st := newCapStore(t)
|
||||||
|
st.SaveHostReport("h1", "c1", reportWith(capCritDegraded), store.HostReportDenorm{})
|
||||||
|
var events []string
|
||||||
|
cc := NewHostCapabilityChecker(st, func(_, et, _, _, _, _ string) { events = append(events, et) }, log.New(io.Discard, "", 0))
|
||||||
|
if len(events) != 0 {
|
||||||
|
t.Fatalf("construction must not emit, got %v", events)
|
||||||
|
}
|
||||||
|
if cc.GetState("h1") != "unknown" {
|
||||||
|
t.Fatalf("born-degraded must be left unseeded, GetState = %s want unknown", cc.GetState("h1"))
|
||||||
|
}
|
||||||
|
cc.Check()
|
||||||
|
if len(events) != 1 || events[0] != "agent_capability_degraded" {
|
||||||
|
t.Fatalf("born-degraded first Check → one agent_capability_degraded, got %v", events)
|
||||||
|
}
|
||||||
|
cc.Check() // steady degraded → no duplicate
|
||||||
|
if len(events) != 1 {
|
||||||
|
t.Fatalf("steady degraded must not re-emit, got %v", events)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -56,19 +56,20 @@ func NewHostStalenessChecker(s *store.Store, threshold time.Duration, onEvent Ev
|
|||||||
}
|
}
|
||||||
sc.customerOf[row.HostID] = row.CustomerID
|
sc.customerOf[row.HostID] = row.CustomerID
|
||||||
age := time.Since(row.LastReportAt)
|
age := time.Since(row.LastReportAt)
|
||||||
|
// F2: seed only HEALTHY hosts. A host already stale/down at (re)start is left UNSEEDED so the
|
||||||
|
// first Check() observes oldState=="" and emits once (a box that went silent while the hub was
|
||||||
|
// down would otherwise stay silent). The dispatcher's 1h cooldown dedups the re-emit.
|
||||||
switch {
|
switch {
|
||||||
case age > sc.downAfter:
|
case age > sc.downAfter:
|
||||||
sc.states[row.HostID] = "down"
|
|
||||||
downCount++
|
downCount++
|
||||||
case age > sc.threshold:
|
case age > sc.threshold:
|
||||||
sc.states[row.HostID] = "stale"
|
|
||||||
staleCount++
|
staleCount++
|
||||||
default:
|
default:
|
||||||
sc.states[row.HostID] = "ok"
|
sc.states[row.HostID] = "ok"
|
||||||
okCount++
|
okCount++
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
logger.Printf("[INFO] Host staleness checker initialized: %d ok, %d stale, %d down", okCount, staleCount, downCount)
|
logger.Printf("[INFO] Host staleness checker initialized: %d ok, %d stale, %d down (stale/down left unseeded → first Check emits)", okCount, staleCount, downCount)
|
||||||
return sc
|
return sc
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -105,7 +106,13 @@ func (sc *HostStalenessChecker) Check() {
|
|||||||
|
|
||||||
oldState := sc.states[row.HostID]
|
oldState := sc.states[row.HostID]
|
||||||
if oldState == "" {
|
if oldState == "" {
|
||||||
sc.states[row.HostID] = newState // first observation — no event
|
// F2: first observation (incl. a hub restart). Healthy seeds silently; a host already
|
||||||
|
// stale/down at first sight emits once (else a box that went silent while the hub was
|
||||||
|
// down would never alert). The dispatcher's 1h cooldown dedups a re-emit on hub restart.
|
||||||
|
sc.states[row.HostID] = newState
|
||||||
|
if newState != "ok" {
|
||||||
|
sc.emitTransition(row.HostID, row.CustomerID, "unknown", newState, age)
|
||||||
|
}
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if oldState == newState {
|
if oldState == newState {
|
||||||
|
|||||||
@@ -42,20 +42,25 @@ func TestHostStalenessChecker(t *testing.T) {
|
|||||||
events = append(events, eventType)
|
events = append(events, eventType)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Seed already-stale (40m) → state stale, but NO event on init.
|
// F2: already-stale (40m) at construction → left UNSEEDED (not seeded "stale"), NO event on init.
|
||||||
backdate(t, db, "h1", 40)
|
backdate(t, db, "h1", 40)
|
||||||
sc := NewHostStalenessChecker(st, 30*time.Minute, onEvent, log.New(io.Discard, "", 0))
|
sc := NewHostStalenessChecker(st, 30*time.Minute, onEvent, log.New(io.Discard, "", 0))
|
||||||
if len(events) != 0 {
|
if len(events) != 0 {
|
||||||
t.Fatalf("seed must not emit events, got %v", events)
|
t.Fatalf("construction must not emit, got %v", events)
|
||||||
}
|
}
|
||||||
if sc.GetState("h1") != "stale" {
|
if sc.GetState("h1") != "unknown" {
|
||||||
t.Fatalf("seeded state = %q, want stale", sc.GetState("h1"))
|
t.Fatalf("born-stale must be left unseeded, GetState = %q want unknown", sc.GetState("h1"))
|
||||||
}
|
}
|
||||||
|
|
||||||
// Same age → no transition.
|
// F2: the FIRST Check on a born-stale host emits host_stale once (persistent-down alerting).
|
||||||
sc.Check()
|
sc.Check()
|
||||||
if len(events) != 0 {
|
if len(events) != 1 || last(events) != "host_stale" {
|
||||||
t.Fatalf("no transition expected, got %v", events)
|
t.Fatalf("born-stale first Check → one host_stale, got %v", events)
|
||||||
|
}
|
||||||
|
// Steady stale → no duplicate.
|
||||||
|
sc.Check()
|
||||||
|
if len(events) != 1 {
|
||||||
|
t.Fatalf("steady stale must not re-emit, got %v", events)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Fresh report → host_recovered.
|
// Fresh report → host_recovered.
|
||||||
|
|||||||
Reference in New Issue
Block a user