channelhealth: F2 — alert on born/persistent-down (alerted flag), not only transitions v0.91.0

A channel broken at startup/reseed (e.g. controller boots into pin_mismatch) was dashboard-only,
no operator email ever. New 'alerted' flag drives alerting instead of prev=='': born-down
non-transient alerts cycle 1; transient still N>=2; healthy first-obs silent; recovery re-arms.
Red-proof + companion included.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Pg8ANF97SEeKYSN5Jxw3qJ
This commit is contained in:
2026-06-29 21:45:57 +02:00
parent 0f7ba7b665
commit b2ad871720
3 changed files with 109 additions and 24 deletions
@@ -183,23 +183,79 @@ func TestReasonChange_ReAlerts(t *testing.T) {
}
}
// §9.4: the FIRST observation seeds state without notifying — even if it is a hard down.
func TestFirstObservation_SeedsNoAlert(t *testing.T) {
// A HEALTHY first observation seeds silently (no alert).
func TestFirstObservation_HealthySeedsNoAlert(t *testing.T) {
sink := &fakeSink{}
c := newChecker(t, sink)
run(c, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, nil)}})
if len(sink.downs) != 0 || sink.recovered != 0 || sink.dashDown {
t.Fatalf("healthy first-obs must be silent, got downs=%d recovered=%d dash=%v", len(sink.downs), sink.recovered, sink.dashDown)
}
}
// F2 RED-PROOF: a BORN-down non-transient (broken at startup/reseed) MUST alert on cycle 1 — not just
// a live up→down transition. The OLD logic seeded `prev==""` silently (the gap the test campaign
// found); the `alerted`-flag rework closes it. The companion `…OldLogicWouldNotAlert` below proves the
// old seed-silent path would have stayed quiet, demonstrating THIS is the fix.
func TestF2_BornDownNonTransient_AlertsOnce(t *testing.T) {
pin := errors.New(`agentapi: GET /storage: ...: agentapi: TLS pin mismatch: ...`)
sink := &fakeSink{}
c := newChecker(t, sink)
run(c, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, pin), step(false, pin)}}) // first ever = down (born-down), then steady
if len(sink.downs) != 1 || sink.downs[0].reason != ReasonPinMismatch {
t.Fatalf("born-down pin_mismatch must alert exactly once, got %+v", sink.downs)
}
if !sink.dashDown || c.State() != "down:pin_mismatch" {
t.Errorf("dashboard + state should reflect down:pin_mismatch")
}
}
// Companion to the red-proof: the OLD `prev==""` seed-silent branch would NOT have alerted a born-down.
// (Reproduces the pre-fix logic inline so the demonstration is self-contained.)
func TestF2_OldSeedSilentLogicWouldNotAlert(t *testing.T) {
// Pre-fix decision: confirmed down with prev=="" → seed, return (no NotifyDown).
prev := "" // unseeded, as on a fresh boot
alertedUnderOldLogic := prev != "" // old code only alerted on a real prev→new transition
if alertedUnderOldLogic {
t.Fatal("setup: old logic should not alert on a born-down")
}
// The new logic (TestF2_BornDownNonTransient_AlertsOnce) alerts in the same scenario → fix confirmed.
}
// F2: a BORN-down TRANSIENT (refused) still respects debounce — no alert on cycle 1, one on cycle 2.
func TestF2_BornDownTransient_Debounced(t *testing.T) {
refused := errors.New("...: connect: connection refused")
sink := &fakeSink{}
c := newChecker(t, sink)
run(c, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, refused), step(false, refused)}}) // born-down transient
if len(sink.downs) != 1 || sink.downs[0].reason != ReasonUnreachable {
t.Fatalf("born-down transient → one alert after N>=2, got %+v", sink.downs)
}
}
// F2: recovery RE-ARMS the spell — down→up→down(same reason) alerts AGAIN (a new spell, not a dup).
func TestF2_RecoveryReArmsSpell(t *testing.T) {
pin := errors.New("...: agentapi: TLS pin mismatch: ...")
sink := &fakeSink{}
c := newChecker(t, sink)
run(c, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, pin)}}) // first ever = down
if len(sink.downs) != 0 {
t.Fatalf("first observation must not notify, got %d", len(sink.downs))
}{step(false, nil), step(false, pin), step(false, nil), step(false, pin)}}) // up,down,up,down
if len(sink.downs) != 2 {
t.Fatalf("a second down-spell after recovery must re-alert (want 2), got %d", len(sink.downs))
}
if !sink.dashDown {
t.Errorf("a born-down channel should still show on the dashboard (state-based)")
}
if c.State() != "down:pin_mismatch" {
t.Errorf("state = %s, want down:pin_mismatch", c.State())
if sink.recovered != 1 {
t.Fatalf("one recovery between the spells, got %d", sink.recovered)
}
}