Files
felhom-controller/controller/internal/channelhealth/r271_restart_recovery_test.go
T
admin f885100d29 R-271: the channel recovery after a controller restart is reported
The agent_channel_unauthorized alert's remedy is a re-bootstrap (a restart), and an unseeded->up
first observation was silent, so following the instruction guaranteed no recovery event. A down
alert now leaves a marker in the data dir; the first UP after a restart sends the recovery and
clears it. A restart with no alert outstanding stays silent.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-10-05 21:55:37 +02:00

70 lines
2.4 KiB
Go

package channelhealth
import (
"errors"
"io"
"log"
"os"
"path/filepath"
"testing"
)
// R-271: the 401 alert's remedy is a re-bootstrap (a controller restart), and an unseeded→up first
// observation is silent — so following the instruction guaranteed no recovery event. The consequence
// asserted across a simulated restart: a NEW checker over the same marker file notifies recovery on
// its first UP; a restart with no alert outstanding stays silent (the seeding rule is unchanged).
func TestR271_RecoveryAfterRestartIsNotified(t *testing.T) {
marker := filepath.Join(t.TempDir(), "channelhealth-down-alerted")
lg := log.New(io.Discard, "", 0)
// Process 1: the channel goes down with a 401 and the alert is sent.
s1 := &fakeSink{}
c1 := New(nil, s1, lg).WithAlertMarker(marker)
run(c1, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, nil), step(false, errors.New("agentapi: GET /storage: HTTP 401"))}})
if len(s1.downs) != 1 {
t.Fatalf("setup: want one down alert, got %d", len(s1.downs))
}
// Restart (the remedy). Process 2 sees the channel UP on its first probe.
s2 := &fakeSink{}
c2 := New(nil, s2, lg).WithAlertMarker(marker)
run(c2, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, nil), step(false, nil)}})
if s2.recovered != 1 {
t.Fatalf("R-271: the recovery after a restart must be notified exactly once, got %d", s2.recovered)
}
if _, err := os.Stat(marker); !os.IsNotExist(err) {
t.Errorf("the marker must be cleared once the recovery is sent (stat err=%v)", err)
}
// Process 3: a restart with nothing outstanding stays silent (first observation seeds).
s3 := &fakeSink{}
c3 := New(nil, s3, lg).WithAlertMarker(marker)
run(c3, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, nil)}})
if s3.recovered != 0 {
t.Errorf("a restart with no alert outstanding must not send a recovery, got %d", s3.recovered)
}
// A live recovery (no restart) clears the marker too, so a later restart is silent.
s4 := &fakeSink{}
c4 := New(nil, s4, lg).WithAlertMarker(marker)
run(c4, &scriptedProbe{steps: []struct {
cons bool
err error
}{step(false, nil), step(false, errors.New("agentapi: GET /storage: HTTP 401")), step(false, nil)}})
if s4.recovered != 1 {
t.Fatalf("live recovery: want 1, got %d", s4.recovered)
}
if _, err := os.Stat(marker); !os.IsNotExist(err) {
t.Errorf("a live recovery must clear the marker (stat err=%v)", err)
}
}