37ae31fd44
gates / gates (push) Successful in 21s
R-549 (operator ruling A): controllerStatus hardcoded 30m/1h while both staleness checkers and hostStatus read alerting.stale_threshold. Moving the threshold to 45m would have painted a customer amber 15 minutes before the alarm could fire - the second definition rollup.go's header forbids. It now reads the same value, down at 2x. Both 'checker initialized' log lines print the threshold, which no line did before. R-539 (ruling 3 of 2026-09-16): controller_slow_crashloop (warning, operator-only), minted when the agent's slow_crashloop_since moves, with the fast sibling's first-sight rule. R-550: restore_interrupted (warning, for the household) allowlisted with a Hungarian customer message. Red-proofs, each seen failing then passing: the status test with the old hardcoded numbers; the checker test with the movement branch removed; the operator-only test with the registration removed; the household-message test with the Hungarian entry removed (asserted on the SUBJECT - the body legitimately repeats the raw message, which my first version of the test mistook for a fallback). go build/vet/test ./... green, 18 packages. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
226 lines
8.6 KiB
Go
226 lines
8.6 KiB
Go
package web
|
||
|
||
// v0.53.0 dead-host roll-up honesty (drill-1 masking observation; operator ruling 2026-07-13).
|
||
// Scenario C is the LIVE Peti-cluster shape: a host down 23h while the controller keeps
|
||
// reporting through the internet — pre-fix the customer row rendered GREEN (the wrong outcome
|
||
// these tests pin). Scenario D bounds the fold: all-ok is a no-regression pass-through, a stale
|
||
// host warns with its own chip, and pending hosts worsen only after onboarding.
|
||
// RED-PROOF (§10 C): short-circuiting foldHostStatus to `return base, ""` (controller-only
|
||
// derivation) → the C assertions fail with the green row / missing chip visible.
|
||
|
||
import (
|
||
"database/sql"
|
||
"fmt"
|
||
"io"
|
||
"log"
|
||
"net/http/httptest"
|
||
"path/filepath"
|
||
"strings"
|
||
"testing"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
||
)
|
||
|
||
// newRollupServer builds a server whose store DB path is known, so tests can backdate host
|
||
// reports over a second connection (the monitor host_staleness_test pattern).
|
||
func newRollupServer(t *testing.T) (*Server, *store.Store, *sql.DB) {
|
||
t.Helper()
|
||
path := filepath.Join(t.TempDir(), "t.db")
|
||
st, err := store.New(path, log.New(io.Discard, "", 0))
|
||
if err != nil {
|
||
t.Fatalf("store.New: %v", err)
|
||
}
|
||
t.Cleanup(func() { st.Close() })
|
||
db, err := sql.Open("sqlite", path)
|
||
if err != nil {
|
||
t.Fatalf("sql.Open: %v", err)
|
||
}
|
||
t.Cleanup(func() { db.Close() })
|
||
// staleThreshold 30m → host "stale" past 30m, "down" past 60m (the single definition).
|
||
s := New(st, "", "", "test", 30*time.Minute, log.New(io.Discard, "", 0))
|
||
return s, st, db
|
||
}
|
||
|
||
func backdateHost(t *testing.T, db *sql.DB, hostID string, minutesAgo int) {
|
||
t.Helper()
|
||
if _, err := db.Exec(`UPDATE hosts SET last_report_at = datetime('now', ?) WHERE host_id = ?`,
|
||
fmt.Sprintf("-%d minutes", minutesAgo), hostID); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
}
|
||
|
||
func renderDashboard(t *testing.T, s *Server) string {
|
||
t.Helper()
|
||
rr := httptest.NewRecorder()
|
||
s.handleDashboard(rr, httptest.NewRequest("GET", "/", nil))
|
||
if rr.Code != 200 {
|
||
t.Fatalf("dashboard = %d", rr.Code)
|
||
}
|
||
return rr.Body.String()
|
||
}
|
||
|
||
func renderCustomer(t *testing.T, s *Server, customerID string) string {
|
||
t.Helper()
|
||
rr := httptest.NewRecorder()
|
||
s.handleCustomerUnified(rr, httptest.NewRequest("GET", "/customers/"+customerID, nil), customerID)
|
||
if rr.Code != 200 {
|
||
t.Fatalf("customer detail = %d", rr.Code)
|
||
}
|
||
return rr.Body.String()
|
||
}
|
||
|
||
// Scenario C — the Peti shape: proxmox1 down 23h, fresh controller report → the customer row
|
||
// must read WARN with the cause chip naming the host; the detail header must say WHICH host.
|
||
// WRONG outcome (pre-fix): a green row.
|
||
func TestRollup_DeadHostMasking(t *testing.T) {
|
||
s, st, db := newRollupServer(t)
|
||
if err := st.SaveCustomerConfig(&store.CustomerConfig{
|
||
CustomerID: "acme", CustomerName: "Acme", APIKey: "k", RetrievalPassword: "p",
|
||
}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
// Fresh controller report (the guest reports hub-direct, independent of the host agent).
|
||
if err := st.SaveReport("acme", []byte(`{"customer_id":"acme","customer_name":"Acme"}`)); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
// The host: reported once, then silent for 23h → "down" by THE definition.
|
||
if err := st.UpsertHost(&store.Host{HostID: "proxmox1", CustomerID: "acme", APIKey: "hk"}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.SaveHostReport("proxmox1", "acme", []byte(`{}`), store.HostReportDenorm{}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
backdateHost(t, db, "proxmox1", 23*60)
|
||
|
||
body := renderDashboard(t, s)
|
||
if !strings.Contains(body, "host down: proxmox1") {
|
||
t.Error("dashboard row lacks the cause chip \"host down: proxmox1\"")
|
||
}
|
||
if strings.Contains(body, "status-badge-ok") {
|
||
t.Error("dashboard row is GREEN over a 23h-dead host — the exact masking bug")
|
||
}
|
||
if !strings.Contains(body, `status-badge-warn">host down: proxmox1`) {
|
||
t.Error("cause chip not rendered as a warn badge")
|
||
}
|
||
|
||
// Detail header: WHICH host.
|
||
detail := renderCustomer(t, s, "acme")
|
||
if !strings.Contains(detail, "host down: proxmox1") {
|
||
t.Error("customer detail header does not name the down host")
|
||
}
|
||
}
|
||
|
||
// Scenario D — boundaries: all-ok pass-through (no regression), stale-host warn chip, and the
|
||
// onboarding rule for pending hosts.
|
||
func TestRollup_Boundaries(t *testing.T) {
|
||
t.Run("all hosts ok leaves controller status untouched", func(t *testing.T) {
|
||
s, st, _ := newRollupServer(t)
|
||
if err := st.SaveReport("acme", []byte(`{"customer_id":"acme"}`)); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.UpsertHost(&store.Host{HostID: "h-ok", CustomerID: "acme", APIKey: "hk"}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.SaveHostReport("h-ok", "acme", []byte(`{}`), store.HostReportDenorm{}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
body := renderDashboard(t, s)
|
||
if !strings.Contains(body, "status-badge-ok") {
|
||
t.Error("healthy customer + healthy host must stay OK")
|
||
}
|
||
if strings.Contains(body, "host down") || strings.Contains(body, "host stale") || strings.Contains(body, "host pending") {
|
||
t.Error("cause chip rendered with every host ok")
|
||
}
|
||
})
|
||
|
||
t.Run("single stale host caps at warn with stale chip", func(t *testing.T) {
|
||
s, st, db := newRollupServer(t)
|
||
if err := st.SaveReport("acme", []byte(`{"customer_id":"acme"}`)); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.UpsertHost(&store.Host{HostID: "h-stale", CustomerID: "acme", APIKey: "hk"}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.SaveHostReport("h-stale", "acme", []byte(`{}`), store.HostReportDenorm{}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
backdateHost(t, db, "h-stale", 45) // between stale (30m) and down (60m)
|
||
body := renderDashboard(t, s)
|
||
if !strings.Contains(body, "host stale: h-stale") {
|
||
t.Error("stale-host cause chip missing")
|
||
}
|
||
if strings.Contains(body, "status-badge-ok") {
|
||
t.Error("customer stayed green over a stale host")
|
||
}
|
||
})
|
||
|
||
t.Run("pending host during onboarding does not worsen", func(t *testing.T) {
|
||
s, st, _ := newRollupServer(t)
|
||
// Config-only customer (never reported) + freshly-minted host (never reported).
|
||
if err := st.SaveCustomerConfig(&store.CustomerConfig{
|
||
CustomerID: "newbie", CustomerName: "Newbie", APIKey: "k", RetrievalPassword: "p",
|
||
}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.UpsertHost(&store.Host{HostID: "h-new", CustomerID: "newbie", APIKey: "hk"}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
body := renderDashboard(t, s)
|
||
if !strings.Contains(body, "status-badge-pending") {
|
||
t.Error("onboarding customer must render PENDING")
|
||
}
|
||
if strings.Contains(body, "host pending") {
|
||
t.Error("a never-reported host worsened a never-reported customer (onboarding exclusion violated)")
|
||
}
|
||
})
|
||
|
||
t.Run("pending host after onboarding worsens", func(t *testing.T) {
|
||
s, st, _ := newRollupServer(t)
|
||
if err := st.SaveReport("acme", []byte(`{"customer_id":"acme"}`)); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := st.UpsertHost(&store.Host{HostID: "h-silent", CustomerID: "acme", APIKey: "hk"}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
body := renderDashboard(t, s)
|
||
if !strings.Contains(body, "host pending: h-silent") {
|
||
t.Error("a never-reported host must worsen a LIVE customer (post-onboarding)")
|
||
}
|
||
if strings.Contains(body, "status-badge-ok") {
|
||
t.Error("customer stayed green over a never-reported host")
|
||
}
|
||
})
|
||
}
|
||
|
||
// R-549 (operator ruling A, 2026-09-17): the "box went quiet" threshold moved from 30 m to 45 m in
|
||
// configuration. The customer status on the dashboard must move WITH it — the file header forbids a
|
||
// second definition, and before this change controllerStatus hardcoded 30 m / 1 h while the checkers
|
||
// read the config. The consequence asserted: with the configured threshold at 45 m, a report 40 m old
|
||
// is NOT amber (the alarm has not fired either), 50 m is amber, 95 m is down (2 × threshold, the same
|
||
// multiplier the checkers use).
|
||
//
|
||
// RED-PROOF: controllerStatus ignoring its threshold (the hardcoded 30 m / 1 h) → the 40 m case
|
||
// reads "warn" and the 70 m case reads "down".
|
||
func TestControllerStatus_FollowsConfiguredThreshold(t *testing.T) {
|
||
th := 45 * time.Minute
|
||
for _, tc := range []struct {
|
||
age time.Duration
|
||
want string
|
||
}{
|
||
{40 * time.Minute, "ok"},
|
||
{50 * time.Minute, "warn"},
|
||
{70 * time.Minute, "warn"},
|
||
{95 * time.Minute, "down"},
|
||
} {
|
||
c := &store.CustomerSummary{TimeSinceReport: tc.age, HealthStatus: "ok"}
|
||
if got := controllerStatus(c, th); got != tc.want {
|
||
t.Errorf("threshold %s, report %s old: status %q, want %q — the dashboard and the alarm disagree", th, tc.age, got, tc.want)
|
||
}
|
||
}
|
||
// A zero threshold (a Server built without one) keeps the historical default, never "everything down".
|
||
if got := controllerStatus(&store.CustomerSummary{TimeSinceReport: 10 * time.Minute, HealthStatus: "ok"}, 0); got != "ok" {
|
||
t.Errorf("zero threshold: 10m-old report reads %q, want ok", got)
|
||
}
|
||
}
|