package monitor import ( "encoding/json" "errors" "testing" "time" "gitea.dooplex.hu/admin/felhom-hub/internal/hetznerapi" "gitea.dooplex.hu/admin/felhom-hub/internal/tenantsync" ) // R-339 — box REACHABILITY. These tests cover the signal that was MISSING on 2026-08-18, when ep0's // PBS proxy was wedged for 9 h 37 m and the hub said nothing on the operator channel. // // Every assertion here checks a VERDICT — the event type, its severity, and the specific details // fields — not merely "an event was emitted" or "the count is N". That is deliberate and recent: // this afternoon's due_checks_gate red-proof passed for the wrong reason because its only assertion // was an exit code that a crash produced just as well as the logic under test. A bare count is the // same trap in a different costume. // capturedFull records ALL SIX EventNotifyFunc arguments, unlike capturedBox which keeps three. // Reachability assertions need the details payload (consecutive_failures, last_ok, blind_for), and a // recorder that discards it cannot tell a correct event from a plausible one. type capturedFull struct { cust, typ, sev, msg, details, src []string } func (c *capturedFull) fn(cust, et, sev, msg, details, src string) { c.cust = append(c.cust, cust) c.typ = append(c.typ, et) c.sev = append(c.sev, sev) c.msg = append(c.msg, msg) c.details = append(c.details, details) c.src = append(c.src, src) } func (c *capturedFull) n() int { return len(c.typ) } // ofType returns the indices of events of a given type — so a test can assert "exactly these" rather // than "at least one somewhere". func (c *capturedFull) ofType(t string) []int { var out []int for i, et := range c.typ { if et == t { out = append(out, i) } } return out } func (c *capturedFull) detail(i int) map[string]any { var m map[string]any _ = json.Unmarshal([]byte(c.details[i]), &m) return m } // ── Group A — Scenario A: the endpoint goes dark and STAYS dark ────────────────────────────── func TestPBSDRBox_Unreachable_SustainedOutage(t *testing.T) { f := &fakeUsage{usage: usageBytes(40, 8)} ev := &capturedFull{} var cur time.Time c := NewPBSDRBoxChecker(f, 80, 90, 0, ev.fn, quietLog()) // 0 → default threshold 3 c.now = func() time.Time { return cur } base := time.Now().UTC() cur = base c.Check() // one healthy read if ev.n() != 0 { t.Fatalf("healthy read emitted %d event(s), want 0: %v", ev.n(), ev.typ) } bandBefore := c.FillState() f.set(tenantsync.BoxUsage{}, errors.New("dial tcp 10.77.0.1:8007: i/o timeout")) for i := 1; i <= 6; i++ { cur = base.Add(time.Duration(i) * boxFetchInterval) c.Check() } idx := ev.ofType("pbsdr_box_unreachable") if len(idx) != 4 { t.Fatalf("want 4 unreachable events across 6 failed windows (silent on 1 and 2), got %d (all: %v)", len(idx), ev.typ) } for _, i := range idx { if ev.sev[i] != "warning" { t.Fatalf("event %d severity = %q, want \"warning\"", i, ev.sev[i]) } if ev.cust[i] != pbsdrBoxScope { t.Fatalf("event %d scope = %q, want %q", i, ev.cust[i], pbsdrBoxScope) } if ev.src[i] != "hub" { t.Fatalf("event %d source = %q, want \"hub\"", i, ev.src[i]) } } // The counter is the escalation signal — first alert at 3, last at 6. if got := ev.detail(idx[0])["consecutive_failures"]; got != float64(3) { t.Fatalf("first alert consecutive_failures = %v, want 3", got) } if got := ev.detail(idx[len(idx)-1])["consecutive_failures"]; got != float64(6) { t.Fatalf("last alert consecutive_failures = %v, want 6", got) } if _, ok := ev.detail(idx[0])["last_ok"]; !ok { t.Fatalf("first alert must carry last_ok — there WAS a successful read before the outage") } // The fill signal must be untouched throughout: missing data never moves a band. if len(ev.ofType("pbsdr_box_fill")) != 0 { t.Fatalf("an outage emitted a FILL event — missing data must never drive a band transition") } if c.FillState() != bandBefore { t.Fatalf("fill band moved during an outage: %q → %q", bandBefore, c.FillState()) } } // ── Group B — Scenario B: a transient blip below threshold ─────────────────────────────────── func TestPBSDRBox_Unreachable_BlipBelowThreshold(t *testing.T) { f := &fakeUsage{usage: usageBytes(40, 8)} ev := &capturedFull{} var cur time.Time c := NewPBSDRBoxChecker(f, 80, 90, 0, ev.fn, quietLog()) c.now = func() time.Time { return cur } base := time.Now().UTC() cur = base c.Check() f.set(tenantsync.BoxUsage{}, errors.New("transient")) cur = base.Add(boxFetchInterval) c.Check() cur = base.Add(2 * boxFetchInterval) c.Check() if ev.n() != 0 { t.Fatalf("two failed windows emitted %v, want silence below the threshold", ev.typ) } f.set(usageBytes(40, 8), nil) cur = base.Add(3 * boxFetchInterval) c.Check() if ev.n() != 0 { t.Fatalf("recovery below threshold emitted %v — there was no alert to clear", ev.typ) } // The counter reset is proven by the NEGATIVE: one more failure must not reach the threshold. // Reading the private field would prove the field, not the behaviour. f.set(tenantsync.BoxUsage{}, errors.New("transient again")) cur = base.Add(4 * boxFetchInterval) c.Check() if ev.n() != 0 { t.Fatalf("a single failure after a success emitted %v — the counter did not reset", ev.typ) } } // ── Group C — Scenario C: recovery after an alert ──────────────────────────────────────────── func TestPBSDRBox_Unreachable_Recovery(t *testing.T) { f := &fakeUsage{usage: usageBytes(40, 8)} ev := &capturedFull{} var cur time.Time c := NewPBSDRBoxChecker(f, 80, 90, 0, ev.fn, quietLog()) c.now = func() time.Time { return cur } base := time.Now().UTC() cur = base c.Check() f.set(tenantsync.BoxUsage{}, errors.New("down")) for i := 1; i <= 3; i++ { cur = base.Add(time.Duration(i) * boxFetchInterval) c.Check() } if len(ev.ofType("pbsdr_box_unreachable")) == 0 { t.Fatalf("setup failed: no unreachable event to recover from") } f.set(usageBytes(40, 8), nil) cur = base.Add(4 * boxFetchInterval) c.Check() rec := ev.ofType("pbsdr_box_recovered") if len(rec) != 1 { t.Fatalf("want exactly 1 recovery event, got %d (all: %v)", len(rec), ev.typ) } if ev.sev[rec[0]] != "info" { t.Fatalf("recovery severity = %q, want \"info\"", ev.sev[rec[0]]) } if ev.cust[rec[0]] != pbsdrBoxScope { t.Fatalf("recovery scope = %q, want %q", ev.cust[rec[0]], pbsdrBoxScope) } bf, _ := ev.detail(rec[0])["blind_for"].(string) if bf == "" || bf == "unknown" { t.Fatalf("recovery blind_for = %q, want a real duration", bf) } before := ev.n() cur = base.Add(5 * boxFetchInterval) c.Check() if ev.n() != before { t.Fatalf("a second consecutive success emitted %v — the alerted flag did not clear", ev.typ[before:]) } } // ── Group D — Scenario D: ErrUsageUnsupported is NOT blindness ─────────────────────────────── func TestPBSDRBox_UsageUnsupported_IsNotBlindness(t *testing.T) { f := &fakeUsage{err: tenantsync.ErrUsageUnsupported} ev := &capturedFull{} var cur time.Time c := NewPBSDRBoxChecker(f, 80, 90, 0, ev.fn, quietLog()) c.now = func() time.Time { return cur } base := time.Now().UTC() for i := 0; i < 10; i++ { cur = base.Add(time.Duration(i) * boxFetchInterval) c.Check() } if ev.n() != 0 { t.Fatalf("ErrUsageUnsupported emitted %v — an expected pre-update condition must never alert", ev.typ) } snap, ok := c.Snapshot() if !ok || snap.State != PBSStateUnavailable { t.Fatalf("state = %q (have=%v), want %q", snap.State, ok, PBSStateUnavailable) } // The counter must never have advanced: one REAL error now must still be below the threshold. f.set(tenantsync.BoxUsage{}, errors.New("real transport failure")) cur = base.Add(10 * boxFetchInterval) c.Check() if ev.n() != 0 { t.Fatalf("a single real error after 10 unsupported windows emitted %v — the unsupported "+ "branch advanced the counter, which would alert on a healthy pre-update endpoint", ev.typ) } } // ── Group E — Scenario E: born blind (hub restarted INTO an outage) ────────────────────────── func TestPBSDRBox_BornBlind_StillReports(t *testing.T) { f := &fakeUsage{err: errors.New("connection refused")} ev := &capturedFull{} var cur time.Time c := NewPBSDRBoxChecker(f, 80, 90, 0, ev.fn, quietLog()) // no snapshot, ever c.now = func() time.Time { return cur } base := time.Now().UTC() for i := 0; i < 3; i++ { cur = base.Add(time.Duration(i) * boxFetchInterval) c.Check() } idx := ev.ofType("pbsdr_box_unreachable") if len(idx) != 1 { t.Fatalf("born-blind checker emitted %d unreachable events, want exactly 1 (all: %v)", len(idx), ev.typ) } d := ev.detail(idx[0]) if v, present := d["last_ok"]; present { t.Fatalf("last_ok = %v, but there has NEVER been a successful read — a fabricated timestamp "+ "reads as \"it was fine until then\", which is a lie", v) } if got := d["consecutive_failures"]; got != float64(3) { t.Fatalf("consecutive_failures = %v, want 3", got) } if ev.sev[idx[0]] != "warning" { t.Fatalf("severity = %q, want \"warning\"", ev.sev[idx[0]]) } } // ── Group F — Scenario F: the storage box, same rules ──────────────────────────────────────── func TestOffsiteBox_Unreachable_AndRecovery(t *testing.T) { st := newDiskStore(t) fake := hetznerapi.NewFake() fake.Boxes[1] = boxWith(1000*gib, 100*gib) // 10% — comfortably ok ev := &capturedFull{} var cur time.Time c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, ev.fn, quietLog()) c.now = func() time.Time { return cur } base := time.Now().UTC() cur = base c.Check() fillBefore, oversubBefore := c.FillState(), c.OversubState() fake.FailGetBox = errors.New("hetzner api: 503 service unavailable") for i := 1; i <= 3; i++ { cur = base.Add(time.Duration(i) * boxFetchInterval) c.Check() } idx := ev.ofType("offsite_box_unreachable") if len(idx) != 1 { t.Fatalf("want exactly 1 unreachable at the third window, got %d (all: %v)", len(idx), ev.typ) } if ev.sev[idx[0]] != "warning" || ev.cust[idx[0]] != offsiteBoxScope { t.Fatalf("event severity/scope = %q/%q, want warning/%s", ev.sev[idx[0]], ev.cust[idx[0]], offsiteBoxScope) } if got := ev.detail(idx[0])["consecutive_failures"]; got != float64(3) { t.Fatalf("consecutive_failures = %v, want 3", got) } fake.FailGetBox = nil cur = base.Add(4 * boxFetchInterval) c.Check() rec := ev.ofType("offsite_box_recovered") if len(rec) != 1 { t.Fatalf("want exactly 1 recovery, got %d (all: %v)", len(rec), ev.typ) } if ev.sev[rec[0]] != "info" { t.Fatalf("recovery severity = %q, want \"info\"", ev.sev[rec[0]]) } // Neither capacity band may move across the whole outage. if len(ev.ofType("offsite_box_fill")) != 0 || len(ev.ofType("offsite_box_oversub")) != 0 { t.Fatalf("an outage emitted a capacity event: %v", ev.typ) } if c.FillState() != fillBefore || c.OversubState() != oversubBefore { t.Fatalf("bands moved during an outage: fill %q→%q oversub %q→%q", fillBefore, c.FillState(), oversubBefore, c.OversubState()) } } // A zero-capacity read is a SUCCESS for reachability: we reached the endpoint and it answered. The // answer being unusable for a percentage is a different complaint, and conflating the two would // leave a blindness alert outstanding forever on a box that is actually responding. func TestPBSDRBox_ZeroCapacitySuccess_ClearsBlindness(t *testing.T) { f := &fakeUsage{usage: usageBytes(40, 8)} ev := &capturedFull{} var cur time.Time c := NewPBSDRBoxChecker(f, 80, 90, 0, ev.fn, quietLog()) c.now = func() time.Time { return cur } base := time.Now().UTC() cur = base c.Check() f.set(tenantsync.BoxUsage{}, errors.New("down")) for i := 1; i <= 3; i++ { cur = base.Add(time.Duration(i) * boxFetchInterval) c.Check() } if len(ev.ofType("pbsdr_box_unreachable")) == 0 { t.Fatal("setup failed: expected an unreachable alert") } f.set(tenantsync.BoxUsage{Total: 0, Used: 0}, nil) // reached, answered, unusable payload cur = base.Add(4 * boxFetchInterval) c.Check() if len(ev.ofType("pbsdr_box_recovered")) != 1 { t.Fatalf("a zero-capacity SUCCESS did not clear the blindness alert (events: %v)", ev.typ) } snap, _ := c.Snapshot() if snap.State != PBSStateDegraded { t.Fatalf("snapshot state = %q, want %q — the fill signal is still degraded, correctly", snap.State, PBSStateDegraded) } }