ab2262c91c
gates / gates (push) Successful in 14s
THE GAP, measured not supposed. On 2026-08-18 ep0's PBS proxy was wedged for
9 h 37 m and the hub emitted NOTHING on the operator channel. Both box
checkers hold their last snapshot and return silently on a failed fetch --
correct for a FILL signal, since a missing reading must never be read as 0%,
but it makes a dead off-site endpoint and a healthy one indistinguishable.
The only mails that morning came from the boxes' own backup failures, and
only because the WEEKLY offsite run happened to land inside the window. Two
days earlier nothing would have fired at all.
REACHABILITY is now a second, independent signal on both checkers:
consecutive failed fetch windows, reported past a default 3 windows
(~30-45 min) as pbsdr_box_unreachable / offsite_box_unreachable (warning) on
the customer-less pbsdr-box / pool-box scopes, each with a paired *_recovered
all-clear. Tunable via alerting.box_unreachable_windows (0/invalid -> 3).
THE FILL LOGIC IS UNTOUCHED. No threshold, throttle, band or escalate-once
behaviour changed; a degraded read still drives no transition.
Three decisions a later reader would otherwise "fix" back, so each is
argued in-code:
- the unreachable event REPEATS rather than escalating once. The band shape
would give exactly ONE mail at ~minute 30 of a nine-hour outage, and one
mail is missable. It leans on the dispatcher's 1 h operator cooldown to
become an hourly "still blind" heartbeat.
- ErrUsageUnsupported is NOT blindness: an old ep0 answers "no such op",
which means we reached it. Counting it would alert for days on a healthy
pre-update endpoint.
- born-blind is reported: the counter is not gated on having a snapshot, so
a hub restarted INTO an outage still speaks. last_ok is OMITTED rather
than zero-valued -- a fabricated timestamp reads as "it was fine until
then".
Both recoveries are severity "info" and severityNotifies drops "info", so
they are registered in recoveredPairedDownTypes or the operator hears that
the tier broke and never that it healed. A cross-package test drives
ProcessEvent and asserts an actual operator MAIL, not a map entry -- a green
checker test proves nothing about the seam (agent v0.91.0 shipped fully green
with SetAuthSink never called).
Tests: box_reachability_test.go (Scenarios A-F) + dispatcher_box_reachability
_test.go (wiring). Three red-proofs run and reverted, each seen failing with a
message naming the right cause: threshold 3->1, the sentinel counter guard,
the pairing entry.
Register: R-339 filed and marked SHIPPED (PROVEN-LIVE still owed -- no real or
constructed outage has exercised the emit path, and one cannot be manufactured
against Tier-2 ep0). R-340 filed: the reachability read rides ep0's LOCAL API
daemon, which the incident explicitly cleared, so this check would have shown
GREEN for all 9 h 37 m -- the honest boundary, recorded rather than glossed.
R-336's next-step corrected: pvestatd's interval is NOT tunable (Proxmox staff
have said so); the only lever is disabling the storage entry, which collides
with the agent's consume-the-one-time-secret path. Doc-only, no agent code
touched.
229 lines
8.5 KiB
Go
229 lines
8.5 KiB
Go
package monitor
|
||
|
||
import (
|
||
"fmt"
|
||
"math"
|
||
"testing"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-hub/internal/hetznerapi"
|
||
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
||
)
|
||
|
||
const gib = int64(1) << 30
|
||
|
||
// boxWith builds a fake StorageBox with the given capacity + used bytes (data-only, no snapshots).
|
||
func boxWith(capacityBytes, usedBytes int64) hetznerapi.StorageBox {
|
||
return hetznerapi.StorageBox{
|
||
ID: 1, Status: "active",
|
||
StorageBoxType: hetznerapi.StorageBoxType{Name: "bx11", Size: capacityBytes},
|
||
Stats: hetznerapi.StorageBoxStats{Size: usedBytes, SizeData: usedBytes},
|
||
}
|
||
}
|
||
|
||
// seedOffsiteCfg writes a customer config whose ConfigJSON carries an offsite descriptor.
|
||
func seedOffsiteCfg(t *testing.T, st *store.Store, id string, enabled bool, typ string, quotaGB int) {
|
||
t.Helper()
|
||
cfgJSON := fmt.Sprintf(`{"offsite":{"enabled":%v,"type":%q,"quota_gb":%d}}`, enabled, typ, quotaGB)
|
||
if err := st.SaveCustomerConfig(&store.CustomerConfig{CustomerID: id, APIKey: "k-" + id, RetrievalPassword: "p", ConfigJSON: cfgJSON}); err != nil {
|
||
t.Fatalf("seed config %s: %v", id, err)
|
||
}
|
||
}
|
||
|
||
func noEvent(_, _, _, _, _, _ string) {}
|
||
|
||
// Scenario A — fetch is throttled: ~4 API reads over an hour of 60 s sweeps, not ~60.
|
||
// Red-proof (i): drop the throttle guard in Check() → GetBoxCalls ≈ 60 → this fails.
|
||
func TestOffsiteBox_FetchThrottle(t *testing.T) {
|
||
st := newDiskStore(t)
|
||
fake := hetznerapi.NewFake()
|
||
fake.Boxes[1] = boxWith(1<<40, 300*gib) // 1 TiB, 300 GiB used
|
||
var cur time.Time
|
||
c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, noEvent, quietLog())
|
||
c.now = func() time.Time { return cur }
|
||
|
||
base := time.Now().UTC()
|
||
for i := 0; i < 60; i++ { // 60 sweeps at 60 s spacing = one simulated hour
|
||
cur = base.Add(time.Duration(i) * 60 * time.Second)
|
||
c.Check()
|
||
}
|
||
if fake.GetBoxCalls < 3 || fake.GetBoxCalls > 5 {
|
||
t.Fatalf("GetStorageBox called %d times over an hour of 60 s sweeps, want ≈4 (fetch-throttled to 15 min)", fake.GetBoxCalls)
|
||
}
|
||
snap, ok := c.Snapshot()
|
||
if !ok || snap.CapacityBytes != 1<<40 || snap.UsedBytes != 300*gib {
|
||
t.Fatalf("snapshot wrong: %+v ok=%v", snap, ok)
|
||
}
|
||
}
|
||
|
||
// Scenario B — oversubscription math: Σ = shared+enabled only (A+B); dedicated (C) and disabled (D)
|
||
// excluded; ratio = Σquota / CAPACITY (not used).
|
||
// Red-proof (ii): include dedicated/disabled in the sum → this fails.
|
||
func TestOffsiteBox_OversubMath(t *testing.T) {
|
||
st := newDiskStore(t)
|
||
seedOffsiteCfg(t, st, "a", true, "shared", 500)
|
||
seedOffsiteCfg(t, st, "b", true, "shared", 700)
|
||
seedOffsiteCfg(t, st, "c", true, "dedicated", 9999) // excluded: dedicated
|
||
seedOffsiteCfg(t, st, "d", false, "shared", 300) // excluded: disabled
|
||
|
||
fake := hetznerapi.NewFake()
|
||
fake.Boxes[1] = boxWith(1024*gib, 100*gib) // capacity 1024 GB
|
||
c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, noEvent, quietLog())
|
||
c.Check()
|
||
|
||
snap, _ := c.Snapshot()
|
||
if snap.SumQuotaGB != 1200 {
|
||
t.Fatalf("Σquota = %d GB, want 1200 (A 500 + B 700 only)", snap.SumQuotaGB)
|
||
}
|
||
want := 1200.0 / 1024.0
|
||
if math.Abs(snap.Ratio-want) > 0.001 {
|
||
t.Fatalf("ratio = %.4f, want %.4f (Σquota/capacity)", snap.Ratio, want)
|
||
}
|
||
}
|
||
|
||
// captureEvents records (eventType, severity, customerID) per emit.
|
||
type capturedBox struct{ typ, sev, cust []string }
|
||
|
||
func (c *capturedBox) fn(cust, et, sev, _, _, _ string) {
|
||
c.cust = append(c.cust, cust)
|
||
c.typ = append(c.typ, et)
|
||
c.sev = append(c.sev, sev)
|
||
}
|
||
|
||
// Scenario C (fill) — escalation-only, in-band no re-emit, recovery re-arms. Every emit carries the
|
||
// "pool-box" scope. Red-proof (iii): remove the escalation-only guard → C3 fails.
|
||
func TestOffsiteBox_FillBands(t *testing.T) {
|
||
st := newDiskStore(t)
|
||
fake := hetznerapi.NewFake()
|
||
cap := int64(1000) * gib
|
||
fake.Boxes[1] = boxWith(cap, 750*gib) // 75%
|
||
ev := &capturedBox{}
|
||
var cur time.Time
|
||
c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, ev.fn, quietLog())
|
||
c.now = func() time.Time { return cur }
|
||
base := time.Now().UTC()
|
||
step := func(usedGiB int64) {
|
||
cur = cur.Add(16 * time.Minute) // bust the 15-min throttle each step
|
||
fake.Boxes[1] = boxWith(cap, usedGiB*gib)
|
||
c.Check()
|
||
}
|
||
cur = base
|
||
|
||
c.Check() // C1: 75% → no emit
|
||
if len(ev.typ) != 0 {
|
||
t.Fatalf("C1 75%% must not emit, got %v", ev.typ)
|
||
}
|
||
step(820) // C2: 82% → one warning
|
||
if count(ev.typ, "offsite_box_fill") != 1 || ev.sev[0] != "warning" {
|
||
t.Fatalf("C2 82%% must warn once, got typ=%v sev=%v", ev.typ, ev.sev)
|
||
}
|
||
step(850) // C3: 85% same band → no second emit
|
||
if count(ev.typ, "offsite_box_fill") != 1 {
|
||
t.Fatalf("C3 same-band must NOT re-emit, got %d", count(ev.typ, "offsite_box_fill"))
|
||
}
|
||
step(920) // C4: 92% → escalate to critical
|
||
if count(ev.typ, "offsite_box_fill") != 2 || ev.sev[len(ev.sev)-1] != "critical" {
|
||
t.Fatalf("C4 92%% must escalate to critical, got typ=%v sev=%v", ev.typ, ev.sev)
|
||
}
|
||
step(700) // C5: 70% → recovery re-arm (no emit)
|
||
if c.FillState() != bandOK {
|
||
t.Fatalf("C5 recovery must re-arm to ok, got %s", c.FillState())
|
||
}
|
||
if count(ev.typ, "offsite_box_fill") != 2 {
|
||
t.Fatalf("C5 recovery must be silent, got %d emits", count(ev.typ, "offsite_box_fill"))
|
||
}
|
||
step(920) // re-breach after recovery → emits again
|
||
if count(ev.typ, "offsite_box_fill") != 3 {
|
||
t.Fatalf("re-breach after recovery must emit again, got %d", count(ev.typ, "offsite_box_fill"))
|
||
}
|
||
for _, cust := range ev.cust {
|
||
if cust != "pool-box" {
|
||
t.Fatalf("every emit must carry the pool-box scope, got %q", cust)
|
||
}
|
||
}
|
||
}
|
||
|
||
// Scenario C6 — oversubscription is an INDEPENDENT signal: it fires on ratio alone even when fill is
|
||
// nominal, with its own event type.
|
||
func TestOffsiteBox_OversubIndependent(t *testing.T) {
|
||
st := newDiskStore(t)
|
||
seedOffsiteCfg(t, st, "a", true, "shared", 2300) // Σ 2300 GB
|
||
fake := hetznerapi.NewFake()
|
||
fake.Boxes[1] = boxWith(1000*gib, 500*gib) // 50% fill (nominal), ratio 2.3×
|
||
ev := &capturedBox{}
|
||
c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, ev.fn, quietLog())
|
||
c.Check()
|
||
if count(ev.typ, "offsite_box_oversub") != 1 {
|
||
t.Fatalf("2.3× oversub must warn once, got %v", ev.typ)
|
||
}
|
||
if count(ev.typ, "offsite_box_fill") != 0 {
|
||
t.Fatalf("fill is nominal (50%%) — no fill emit, got %v", ev.typ)
|
||
}
|
||
if ev.cust[0] != "pool-box" {
|
||
t.Fatalf("oversub emit scope = %q, want pool-box", ev.cust[0])
|
||
}
|
||
}
|
||
|
||
// Scenario D — API failure honesty: the last snapshot is KEPT (marked degraded), no band transition,
|
||
// and a failed fetch NEVER zeroes the snapshot (the recovery re-arm must not trigger off missing data).
|
||
// Red-proof (iv): let a failed fetch zero the snapshot → this fails.
|
||
func TestOffsiteBox_FailedFetchHonesty(t *testing.T) {
|
||
st := newDiskStore(t)
|
||
fake := hetznerapi.NewFake()
|
||
cap := int64(1000) * gib
|
||
fake.Boxes[1] = boxWith(cap, 920*gib) // 92% → critical
|
||
ev := &capturedBox{}
|
||
var cur time.Time
|
||
c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, ev.fn, quietLog())
|
||
c.now = func() time.Time { return cur }
|
||
base := time.Now().UTC()
|
||
cur = base
|
||
|
||
c.Check() // establishes a 92% critical snapshot + emit
|
||
if count(ev.typ, "offsite_box_fill") != 1 {
|
||
t.Fatalf("setup: want one critical emit, got %v", ev.typ)
|
||
}
|
||
before, _ := c.Snapshot()
|
||
|
||
// Now the API fails.
|
||
fake.FailGetBox = fmt.Errorf("hetzner timeout")
|
||
cur = cur.Add(16 * time.Minute)
|
||
c.Check()
|
||
|
||
after, ok := c.Snapshot()
|
||
if !ok {
|
||
t.Fatal("a failed fetch must KEEP the last snapshot, not drop it")
|
||
}
|
||
if after.UsedBytes != before.UsedBytes || after.CapacityBytes != before.CapacityBytes {
|
||
t.Fatalf("a failed fetch must not alter the last-known values (before %d/%d, after %d/%d)",
|
||
before.UsedBytes, before.CapacityBytes, after.UsedBytes, after.CapacityBytes)
|
||
}
|
||
if !after.Degraded {
|
||
t.Fatal("a failed fetch must mark the served snapshot Degraded (visible staleness)")
|
||
}
|
||
if c.FillState() != bandCritical {
|
||
t.Fatalf("a failed fetch must NOT transition the band (missing data ≠ 0%%), got %s", c.FillState())
|
||
}
|
||
// No new emit on the failure.
|
||
if count(ev.typ, "offsite_box_fill") != 1 {
|
||
t.Fatalf("a failed fetch must not emit, got %d total", count(ev.typ, "offsite_box_fill"))
|
||
}
|
||
}
|
||
|
||
// Not-configured / zero-capacity guard: a zero-capacity box marks degraded, never divides, never alerts.
|
||
func TestOffsiteBox_ZeroCapacityGuard(t *testing.T) {
|
||
st := newDiskStore(t)
|
||
fake := hetznerapi.NewFake()
|
||
fake.Boxes[1] = boxWith(0, 0) // initializing box — no capacity yet
|
||
ev := &capturedBox{}
|
||
c := NewOffsiteBoxChecker(fake, 1, st, 80, 90, 2.0, 0, ev.fn, quietLog())
|
||
c.Check()
|
||
snap, ok := c.Snapshot()
|
||
if !ok || !snap.Degraded {
|
||
t.Fatalf("zero capacity must yield a degraded snapshot, got %+v", snap)
|
||
}
|
||
if len(ev.typ) != 0 {
|
||
t.Fatalf("zero capacity must not alert, got %v", ev.typ)
|
||
}
|
||
}
|