Files
felhom.eu/hub/internal/monitor/offsite_r431_test.go
T
admin db38f4c800
gates / gates (push) Successful in 17s
hub v0.111.1: the alarm stops promising a rescue that does not exist, and the arc is closed for beta
R-434 CLOSED — and the row's own "blocked on R-433" verdict was wrong, which is the point.
The fix is a DELETION, not a replacement: withdraw the promise instead of swapping it for a
new one, and the sentence is true under every possible answer to the provider questions, so
it never needs a second rewrite. A replacement would have been blocked; a withdrawal is not.

  was:  "...still hold the older copy, so this is recoverable file-by-file; it is NOT
         confirmed data loss. Check whether a deletion ran on the box before restoring."
  now:  "...still hold the older copy. The route back out of them is not yet established,
         so treat this as neither confirmed data loss nor confirmed recovery. Get in touch
         before restoring anything, and check whether a deletion ran on the box."

It must not swing the other way either: "your backups are gone" is still usually false.
Clause (a) — the box cannot WRITE into the snapshot area — stands and is re-confirmed.

Tests: offsite_r434_test.go, three, all driving the production path so they assert the
sentence an operator RECEIVES. ASCII-only fragments, positive and negative controls.
RED-PROOF: restoring the v0.111.0 sentence failed all three, on every fragment, with the
offending sentence printed. TestR431_FiresOnAMassDeletion asserted "NOT confirmed data
loss" and caught this fix correctly; its wording fragment is REMOVED rather than updated,
so the wording keeps ONE home.

R-435 written into the detector's own documentation, no threshold changed: it sees a mass
deletion, not one app being wiped (69 across 9 apps -> ~35 needed, one tag is ~9, and
forget --prune groups by host,tags). Says explicitly not to lower the numbers.

THE STOPPING LINE, in all three places — register, 07 section 8 head, STATUS.md.
Deferred set ENUMERATED, not described: 07 section 8 rows 4, 8, 9, 10, 11 (+11b), 12,
each tagged [BETA-DEFERRED]. A number in the brief was wrong and is corrected in place:
six rows are DEFERRED, ELEVEN carry a blank RTO (4,5,8,9,10,11,11b,12,13,14,15); the other
five are blank for reasons that are not deferred work, and row 15 is an open DEFECT (R-104)
that the stopping line does NOT cover. NO STATUS MOVED — nothing was proven today.

Two provider questions drafted, not sent, no API called (11-D stands):
documentation/runbooks/provider-questions-2026-09-01.md, linked from R-95 and R-433, and
tracked by a dated DUE-CHECKS row (2026-09-15) — the 2026-07-27 check that sat unconfirmed
for 36 days is the scar that block exists for.

R-95, R-433 BLOCKED-ON-PROVIDER. R-95's one-day demotion on a clause that did not hold is
recorded; the proposal to rank it back near the top is stated and NOT acted on. R-430 marked
LATENT with its trigger: it becomes live the moment delete is withdrawn, so it is a
precondition on the R-95 build, not a follow-up. The stale ranking paragraph ("armed",
"zero snapshots") is corrected in place, order unchanged.

Register 621 -> 688 lines; 181 rows throughout; open-state 170 -> 169.
No controller or agent change. No golden owed, no floor change.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LB8FmJaGd2cyjvy6dbEjpM
2026-09-01 18:34:52 +02:00

287 lines
11 KiB
Go

package monitor
import (
"encoding/json"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"time"
)
// R-431 — the snapshot-drop signal, proven in BOTH directions.
//
// The acceptance test is TestR431_RealHistoryProducesZeroAlarms: a detector that fires on healthy
// boxes is the mistake this project caught twice in one week, and it is the one that would get this
// signal switched off within a fortnight.
// dropJSON builds a trustworthy offsite object (stats_known, no declared state, last run ok).
func dropJSON(count int, statsKnown bool, state, lastStatus string) string {
ts := time.Now().UTC().Add(-1 * time.Hour).Format(time.RFC3339)
success := ts
if lastStatus != "ok" {
success = ""
}
return fmt.Sprintf(
`{"enabled":true,"escrow_state":"escrowed","last_run":%q,"last_status":%q,"last_success":%q,`+
`"snapshot_count":%d,"repo_size_bytes":1073741824,"quota_gb":0,"stats_known":%v,"state":%q}`,
ts, lastStatus, success, count, statsKnown, state)
}
// TestR431_FiresOnAMassDeletion — direction 1. A drop past the threshold alarms EXACTLY once.
//
// RED-PROOF (run 2026-09-01, recorded in REPORT.md): setting snapshotDropFraction to 0.99 makes this
// fail — 69 → 4 is a 94% fall and would no longer qualify, which is what an over-loose threshold
// looks like in production.
func TestR431_FiresOnAMassDeletion(t *testing.T) {
st := newDiskStore(t)
var got []struct{ et, sev, msg string }
saveOffsiteReport(t, st, "victim", dropJSON(69, true, "", "ok"))
oc := NewOffsiteChecker(st, 48*time.Hour, func(_, et, sev, msg, _, _ string) {
got = append(got, struct{ et, sev, msg string }{et, sev, msg})
}, quietLog())
// the constructor seeded the baseline at 69; now the store is emptied
saveOffsiteReport(t, st, "victim", dropJSON(4, true, "", "ok"))
oc.Check()
var drops []struct{ et, sev, msg string }
for _, g := range got {
if g.et == "offsite_snapshots_dropped" {
drops = append(drops, g)
}
}
if len(drops) != 1 {
t.Fatalf("want exactly 1 offsite_snapshots_dropped, got %d (%v)", len(drops), got)
}
// The severity MUST be in the hub's exact vocabulary — anything else is coerced to info and
// mailed to nobody (08 §6.1, shipped twice).
switch drops[0].sev {
case "info", "warning", "error", "critical":
default:
t.Fatalf("severity %q is outside the hub vocabulary — it would be coerced to info and reach nobody", drops[0].sev)
}
// These fragments belong to the SIGNAL — the two counts, and where the older copy still is.
//
// THE WORDING FRAGMENT THAT USED TO SIT HERE IS GONE ON PURPOSE. This list asserted
// "NOT confirmed data loss" until 2026-09-01, and it caught the R-434 fix, correctly — the
// sentence changed because the alarm was promising a recovery that R-433 showed cannot be
// performed. The wording now has ONE home, `offsite_r434_test.go`, which pins both what the
// message must say and what it must never say again. Duplicating it here would create the
// second source that makes the next correction land in one file and not the other.
for _, frag := range []string{"69", "4", "read-only"} {
if !strings.Contains(drops[0].msg, frag) {
t.Fatalf("message must contain %q; got: %s", frag, drops[0].msg)
}
}
if strings.Contains(drops[0].msg, "data is lost") || strings.Contains(drops[0].msg, "data lost") {
t.Fatalf("the message must NOT claim data loss — the snapshots usually still hold it: %s", drops[0].msg)
}
}
// TestR431_EscalationOnlyLatch — a CONTINUING deletion must not page on every report cycle.
//
// WRITTEN AFTER A HOLLOW FIRST ATTEMPT, and the failure is recorded because it is instructive: the
// original assertion re-swept the SAME report and called that "escalation-only". It proved nothing —
// the baseline had already moved to the new count, so the second sweep saw a drop of zero and the
// latch was never consulted. Its red-proof (removing the latch) PASSED, which is how it was caught.
//
// This drives a count that keeps FALLING, which is the only shape where the latch is load-bearing.
//
// RED-PROOF (run 2026-09-01): replacing `dropped && oc.dropStates[...] != "dropped"` with `dropped`
// makes this fail with 2 alarms — one per report cycle, which is the noise that trains an operator
// to ignore the alarm.
func TestR431_EscalationOnlyLatch(t *testing.T) {
st := newDiskStore(t)
var n int
saveOffsiteReport(t, st, "sliding", dropJSON(69, true, "", "ok"))
oc := NewOffsiteChecker(st, 48*time.Hour, func(_, et, _, _, _, _ string) {
if et == "offsite_snapshots_dropped" {
n++
}
}, quietLog())
saveOffsiteReport(t, st, "sliding", dropJSON(30, true, "", "ok")) // 69 -> 30: alarm
oc.Check()
if n != 1 {
t.Fatalf("the first large drop must alarm exactly once; got %d", n)
}
saveOffsiteReport(t, st, "sliding", dropJSON(2, true, "", "ok")) // 30 -> 2: still falling
oc.Check()
if n != 1 {
t.Fatalf("a CONTINUING deletion must not re-page while the latch is set; got %d alarms", n)
}
if oc.GetDropState("sliding") != "dropped" {
t.Fatalf("the latch must be held, got %q", oc.GetDropState("sliding"))
}
// RECOVERY RE-ARMS: a clean sweep clears the latch, so a LATER deletion is caught again.
saveOffsiteReport(t, st, "sliding", dropJSON(40, true, "", "ok")) // rebuilt, no drop
oc.Check()
if oc.GetDropState("sliding") != "ok" {
t.Fatalf("a clean sweep must re-arm the latch, got %q", oc.GetDropState("sliding"))
}
saveOffsiteReport(t, st, "sliding", dropJSON(1, true, "", "ok")) // deleted again
oc.Check()
if n != 2 {
t.Fatalf("after re-arming, a NEW deletion must alarm again; got %d", n)
}
}
// TestR431_SilentWhenNotTrustworthy — direction 2, the three pre-conditions, each with its scar.
func TestR431_SilentWhenNotTrustworthy(t *testing.T) {
cases := []struct {
name, first, second string
}{
{"stats_known absent (R-331: a zero that means UNMEASURED)",
dropJSON(69, true, "", "ok"), dropJSON(0, false, "", "ok")},
{"a DECLARED state (R-204: the box says what happened)",
dropJSON(69, true, "", "ok"), dropJSON(0, true, "needs_credential", "ok")},
{"the run FAILED (R-100: presence is not success)",
dropJSON(69, true, "", "ok"), dropJSON(0, true, "", "error")},
{"the run was INCOMPLETE (R-203: a partial run counts less)",
dropJSON(69, true, "", "ok"), dropJSON(0, true, "", "incomplete")},
}
for i, c := range cases {
t.Run(c.name, func(t *testing.T) {
st := newDiskStore(t)
cid := fmt.Sprintf("c%d", i)
var n int
saveOffsiteReport(t, st, cid, c.first)
oc := NewOffsiteChecker(st, 48*time.Hour, func(_, et, _, _, _, _ string) {
if et == "offsite_snapshots_dropped" {
n++
}
}, quietLog())
saveOffsiteReport(t, st, cid, c.second)
oc.Check()
if n != 0 {
t.Fatalf("%s: must NOT alarm; got %d", c.name, n)
}
// AND the baseline must be untouched, so the RECOVERY does not read as a rise-then-drop.
oc.mu.Lock()
base := oc.lastCounts[cid]
oc.mu.Unlock()
if base != 69 {
t.Fatalf("%s: an untrustworthy report must not overwrite the baseline; got %d", c.name, base)
}
})
}
}
// TestR431_OrdinaryRetentionIsSilent — a fall that retention CAN explain must not alarm.
func TestR431_OrdinaryRetentionIsSilent(t *testing.T) {
for _, c := range []struct {
name string
from, to int
wantAlarms int
}{
{"69 -> 60 (9 gone, under half)", 69, 60, 0},
{"10 -> 7 (3 gone, under the floor of 5)", 10, 7, 0},
{"10 -> 5 (5 gone, exactly half — NOT more than half)", 10, 5, 0},
{"69 -> 34 (35 gone, more than half)", 69, 34, 1},
} {
t.Run(c.name, func(t *testing.T) {
st := newDiskStore(t)
cid := fmt.Sprintf("r%d%d", c.from, c.to)
var n int
saveOffsiteReport(t, st, cid, dropJSON(c.from, true, "", "ok"))
oc := NewOffsiteChecker(st, 48*time.Hour, func(_, et, _, _, _, _ string) {
if et == "offsite_snapshots_dropped" {
n++
}
}, quietLog())
saveOffsiteReport(t, st, cid, dropJSON(c.to, true, "", "ok"))
oc.Check()
if n != c.wantAlarms {
t.Fatalf("%s: want %d alarm(s), got %d", c.name, c.wantAlarms, n)
}
})
}
}
// TestR431_RealHistoryProducesZeroAlarms — THE ACCEPTANCE STEP.
//
// Replays the ACTUAL snapshot-count history of both live boxes, exported from the hub's own reports
// table on 2026-09-01, through the real detector. It must produce ZERO alarms. A warning that fires
// on healthy boxes is worse than no warning at all.
//
// The fixture is committed beside this test so the assertion does not depend on a live database.
// If it is absent the test FAILS rather than skipping — a silent skip is how a green tick comes to
// mean nothing.
func TestR431_RealHistoryProducesZeroAlarms(t *testing.T) {
path := filepath.Join("testdata", "r431_real_history.json")
raw, err := os.ReadFile(path)
if err != nil {
t.Fatalf("the real-history fixture is missing (%v) — this is a FAILURE, never a skip: "+
"without it the acceptance step asserts nothing", err)
}
var hist map[string][]struct {
At string `json:"at"`
Count int `json:"count"`
StatsKnown bool `json:"stats_known"`
State string `json:"state"`
LastStatus string `json:"last_status"`
}
if err := json.Unmarshal(raw, &hist); err != nil {
t.Fatal(err)
}
if len(hist) == 0 {
t.Fatal("the fixture is empty — it would pass vacuously")
}
total := 0
for cust, points := range hist {
if len(points) < 2 {
t.Fatalf("%s: fewer than 2 points — nothing to compare", cust)
}
total += len(points)
st := newDiskStore(t)
var alarms int
var first = points[0]
saveOffsiteReport(t, st, cust, dropJSON(first.Count, first.StatsKnown, first.State, first.LastStatus))
oc := NewOffsiteChecker(st, 48*time.Hour, func(_, et, _, msg, _, _ string) {
if et == "offsite_snapshots_dropped" {
alarms++
t.Errorf("%s: FIRED ON REAL HISTORY: %s", cust, msg)
}
}, quietLog())
// Feed the remaining points straight through the detector. Driving Check() per point would
// need a 1.1s sleep each time (received_at is second-resolution) — hours for 5 000 points —
// so the sweep's own decision path is exercised directly instead, with the same guards.
for _, p := range points[1:] {
off := &offsiteReport{
SnapshotCount: p.Count, StatsKnown: p.StatsKnown, State: p.State,
LastStatus: p.LastStatus, LastSuccess: "2026-09-01T00:00:00Z",
}
if p.LastStatus != "ok" {
off.LastSuccess = ""
}
oc.mu.Lock()
if oc.countIsTrustworthy(off) {
dropped, prev, cur := oc.snapshotDropped(cust, off)
if dropped && oc.dropStates[cust] != "dropped" {
alarms++
t.Errorf("%s at %s: FIRED ON REAL HISTORY %d -> %d", cust, p.At, prev, cur)
oc.dropStates[cust] = "dropped"
} else if !dropped {
oc.dropStates[cust] = "ok"
}
oc.lastCounts[cust] = cur
}
oc.mu.Unlock()
}
if alarms != 0 {
t.Fatalf("%s: %d alarm(s) on real history — the threshold is wrong", cust, alarms)
}
}
// POSITIVE CONTROL: the replay must actually have looked at something. Without this the test
// passes when the fixture is a list of empty lists.
if total < 100 {
t.Fatalf("only %d points replayed — too few for this to mean anything", total)
}
t.Logf("replayed %d real report points across %d customers: ZERO alarms", total, len(hist))
}