b11607b26b
R-81 merged every backup signal into one 'newest' against a single 26h limit. backupStaleAfter's own comment recorded why that stops being right under a weekly offsite tier. Each tier is now judged against its own threshold; R-81's structure (three verdicts, anchored absence, distinct reasons) and its boundary test are preserved intact. - offsiteBackupStaleAfter = 8d (7d cadence + headroom); backupStaleAfter keeps 26h and now names the HOST tier only - splitTiers / assessTier / newestBackupEvidenceByTier Slice-A.4 rule implemented: a PBS-targeted vzdump appears in BOTH arrays, so classification is by TARGET TYPE (target_id -> storage_targets[].name -> type), never by array membership — otherwise a PBS backup makes a stale host tier look fresh. storage_targets is used rather than pbs_dr.storage_id because the latter is null on a box with a PBS storage but no DR descriptor. A tier is only judged when the box HAS it, else every box without an offsite tier would alarm once the anchor elapsed — R-81's mistake one level down. With neither tier identifiable (old agent) the pre-Slice-C path runs unchanged. Intended behaviour change: a 30h offsite snapshot no longer alarms. Three fixtures asserted the merged threshold; each still asserts an alarm at the correct limit. No assertion was weakened. RECORDED LIMITATION: the hub infers 'PBS => weekly' from storage type. defaultBackupTarget is felhom-pbs, so a box that never sets local_backup_target would run PBS as its DAILY tier and be judged against 8 days — 7 days of blindness. No box is in that shape today; the real fix is the agent reporting per-tier cadences. Own task. Red-proof observed. Replayed live: demo-felhom OK, demo-hp UNKNOWN (defers correctly), drill-r50 MISSED (true positive). No customer email would be sent.
440 lines
19 KiB
Go
440 lines
19 KiB
Go
package monitor
|
|
|
|
import (
|
|
"encoding/json"
|
|
"log"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
|
)
|
|
|
|
// R-81 — "no signal" is not "bad signal".
|
|
//
|
|
// Origin: 2026-07-26 03:00 UTC, expected_backup_missed fired on demo-felhom, demo-hp and
|
|
// drill-r50 at once. The agent's backup record store is IN-MEMORY
|
|
// (felhom-agent/internal/backup/store.go — "lost on restart; the cadence re-populates"), so
|
|
// the R-50 island migration's fleet restart at 12:44 UTC emptied the host-report `backups`
|
|
// array until the next backup at 07:03. The check read empty as "no backup exists". The
|
|
// vzdump had in fact run: three archives were on disk (07-24, 07-25, 07-26).
|
|
//
|
|
// The fix is an ANCHOR, not silence — the v0.73.0 offsite never-ran shape. These tests pin
|
|
// BOTH halves: absence must stop crying wolf (A, C) AND must still alarm when it is real (B).
|
|
// A test suite that only proved A would pass against an implementation that never alarms,
|
|
// which is strictly worse than the bug it replaces.
|
|
|
|
// evidenceAt builds a backupEvidence with hub-history evidence at `ago` before now, and a
|
|
// first-contact anchor `watched` before now.
|
|
func evidenceAt(now time.Time, ago, watched time.Duration) backupEvidence {
|
|
return backupEvidence{
|
|
newestSeen: now.Add(-ago),
|
|
haveSeen: true,
|
|
firstReportAt: now.Add(-watched),
|
|
}
|
|
}
|
|
|
|
// noEvidence builds a backupEvidence with NO backup evidence, watched for `watched`.
|
|
func noEvidence(now time.Time, watched time.Duration) backupEvidence {
|
|
return backupEvidence{firstReportAt: now.Add(-watched)}
|
|
}
|
|
|
|
// ── Scenario A — the 2026-07-26 case must NOT alarm ──────────────────────────────────────
|
|
|
|
// COMPANION RED-PROOF (observed): removing the ev.haveSeen fold-in from
|
|
// assessBackupFreshness (pre-R-81 shape: judge the latest report alone) makes this test fail
|
|
// with:
|
|
//
|
|
// deadline_anchor_test.go: 07-26 shape must NOT alarm; got verdict=2 reason=
|
|
// "newest backup is 176h0m0s old (limit 26h0m0s)"
|
|
//
|
|
// which is verbatim the message demo-felhom actually sent that morning. Restored after.
|
|
func TestBackupFreshness_AgentRestartBlindWindow_NoAlarm(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
// The real demo-felhom shape: backups:[] (agent forgot), pbs_snapshots:[one from 07-18].
|
|
report := `{"pbs_snapshots":[{"backup_time":"2026-07-18T18:31:06Z","verify_state":"ok"}],"backups":[]}`
|
|
// The hub's retained window still holds the 07-25 06:30 vzdump — ~20.5h before the check.
|
|
ev := evidenceAt(now, 20*time.Hour+30*time.Minute, 8*24*time.Hour)
|
|
|
|
got := assessBackupFreshness(report, ev, now)
|
|
if got.missed() {
|
|
t.Fatalf("07-26 shape must NOT alarm; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
if got.verdict != verdictOK {
|
|
t.Fatalf("evidence exists in the window → verdict must be OK, not a deferred UNKNOWN; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
}
|
|
|
|
// The demo-hp / drill-r50 shape: BOTH arrays empty, nothing in the window either, but the
|
|
// hub has been watching less than the threshold → UNKNOWN, not an alarm.
|
|
func TestBackupFreshness_EmptyArraysWithinGrace_Unknown(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
got := assessBackupFreshness(`{"pbs_snapshots":[],"backups":[]}`, noEvidence(now, 10*time.Hour), now)
|
|
if got.missed() {
|
|
t.Fatalf("absence inside the anchored grace must NOT alarm; got reason=%q", got.reason)
|
|
}
|
|
if got.verdict != verdictUnknown {
|
|
t.Fatalf("want verdictUnknown, got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
}
|
|
|
|
// ── Scenario B — a genuinely dead box MUST still alarm ───────────────────────────────────
|
|
|
|
// THE TEST THAT MAKES SCENARIO A SAFE. Without it, "return silent on absence" passes A and C.
|
|
//
|
|
// COMPANION RED-PROOF (observed): replacing the elapsed-anchor branch with an unconditional
|
|
// `return backupAssessment{verdict: verdictUnknown, ...}` (the naive over-suppression fix)
|
|
// makes this test fail with:
|
|
//
|
|
// deadline_anchor_test.go: a box with NO backup evidence for 240h MUST alarm; got verdict=1
|
|
// reason="no backup evidence yet, but only watching for 240h0m0s ..."
|
|
//
|
|
// Restored after.
|
|
func TestBackupFreshness_NoEvidenceBeyondAnchor_Alarms(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
got := assessBackupFreshness(`{"pbs_snapshots":[],"backups":[]}`, noEvidence(now, 240*time.Hour), now)
|
|
if !got.missed() {
|
|
t.Fatalf("a box with NO backup evidence for 240h MUST alarm; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
// Scenario E: the reason must name absence-over-time, NOT bare absence and NOT staleness.
|
|
if !strings.Contains(got.reason, "no backup evidence in any host-report for") {
|
|
t.Fatalf("absence-over-time needs its own reason string; got %q", got.reason)
|
|
}
|
|
if strings.Contains(got.reason, "newest backup is") {
|
|
t.Fatalf("absence must NOT reuse the stale-timestamp string; got %q", got.reason)
|
|
}
|
|
}
|
|
|
|
// A zero anchor (hub holds no first-contact time) must fail toward VISIBILITY, not silence —
|
|
// the v0.73.0 legacy zero-anchor precedent.
|
|
func TestBackupFreshness_NoEvidenceNoAnchor_Alarms(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
got := assessBackupFreshness(`{"pbs_snapshots":[],"backups":[]}`, backupEvidence{}, now)
|
|
if !got.missed() {
|
|
t.Fatalf("unanchored absence must fail toward visibility; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
if !strings.Contains(got.reason, "no first-contact anchor") {
|
|
t.Fatalf("unanchored absence needs its own reason string; got %q", got.reason)
|
|
}
|
|
}
|
|
|
|
// ── Scenario C — a fresh box is not born failing ─────────────────────────────────────────
|
|
|
|
// THE NAMED BOUNDARY CONTRACT (Part 2). "No evidence + no elapsed window → no alarm" is
|
|
// pinned here as a contract with an obvious name, so re-introducing the bug requires deleting
|
|
// a test that says what it is protecting. Exercises both sides of the boundary.
|
|
//
|
|
// COMPANION RED-PROOF (observed): pre-fix (`if !havePBS && !haveVzdump { return missed }`),
|
|
// the 1-minute-old newborn fails with:
|
|
//
|
|
// deadline_anchor_test.go: CONTRACT VIOLATED: no evidence + no elapsed window must NOT
|
|
// alarm (watched=1m0s, limit=26h0m0s); got reason="no PBS snapshot or successful backup in
|
|
// the latest host-report"
|
|
//
|
|
// Restored after.
|
|
func TestBackupFreshness_Contract_AbsenceIsUnknownUntilAnchorElapses(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
empty := `{"pbs_snapshots":[],"backups":[]}`
|
|
|
|
cases := []struct {
|
|
name string
|
|
watched time.Duration
|
|
wantMissed bool
|
|
}{
|
|
{"newborn, 1 minute", time.Minute, false},
|
|
{"newborn, 1 hour", time.Hour, false},
|
|
{"just inside the window", backupStaleAfter - time.Minute, false},
|
|
{"exactly at the window", backupStaleAfter, false}, // <= is grace, not fault
|
|
{"just outside the window", backupStaleAfter + time.Minute, true},
|
|
{"long past the window", 30 * 24 * time.Hour, true},
|
|
}
|
|
for _, c := range cases {
|
|
t.Run(c.name, func(t *testing.T) {
|
|
got := assessBackupFreshness(empty, noEvidence(now, c.watched), now)
|
|
if got.missed() != c.wantMissed {
|
|
if c.wantMissed {
|
|
t.Fatalf("CONTRACT VIOLATED: absence beyond the window MUST alarm (watched=%s, limit=%s); got verdict=%d reason=%q",
|
|
c.watched, backupStaleAfter, got.verdict, got.reason)
|
|
}
|
|
t.Fatalf("CONTRACT VIOLATED: no evidence + no elapsed window must NOT alarm (watched=%s, limit=%s); got reason=%q",
|
|
c.watched, backupStaleAfter, got.reason)
|
|
}
|
|
if !c.wantMissed && got.verdict != verdictUnknown {
|
|
t.Fatalf("deferred absence must be UNKNOWN (not OK) so it stays visible; got verdict=%d", got.verdict)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// ── Scenario D — the three existing behaviours are untouched ─────────────────────────────
|
|
|
|
// Pins the pre-R-81 outcomes byte-for-byte, INCLUDING the reason strings, under the new
|
|
// signature. Hub-history evidence is present and fresh in each case so it cannot be the
|
|
// thing producing the result — these must hold on the latest report's own merits.
|
|
func TestBackupFreshness_ExistingBehavioursUnchanged(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
at := func(d time.Duration) string { return now.Add(d).Format(time.RFC3339) }
|
|
anchor := 30 * 24 * time.Hour
|
|
|
|
t.Run("1 fresh verified PBS stays silent", func(t *testing.T) {
|
|
r := `{"pbs_snapshots":[{"backup_time":"` + at(-3*time.Hour) + `","verify_state":"ok"}]}`
|
|
got := assessBackupFreshness(r, evidenceAt(now, 3*time.Hour, anchor), now)
|
|
if got.missed() || got.verdict != verdictOK {
|
|
t.Fatalf("want silent OK; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
})
|
|
|
|
t.Run("2 a backup past its tier limit still alarms", func(t *testing.T) {
|
|
// Slice C: the HOST tier keeps the 26h limit verbatim, so this is the same assertion the
|
|
// merged code made — now scoped to the tier it actually belongs to.
|
|
r := `{"backups":[{"target_id":"local","started_at":"` + at(-30*time.Hour) + `","success":true}]}`
|
|
ev := backupEvidence{newestHost: now.Add(-30 * time.Hour), haveHost: true, firstReportAt: now.Add(-anchor)}
|
|
got := assessBackupFreshness(r, ev, now)
|
|
if !got.missed() {
|
|
t.Fatalf("stale host backup must still alarm; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
if got.reason != "host tier: newest backup is 30h0m0s old (limit 26h0m0s)" {
|
|
t.Fatalf("stale reason string changed: %q", got.reason)
|
|
}
|
|
})
|
|
|
|
t.Run("3 failed verify still alarms with the same string", func(t *testing.T) {
|
|
r := `{"pbs_snapshots":[{"backup_time":"` + at(-2*time.Hour) + `","verify_state":"failed"}]}`
|
|
got := assessBackupFreshness(r, evidenceAt(now, 2*time.Hour, anchor), now)
|
|
if !got.missed() {
|
|
t.Fatalf("failed verify must still alarm; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
// Slice C prefixes the tier. The FAILURE MODE is unchanged and still has its own distinct
|
|
// string; only the tier name was added, which is the point of tier-aware reasons.
|
|
if got.reason != "offsite tier: newest PBS snapshot failed verification" {
|
|
t.Fatalf("verify-failed reason string changed: %q", got.reason)
|
|
}
|
|
})
|
|
|
|
t.Run("unparseable report still alarms", func(t *testing.T) {
|
|
got := assessBackupFreshness(`not json`, evidenceAt(now, time.Hour, anchor), now)
|
|
if !got.missed() || got.reason != "latest host-report could not be parsed" {
|
|
t.Fatalf("unparseable behaviour changed: verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
})
|
|
}
|
|
|
|
// Fresh hub-history evidence must NOT rescue a failed verify — behaviour 3 is about the
|
|
// snapshot's integrity, not its age, so the anchor has no business suppressing it.
|
|
func TestBackupFreshness_WindowEvidenceDoesNotRescueFailedVerify(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
r := `{"pbs_snapshots":[{"backup_time":"` + now.Add(-2*time.Hour).Format(time.RFC3339) + `","verify_state":"failed"}]}`
|
|
got := assessBackupFreshness(r, evidenceAt(now, time.Minute, 30*24*time.Hour), now)
|
|
if !got.missed() {
|
|
t.Fatalf("fresh window evidence must not suppress a failed verify; got verdict=%d reason=%q", got.verdict, got.reason)
|
|
}
|
|
}
|
|
|
|
// ── Scenario E — reason strings stay diagnostic ──────────────────────────────────────────
|
|
|
|
// Every failure mode must produce a DISTINCT message. The 2026-07-26 diagnosis turned
|
|
// entirely on reading the exact string; collapsing them would have made it impossible.
|
|
func TestBackupFreshness_ReasonStringsAreDistinct(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
at := func(d time.Duration) string { return now.Add(d).Format(time.RFC3339) }
|
|
|
|
reasons := map[string]string{
|
|
"unparseable": assessBackupFreshness(`nope`, backupEvidence{}, now).reason,
|
|
"stale": assessBackupFreshness(`{"pbs_snapshots":[{"backup_time":"`+at(-9*24*time.Hour)+`","verify_state":"ok"}]}`, backupEvidence{}, now).reason,
|
|
"verify failed": assessBackupFreshness(`{"pbs_snapshots":[{"backup_time":"`+at(-2*time.Hour)+`","verify_state":"failed"}]}`, evidenceAt(now, 2*time.Hour, 30*24*time.Hour), now).reason,
|
|
"absence timed": assessBackupFreshness(`{"backups":[]}`, noEvidence(now, 240*time.Hour), now).reason,
|
|
"absence unanch": assessBackupFreshness(`{"backups":[]}`, backupEvidence{}, now).reason,
|
|
"absence grace": assessBackupFreshness(`{"backups":[]}`, noEvidence(now, time.Hour), now).reason,
|
|
}
|
|
seen := map[string]string{}
|
|
for name, r := range reasons {
|
|
if r == "" {
|
|
t.Fatalf("%s produced an empty reason", name)
|
|
}
|
|
if prev, dup := seen[r]; dup {
|
|
t.Fatalf("reason strings collapsed: %q and %q both produce %q", prev, name, r)
|
|
}
|
|
seen[r] = name
|
|
}
|
|
}
|
|
|
|
// ── newestBackupEvidence — the window scan ───────────────────────────────────────────────
|
|
|
|
func reportRow(t *testing.T, receivedAt time.Time, pbs []string, vzdump []struct {
|
|
at string
|
|
ok bool
|
|
}) store.HostReportRow {
|
|
t.Helper()
|
|
type snap struct {
|
|
BackupTime string `json:"backup_time"`
|
|
VerifyState string `json:"verify_state"`
|
|
}
|
|
type bk struct {
|
|
StartedAt string `json:"started_at"`
|
|
Success bool `json:"success"`
|
|
}
|
|
p := struct {
|
|
PBSSnapshots []snap `json:"pbs_snapshots"`
|
|
Backups []bk `json:"backups"`
|
|
}{}
|
|
for _, s := range pbs {
|
|
p.PBSSnapshots = append(p.PBSSnapshots, snap{BackupTime: s, VerifyState: "ok"})
|
|
}
|
|
for _, b := range vzdump {
|
|
p.Backups = append(p.Backups, bk{StartedAt: b.at, Success: b.ok})
|
|
}
|
|
out, err := json.Marshal(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return store.HostReportRow{ReceivedAt: receivedAt, ReportJSON: string(out)}
|
|
}
|
|
|
|
// The scan must reach PAST the empty reports the restart produced and find the vzdump that
|
|
// an older report still carries. This is the mechanism behind Scenario A.
|
|
func TestNewestBackupEvidence_ReachesPastEmptyReports(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
vz := func(at string, ok bool) []struct {
|
|
at string
|
|
ok bool
|
|
} {
|
|
return []struct {
|
|
at string
|
|
ok bool
|
|
}{{at, ok}}
|
|
}
|
|
rows := []store.HostReportRow{
|
|
// Newest first, as the store returns them. Post-restart reports carry nothing.
|
|
reportRow(t, now.Add(-1*time.Minute), nil, nil),
|
|
reportRow(t, now.Add(-1*time.Hour), nil, nil),
|
|
reportRow(t, now.Add(-14*time.Hour), nil, nil),
|
|
// The last pre-restart report still holds the 20.5h-old vzdump.
|
|
reportRow(t, now.Add(-15*time.Hour), nil, vz(now.Add(-20*time.Hour-30*time.Minute).Format(time.RFC3339), true)),
|
|
reportRow(t, now.Add(-40*time.Hour), nil, vz(now.Add(-44*time.Hour).Format(time.RFC3339), true)),
|
|
}
|
|
got, ok := newestBackupEvidence(rows, now)
|
|
if !ok {
|
|
t.Fatal("must find the vzdump carried by the pre-restart report")
|
|
}
|
|
if want := now.Add(-20*time.Hour - 30*time.Minute); !got.Equal(want) {
|
|
t.Fatalf("newest evidence = %s, want %s", got, want)
|
|
}
|
|
}
|
|
|
|
// A FAILED vzdump is not evidence of a backup.
|
|
func TestNewestBackupEvidence_IgnoresFailedVzdump(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
rows := []store.HostReportRow{
|
|
reportRow(t, now.Add(-time.Hour), nil, []struct {
|
|
at string
|
|
ok bool
|
|
}{{now.Add(-2 * time.Hour).Format(time.RFC3339), false}}),
|
|
}
|
|
if _, ok := newestBackupEvidence(rows, now); ok {
|
|
t.Fatal("a failed vzdump must not count as evidence")
|
|
}
|
|
}
|
|
|
|
// One malformed retained report must not blind the scan to the good ones behind it.
|
|
func TestNewestBackupEvidence_SkipsMalformedRows(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
rows := []store.HostReportRow{
|
|
{ReceivedAt: now.Add(-time.Minute), ReportJSON: `{{{not json`},
|
|
reportRow(t, now.Add(-2*time.Hour), []string{now.Add(-3 * time.Hour).Format(time.RFC3339)}, nil),
|
|
}
|
|
got, ok := newestBackupEvidence(rows, now)
|
|
if !ok {
|
|
t.Fatal("a malformed row must not blind the scan")
|
|
}
|
|
if want := now.Add(-3 * time.Hour); !got.Equal(want) {
|
|
t.Fatalf("newest evidence = %s, want %s", got, want)
|
|
}
|
|
}
|
|
|
|
// No rows at all → no evidence, and emphatically not a zero timestamp treated as evidence.
|
|
func TestNewestBackupEvidence_EmptyWindow(t *testing.T) {
|
|
now := time.Date(2026, 7, 26, 3, 0, 0, 0, time.UTC)
|
|
if _, ok := newestBackupEvidence(nil, now); ok {
|
|
t.Fatal("an empty window must report no evidence")
|
|
}
|
|
}
|
|
|
|
// ── End-to-end through CheckBackupDeadlines ──────────────────────────────────────────────
|
|
|
|
// The full 2026-07-26 replay against the real store: an agent restart empties the array, the
|
|
// hub's retained window still holds yesterday's vzdump → NO event.
|
|
func TestCheckBackupDeadlines_RestartBlindWindow_NoEvent(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
if _, err := st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
okVz := []struct {
|
|
at string
|
|
ok bool
|
|
}{{rfc(-20 * time.Hour), true}}
|
|
|
|
// Pre-restart report: carries the vzdump. Back-dated so it is not the latest.
|
|
pre := hostReportJSON(t, [][2]string{{"2026-07-18T18:31:06Z", "ok"}}, okVz)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(pre), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := st.SetHostReportsReceivedAtForTest("c1", sqliteAgo(15*time.Hour)); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// Post-restart report: PBS only, 176h stale — exactly demo-felhom's 07-26 shape.
|
|
post := hostReportJSON(t, [][2]string{{"2026-07-18T18:31:06Z", "ok"}}, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(post), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
got := runDeadline(t, st)
|
|
if has(got, "expected_backup_missed") {
|
|
t.Fatalf("the 07-26 restart shape must NOT raise expected_backup_missed; got %v", got)
|
|
}
|
|
}
|
|
|
|
// The counterpart end-to-end: a host that has been reporting for days and has NEVER produced
|
|
// a backup must still alarm through the real store path.
|
|
func TestCheckBackupDeadlines_NeverBackedUpBeyondAnchor_Alarms(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller")
|
|
empty := hostReportJSON(t, nil, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(empty), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// Back-date first contact well beyond the 26h anchor.
|
|
if err := st.SetHostReportsReceivedAtForTest("c1", sqliteAgo(120*time.Hour)); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := st.SaveHostReport("h1", "c1", []byte(empty), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
got := runDeadline(t, st)
|
|
if !has(got, "expected_backup_missed") {
|
|
t.Fatalf("a host with no backup for 120h MUST alarm; got %v", got)
|
|
}
|
|
}
|
|
|
|
// A newborn host must not alarm on its first morning — end-to-end.
|
|
func TestCheckBackupDeadlines_NewbornHost_NoEvent(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller")
|
|
empty := hostReportJSON(t, nil, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(empty), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
var logged strings.Builder
|
|
onEvent := func(customerID, eventType, severity, message, detailsJSON, source string) {
|
|
if customerID == "c1" && eventType == "expected_backup_missed" {
|
|
t.Fatalf("a newborn host must not alarm; got %q", message)
|
|
}
|
|
}
|
|
CheckBackupDeadlines(st, nil, onEvent, log.New(&logged, "", 0))
|
|
|
|
// The deferral must be VISIBLE — quiet must never look like "did not run".
|
|
if !strings.Contains(logged.String(), "verdict UNKNOWN") {
|
|
t.Fatalf("the deferred verdict must be logged; log was:\n%s", logged.String())
|
|
}
|
|
}
|