b11607b26b
R-81 merged every backup signal into one 'newest' against a single 26h limit. backupStaleAfter's own comment recorded why that stops being right under a weekly offsite tier. Each tier is now judged against its own threshold; R-81's structure (three verdicts, anchored absence, distinct reasons) and its boundary test are preserved intact. - offsiteBackupStaleAfter = 8d (7d cadence + headroom); backupStaleAfter keeps 26h and now names the HOST tier only - splitTiers / assessTier / newestBackupEvidenceByTier Slice-A.4 rule implemented: a PBS-targeted vzdump appears in BOTH arrays, so classification is by TARGET TYPE (target_id -> storage_targets[].name -> type), never by array membership — otherwise a PBS backup makes a stale host tier look fresh. storage_targets is used rather than pbs_dr.storage_id because the latter is null on a box with a PBS storage but no DR descriptor. A tier is only judged when the box HAS it, else every box without an offsite tier would alarm once the anchor elapsed — R-81's mistake one level down. With neither tier identifiable (old agent) the pre-Slice-C path runs unchanged. Intended behaviour change: a 30h offsite snapshot no longer alarms. Three fixtures asserted the merged threshold; each still asserts an alarm at the correct limit. No assertion was weakened. RECORDED LIMITATION: the hub infers 'PBS => weekly' from storage type. defaultBackupTarget is felhom-pbs, so a box that never sets local_backup_target would run PBS as its DAILY tier and be judged against 8 days — 7 days of blindness. No box is in that shape today; the real fix is the agent reporting per-tier cadences. Own task. Red-proof observed. Replayed live: demo-felhom OK, demo-hp UNKNOWN (defers correctly), drill-r50 MISSED (true positive). No customer email would be sent.
220 lines
9.5 KiB
Go
220 lines
9.5 KiB
Go
package monitor
|
|
|
|
import (
|
|
"encoding/json"
|
|
"io"
|
|
"log"
|
|
"path/filepath"
|
|
"testing"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
|
)
|
|
|
|
// newDeadlineStore creates an isolated store with one active customer + a host row.
|
|
func newDeadlineStore(t *testing.T) *store.Store {
|
|
t.Helper()
|
|
st, err := store.New(filepath.Join(t.TempDir(), "test.db"), log.New(io.Discard, "", 0))
|
|
if err != nil {
|
|
t.Fatalf("store.New: %v", err)
|
|
}
|
|
t.Cleanup(func() { st.Close() })
|
|
if err := st.SaveCustomerConfig(&store.CustomerConfig{CustomerID: "c1", APIKey: "ck", RetrievalPassword: "p"}); err != nil {
|
|
t.Fatalf("SaveCustomerConfig: %v", err)
|
|
}
|
|
if err := st.UpsertHost(&store.Host{HostID: "h1", CustomerID: "c1", APIKey: "k1"}); err != nil {
|
|
t.Fatalf("UpsertHost: %v", err)
|
|
}
|
|
return st
|
|
}
|
|
|
|
// hostReportJSON builds a host-report payload with the given PBS snapshots and vzdump
|
|
// backups. Each snapshot is {backup_time, verify_state}; each backup is {started_at, success}.
|
|
func hostReportJSON(t *testing.T, pbs [][2]string, backups []struct {
|
|
at string
|
|
ok bool
|
|
}) string {
|
|
t.Helper()
|
|
type snap struct {
|
|
BackupTime string `json:"backup_time"`
|
|
VerifyState string `json:"verify_state"`
|
|
}
|
|
type bk struct {
|
|
StartedAt string `json:"started_at"`
|
|
Success bool `json:"success"`
|
|
}
|
|
payload := struct {
|
|
HostID string `json:"host_id"`
|
|
PBSSnapshots []snap `json:"pbs_snapshots"`
|
|
Backups []bk `json:"backups"`
|
|
}{HostID: "h1"}
|
|
for _, p := range pbs {
|
|
payload.PBSSnapshots = append(payload.PBSSnapshots, snap{BackupTime: p[0], VerifyState: p[1]})
|
|
}
|
|
for _, b := range backups {
|
|
payload.Backups = append(payload.Backups, bk{StartedAt: b.at, Success: b.ok})
|
|
}
|
|
out, err := json.Marshal(payload)
|
|
if err != nil {
|
|
t.Fatalf("marshal report: %v", err)
|
|
}
|
|
return string(out)
|
|
}
|
|
|
|
// runDeadline runs CheckBackupDeadlines and returns the list of emitted event types for c1.
|
|
func runDeadline(t *testing.T, st *store.Store) []string {
|
|
t.Helper()
|
|
var got []string
|
|
onEvent := func(customerID, eventType, severity, message, detailsJSON, source string) {
|
|
if customerID == "c1" {
|
|
got = append(got, eventType)
|
|
}
|
|
}
|
|
// nil staleness → no "down" skip; the check evaluates c1.
|
|
CheckBackupDeadlines(st, nil, onEvent, log.New(io.Discard, "", 0))
|
|
return got
|
|
}
|
|
|
|
func has(events []string, t string) bool {
|
|
for _, e := range events {
|
|
if e == t {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func rfc(d time.Duration) string {
|
|
return time.Now().UTC().Add(d).Format(time.RFC3339)
|
|
}
|
|
|
|
// TestCheckBackupDeadlines_FreshVerifiedPBS_NoBackupAlarm is the COMPANION test.
|
|
// It FAILS against the pre-fix (event-based) check: with a fresh, verified PBS snapshot
|
|
// but no backup_completed event, the old code emitted expected_backup_missed anyway.
|
|
// The repoint reads the host-report instead, so a healthy customer raises no alarm.
|
|
func TestCheckBackupDeadlines_FreshVerifiedPBS_NoBackupAlarm(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
// Make the db-dump half pass too, so the only thing under test is the backup half.
|
|
if _, err := st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
report := hostReportJSON(t, [][2]string{{rfc(-3 * time.Hour), "ok"}}, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(report), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
got := runDeadline(t, st)
|
|
if has(got, "expected_backup_missed") {
|
|
t.Fatalf("fresh+verified PBS must NOT raise expected_backup_missed; got %v", got)
|
|
}
|
|
if has(got, "expected_dbdump_missed") {
|
|
t.Fatalf("db_dump_completed present → no dbdump alarm expected; got %v", got)
|
|
}
|
|
}
|
|
|
|
// TestCheckBackupDeadlines_StalePBS_Alarms: an offsite snapshot past the OFFSITE limit → alarm.
|
|
//
|
|
// R-82 Slice C changed the threshold this fixture must cross, NOT the behaviour it asserts. 30h was
|
|
// "stale" while one 26h limit covered every tier; under a weekly offsite tier a 30h snapshot is
|
|
// healthy, and alarming on it is exactly the cry-wolf `backupStaleAfter`'s comment predicted. The
|
|
// test still proves a stale offsite tier alarms — now at 9 days, past the 8-day offsite limit.
|
|
func TestCheckBackupDeadlines_StalePBS_Alarms(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller")
|
|
report := hostReportJSON(t, [][2]string{{rfc(-9 * 24 * time.Hour), "ok"}}, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(report), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
got := runDeadline(t, st)
|
|
if !has(got, "expected_backup_missed") {
|
|
t.Fatalf("stale (>26h) PBS snapshot must raise expected_backup_missed; got %v", got)
|
|
}
|
|
}
|
|
|
|
// TestCheckBackupDeadlines_FailedVerify_Alarms: fresh snapshot but verify failed → alarm.
|
|
func TestCheckBackupDeadlines_FailedVerify_Alarms(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller")
|
|
report := hostReportJSON(t, [][2]string{{rfc(-2 * time.Hour), "failed"}}, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(report), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
got := runDeadline(t, st)
|
|
if !has(got, "expected_backup_missed") {
|
|
t.Fatalf("failed PBS verify must raise expected_backup_missed; got %v", got)
|
|
}
|
|
}
|
|
|
|
// TestCheckBackupDeadlines_NoHostReport_NoBackupAlarm: a customer with no host-report
|
|
// (legacy/defunct controller-only) must NOT get a backup alarm from this check —
|
|
// liveness is the staleness checker's job.
|
|
func TestCheckBackupDeadlines_NoHostReport_NoBackupAlarm(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
st.SaveEvent("c1", "db_dump_completed", "info", "", "{}", "controller")
|
|
// No SaveHostReport call.
|
|
got := runDeadline(t, st)
|
|
if has(got, "expected_backup_missed") {
|
|
t.Fatalf("no host-report → no backup alarm; got %v", got)
|
|
}
|
|
}
|
|
|
|
// TestCheckBackupDeadlines_DbDumpHalfPreserved: fresh backup (no backup alarm) but a
|
|
// missing db_dump_completed event must still raise expected_dbdump_missed.
|
|
func TestCheckBackupDeadlines_DbDumpHalfPreserved(t *testing.T) {
|
|
st := newDeadlineStore(t)
|
|
report := hostReportJSON(t, [][2]string{{rfc(-2 * time.Hour), "ok"}}, nil)
|
|
if err := st.SaveHostReport("h1", "c1", []byte(report), store.HostReportDenorm{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
got := runDeadline(t, st)
|
|
if has(got, "expected_backup_missed") {
|
|
t.Fatalf("fresh backup → no backup alarm; got %v", got)
|
|
}
|
|
if !has(got, "expected_dbdump_missed") {
|
|
t.Fatalf("missing db_dump_completed must still raise expected_dbdump_missed; got %v", got)
|
|
}
|
|
}
|
|
|
|
// TestAssessBackupFreshness exercises the pure freshness policy directly.
|
|
//
|
|
// R-81: every case here passes a ZERO backupEvidence — no hub-history evidence and no
|
|
// first-contact anchor. That is deliberate: it pins the latest-report-only behaviour
|
|
// unchanged, and the "no snapshots and no backups" row exercises the UNANCHORED absence
|
|
// branch (zero anchor → fail toward visibility, the v0.73.0 legacy-shape precedent).
|
|
// The anchored branches have their own named tests below.
|
|
func TestAssessBackupFreshness(t *testing.T) {
|
|
now := time.Date(2026, 6, 16, 3, 0, 0, 0, time.UTC)
|
|
at := func(d time.Duration) string { return now.Add(d).Format(time.RFC3339) }
|
|
|
|
cases := []struct {
|
|
name string
|
|
report string
|
|
wantMissed bool
|
|
}{
|
|
{"fresh verified PBS", `{"pbs_snapshots":[{"backup_time":"` + at(-3*time.Hour) + `","verify_state":"ok"}]}`, false},
|
|
{"fresh unverified (none) is not a failure", `{"pbs_snapshots":[{"backup_time":"` + at(-3*time.Hour) + `","verify_state":"none"}]}`, false},
|
|
// Slice C: a 30h offsite snapshot is HEALTHY under the weekly (8d) offsite limit.
|
|
{"offsite 30h is fresh under the weekly limit", `{"pbs_snapshots":[{"backup_time":"` + at(-30*time.Hour) + `","verify_state":"ok"}]}`, false},
|
|
{"offsite 6d is fresh under the weekly limit", `{"pbs_snapshots":[{"backup_time":"` + at(-6*24*time.Hour) + `","verify_state":"ok"}]}`, false},
|
|
{"offsite 9d is STALE past the weekly limit", `{"pbs_snapshots":[{"backup_time":"` + at(-9*24*time.Hour) + `","verify_state":"ok"}]}`, true},
|
|
{"fresh but verify failed", `{"pbs_snapshots":[{"backup_time":"` + at(-2*time.Hour) + `","verify_state":"failed"}]}`, true},
|
|
{"no snapshots and no backups", `{"pbs_snapshots":[],"backups":[]}`, true},
|
|
{"host vzdump fresh success", `{"backups":[{"target_id":"local","started_at":"` + at(-4*time.Hour) + `","success":true}]}`, false},
|
|
{"host vzdump only, failed → counts as none", `{"backups":[{"target_id":"local","started_at":"` + at(-4*time.Hour) + `","success":false}]}`, true},
|
|
{"host fresh + offsite within weekly → both healthy", `{"pbs_snapshots":[{"backup_time":"` + at(-40*time.Hour) + `","verify_state":"ok"}],"backups":[{"target_id":"local","started_at":"` + at(-2*time.Hour) + `","success":true}]}`, false},
|
|
// Slice C's core: the tiers are judged SEPARATELY. A fresh host backup must NOT rescue an
|
|
// offsite tier that is genuinely past its own (weekly) limit — the merged threshold did.
|
|
{"host fresh does NOT rescue a 9d offsite tier", `{"pbs_snapshots":[{"backup_time":"` + at(-9*24*time.Hour) + `","verify_state":"ok"}],"backups":[{"target_id":"local","started_at":"` + at(-2*time.Hour) + `","success":true}]}`, true},
|
|
{"offsite fresh does NOT rescue a 30h host tier", `{"pbs_snapshots":[{"backup_time":"` + at(-2*time.Hour) + `","verify_state":"ok"}],"backups":[{"target_id":"local","started_at":"` + at(-30*time.Hour) + `","success":true}]}`, true},
|
|
{"unparseable report", `not json`, true},
|
|
}
|
|
for _, c := range cases {
|
|
t.Run(c.name, func(t *testing.T) {
|
|
got := assessBackupFreshness(c.report, backupEvidence{}, now)
|
|
if got.missed() != c.wantMissed {
|
|
t.Fatalf("missed=%v want=%v (reason=%q)", got.missed(), c.wantMissed, got.reason)
|
|
}
|
|
})
|
|
}
|
|
}
|