package monitor import ( "encoding/json" "fmt" "log" "sync" "time" "gitea.dooplex.hu/admin/felhom-hub/internal/store" ) // OffsiteChecker (SLICE 4) watches each customer's offsite backup health from the controller report's // `offsite` status object. Two independent signals, one checker (sibling of StorageFillChecker — same // born/persistent, escalation-only emit, recovery re-arm shape; NOT bolted onto the disk checkers — // different data source, different remedy text): // // - FILL: repo_size_bytes vs the shared-model soft quota (quota_gb>0) at warn 90% / crit 95% — the // operator's early warning before the controller's own 100% run-refusal bites the customer. // - SNAPSHOT-DROP (R-431): the reported snapshot count falls by more than retention can explain — // the "something deleted this customer's off-site history" detector. See snapshotDrop below for // why it lives HERE and not on the box, and for the measurement behind its threshold. // - STALENESS: enabled + escrowed but no run in >48h (or never) — the silently-STUCK detector. A // RECENTLY-failing offsite is NOT stale (backup_failed already alerts it); staleness is the // complement: nothing is even trying. Pending/disabled targets are normal onboarding, never stale. // // Reports without an `offsite` object (pre-v0.109 controllers, offbox not enabled) are skipped nil-safe. type OffsiteChecker struct { store *store.Store logger *log.Logger onEvent EventNotifyFunc staleAfter time.Duration // legacyWarned tracks the one-shot R-100 legacy-degrade log, per customer. legacyMu sync.Mutex legacyWarned map[string]bool now func() time.Time // injectable clock (tests) mu sync.Mutex fillStates map[string]string // customerID → fill band staleStates map[string]string // customerID → "ok" | "stale" // R-431. dropStates is the escalation latch ("ok" | "dropped"); lastCounts is the baseline the // next sweep compares against. Only a TRUSTWORTHY report updates lastCounts — see snapshotDrop. dropStates map[string]string lastCounts map[string]int } const defaultOffsiteStaleAfter = 48 * time.Hour // offsiteReport mirrors the controller report's `offsite` object (v0.109.0). type offsiteReport struct { Enabled bool `json:"enabled"` EscrowState string `json:"escrow_state"` LastRun string `json:"last_run"` LastStatus string `json:"last_status"` // LastSuccess (R-100) is the last run that actually SUCCEEDED — the staleness anchor. Absent on a // controller older than v0.181.0; isStale degrades explicitly in that case, see there. LastSuccess string `json:"last_success"` SnapshotCount int `json:"snapshot_count"` RepoSizeBytes int64 `json:"repo_size_bytes"` QuotaGB int `json:"quota_gb"` // StatsKnown (R-331/R-225) — the ONLY thing separating "this repository holds nothing" from // "nobody has ever measured this repository". Both are `snapshot_count: 0` on the wire and they // are opposite news. ABSENT on a pre-v0.225.0 controller, and absence means the box CANNOT // ANSWER, never that the answer is no. snapshotDrop refuses to judge without it. StatsKnown bool `json:"stats_known,omitempty"` // State (R-204/R-193) — a DECLARED condition, empty on every healthy box. A box that has // re-initialised, lost its credential or been abandoned reports a real, correct, large drop. // The box knows its own situation; believe it rather than alarming on it. State string `json:"state,omitempty"` } // NewOffsiteChecker builds the checker. Same seeding philosophy as StorageFillChecker: already-breached // customers are left UNSEEDED so their first Check emits (born/persistent); the dispatcher's cooldown // dedups a hub restart. func NewOffsiteChecker(s *store.Store, staleAfter time.Duration, onEvent EventNotifyFunc, logger *log.Logger) *OffsiteChecker { if staleAfter <= 0 { staleAfter = defaultOffsiteStaleAfter } oc := &OffsiteChecker{ store: s, logger: logger, onEvent: onEvent, staleAfter: staleAfter, now: time.Now, fillStates: make(map[string]string), staleStates: make(map[string]string), dropStates: make(map[string]string), lastCounts: make(map[string]int), } customers, err := s.GetCustomers() if err != nil { logger.Printf("[WARN] Offsite checker: failed to seed states: %v", err) return oc } var seeded int for _, c := range customers { off := parseOffsite(c.ReportJSON) if off == nil || s.IsCustomerBlocked(c.CustomerID) { continue } if band := oc.fillBand(off); band == bandOK { oc.fillStates[c.CustomerID] = bandOK seeded++ } if !oc.isStale(c.CustomerID, off) { oc.staleStates[c.CustomerID] = "ok" } // R-431: seed the baseline from the current report so a hub restart does not read the first // sweep as a drop from nothing. Only a trustworthy report seeds — an untrustworthy one leaves // no baseline, and no baseline means no verdict. if oc.countIsTrustworthy(off) { oc.lastCounts[c.CustomerID] = off.SnapshotCount oc.dropStates[c.CustomerID] = "ok" } } logger.Printf("[INFO] Offsite checker initialized: fill warn=90%% crit=95%%, stale after %s, %d ok-seeded", staleAfter, seeded) return oc } func parseOffsite(reportJSON string) *offsiteReport { var r struct { Offsite *offsiteReport `json:"offsite"` } if json.Unmarshal([]byte(reportJSON), &r) != nil { return nil } return r.Offsite // nil when absent (old controller / offbox not enabled) — the caller skips } // fillBand maps the quota usage to a band. quota<=0 (dedicated/unset) never alerts. func (oc *OffsiteChecker) fillBand(off *offsiteReport) string { if off.QuotaGB <= 0 || off.RepoSizeBytes <= 0 { return bandOK } pct := float64(off.RepoSizeBytes) * 100 / float64(int64(off.QuotaGB)<<30) return bandForPercent(pct, 90, 95) } // isStale: enabled + ESCROWED (the only state where runs are expected) with no SUCCESSFUL run in // >staleAfter (or never ran, ANCHORED — see below). Pending/disabled = normal onboarding, never stale. // // R-100 — PRESENCE IS NOT SUCCESS. This counted from `LastRun`, which the controller writes // unconditionally at the end of every run INCLUDING failures. So it asked "how long since we last // TRIED", and a tier failing on every single run refreshed the clock nightly and read as perfectly // fresh forever. It now counts from `LastSuccess`, which only a successful run advances. // // The old comment here said "a recent-but-failing run is NOT stale (backup_failed owns that signal)", // and that WAS true — `backup_failed` does fire, nightly, and reaches the operator. The defect is not // silence, it is DEFEATED DEFENCE IN DEPTH: this checker is the hub-side, pull-based net that exists to // be independent of controller-PUSHED events, and anchoring it on a field the failing controller keeps // refreshing made it depend on the very thing it backs up. F-HUB (the hub dropping an event under // SQLITE_BUSY, no retry) is exactly that loss. INVARIANT pinned by TestIsStale_* in offsite_r100_test.go. // // Part-7 (v0.73.0) — the never-ran branch no longer fires on sight. The 2026-07-23 cry-wolf: // demo-hp's tier was repaired and escrowed at 10:01Z and offsite_stale fired MINUTES later // (`last_run:"" … threshold 48h`), because "enabled + escrowed + never ran" had no time anchor. // Boundary: reaching this code at all means the latest report CARRIES the offsite object — i.e. // the v0.72.0 delivery state is `applied` (Check nil-skips everything else); pre-applied // never-ran shapes are offsite_delivery_stuck's alone — ONE STATE, ONE OWNER, never both. // Anchor: the newest of one_time_secrets.consumed_at (delivery completed) and the customer's // escrow-blob timestamp (host_escrow.updated_at/created_at — runs become POSSIBLE only at the // ceremony). The EXISTING staleAfter threshold, anchored there, IS the grace — no new knob. func (oc *OffsiteChecker) isStale(customerID string, off *offsiteReport) bool { if !off.Enabled || off.EscrowState != "escrowed" { return false } // NEVER RAN AT ALL — unchanged v0.73.0 behaviour, and it must stay unchanged. A newborn box has no // LastRun and no LastSuccess; alarming it on sight is the 2026-07-23 cry-wolf. Checked on LastRun // (not LastSuccess) deliberately: LastRun is the "has anything ever happened here" signal, and a box // whose first run FAILED has a LastRun but no LastSuccess — that is a run, not a newborn, and it // belongs on the anchored path below. if off.LastRun == "" { anchor := oc.neverRanAnchor(customerID) if anchor.IsZero() { return true // legacy shape (no secret timestamps, no escrow row) — fail toward visibility, as before } return oc.now().Sub(anchor) > oc.staleAfter } // LEGACY CONTROLLER — it has run, but sends no last_success (pre-v0.181.0). Handle it EXPLICITLY, // in the direction that preserves known behaviour: count from LastRun exactly as before. // - treating absence as FAILURE would alarm every un-upgraded box in the fleet at once; // - treating it as SUCCESS is the bug. // Preserving today's behaviour is correct here; the controller version floor drives the upgrade. // Same degrade direction as R-88 Part 2's age_state, and for the same reason. if off.LastSuccess == "" { oc.warnLegacyOnce(customerID) t, err := time.Parse(time.RFC3339, off.LastRun) if err != nil { return true // unparseable = unknown-old — fail toward visibility } return oc.now().Sub(t) > oc.staleAfter } // THE ANCHORED VERDICT. Note what is NOT consulted: LastStatus. A single failed night does not make // a tier stale — the threshold simply keeps running from the last good run, so one blip is tolerated // and a persistent failure is caught. Reading LastStatus here ("error ⇒ stale") would alarm on every // transient blip, which is the F-A1 noise path that trains an operator to ignore the alarm. // `running` is a real wire value (a report captured mid-run) and is likewise none of our business. t, err := time.Parse(time.RFC3339, off.LastSuccess) if err != nil { return true // unparseable = unknown-old — fail toward visibility } return oc.now().Sub(t) > oc.staleAfter } // ── SNAPSHOT-DROP (R-431) ─────────────────────────────────────────────────────────────────────── // // WHY THIS LIVES ON THE HUB AND NOT ON THE BOX. The thing being detected is a box deleting its own // off-site backups — so a detector living on that box is a detector the same event can silence. The // hub already receives the count in every report and already keeps the history, so it can notice // without trusting the box's judgement. The box still REPORTS the number, and a sophisticated // attacker could lie about it; that is a far higher bar than deleting files, and the weekly integrity // check (R-359) would then disagree with the lie. // // WHAT IT IS FOR. R-95: the restic credential can delete its own repository, and the sub-account API // has no append-only axis, so prevention needs a transport change. The Storage Box's daily ZFS // snapshots bound the loss — MEASURED 2026-09-01: a write into `/.zfs/snapshot` is REFUSED // (`dest open …: Failure`) while the same write to the account home succeeds. So the remedy exists; // what was missing was NOTICING, and this is that. // // THE THRESHOLD, AND THE MEASUREMENT IT CAME FROM. Measured over the hub's own `reports` table on // 2026-09-01: 12 898 reports, 4 customers, 2026-06-05 → 2026-09-01. // // * In the whole history there are NINE decreases, and EVERY ONE of them lands exactly on ZERO // (36→0, 18→0 ×2, 15→0, 12→0, 8→0, 3→0). There is not one gradual retention decrease anywhere. // * Every one of those nine predates `stats_known`, i.e. they are the R-331 shape — a zero that // means "could not measure", not "nothing is there". Several carry a declared State // (`needs_credential`, `awaiting_recovery_key`), which says so outright. // * In the window where `stats_known` is TRUE (380 reports across both live boxes) there are ZERO // decreases: demo-felhom sat flat at 10, demo-hp moved 67→68→69. Only rises. // // So observed retention churn gives NOTHING to calibrate against, and saying so is the honest answer // rather than inventing a number (R-401's lesson). The threshold is therefore reasoned from what // retention CAN do, not from what it was seen to do: // // the box runs `forget --keep-daily 7 --keep-weekly 4 --keep-monthly 6 --group-by host,tags` // over ~8 apps. On a boundary day several groups can expire at once, so a legitimate pass can // plausibly remove low double digits. **What it can NEVER do is halve the total**: keeping 7 daily // + 4 weekly + 6 monthly per group is a floor, and a mass deletion goes to ~0. // // Hence: **a fall of MORE THAN HALF the previous count, and at least 5 snapshots.** The 50% cannot be // reached by retention; the floor of 5 stops a tiny-count box alarming on ordinary ageing. It is // deliberately NOT sensitive — a detector that cries wolf is switched off within a fortnight, and // this project has proved that twice in a week. // WHAT THIS DETECTOR DOES NOT SEE — R-435, and it must be read wherever "an unexplained fall is // noticed within a day" is claimed, because that claim is true only of falls above the fraction. // // **It sees a MASS deletion. It does not see ONE APP being wiped.** Worked on the live fleet // 2026-09-01: demo-hp's baseline is 69 snapshots across 9 apps, so ~35 must go before this speaks; // one app's tag is ~9 and is invisible. And `offbox.go:1388` runs `forget --prune` **grouped by // host,tags** — a per-tag wipe is exactly the shape a faulty retention or a targeted deletion // produces, so the blind spot sits on the most likely single-app failure, not an exotic one. // // THIS IS DELIBERATE AND THE THRESHOLD SHOULD NOT BE LOWERED TO "FIX" IT. The reasoning is below: a // detector that cries wolf is switched off within a fortnight, and this project has proved that // twice in a week. What is NOT acceptable is claiming coverage this does not have. Anyone adding // per-app detection should add a SECOND signal keyed on the per-tag count, not move these numbers. const ( snapshotDropFraction = 0.5 // more than half the history gone in one step snapshotDropFloor = 5 // and at least this many, so small counts do not twitch ) // countIsTrustworthy — the three pre-conditions, each with a scar behind it. A report failing ANY of // them is not evidence of anything: it neither alarms NOR updates the baseline, because comparing // against a number nobody could measure is how a detector invents its own findings. func (oc *OffsiteChecker) countIsTrustworthy(off *offsiteReport) bool { // 1. R-331 — a zero is not a zero when stats are unknown. Absent `stats_known` means the box // cannot answer. Every historical decrease in this hub's data is this shape. if !off.StatsKnown { return false } // 2. R-204/R-193 — the box DECLARES its own situation. A re-initialised, credential-less or // abandoned box reports a real, correct drop; alarming on it would be blaming the box for // telling the truth. if off.State != "" { return false } // 3. R-100 — PRESENCE IS NOT SUCCESS, and this file is where that lesson was learned. A count // from a failed or half-finished run is not a measurement of the store. "incomplete" (R-203) // is deliberately excluded too: a partial run legitimately counts less. if off.LastStatus != "ok" || off.LastSuccess == "" { return false } return true } // snapshotDropped reports whether `off` is a drop worth alarming about, against the remembered // baseline. Caller holds oc.mu. func (oc *OffsiteChecker) snapshotDropped(customerID string, off *offsiteReport) (bool, int, int) { prev, had := oc.lastCounts[customerID] if !had || prev <= 0 { return false, prev, off.SnapshotCount } drop := prev - off.SnapshotCount if drop < snapshotDropFloor { return false, prev, off.SnapshotCount } return float64(drop) > float64(prev)*snapshotDropFraction, prev, off.SnapshotCount } // emitSnapshotDrop — one event, at a severity inside the hub's exact vocabulary (08 §6.1). // // THE MESSAGE MUST NOT SAY THE DATA IS LOST, because after the 2026-09-01 measurement that is usually // false: the daily Storage Box snapshots are read-only to every account (proven, not cited) and hold // the older copy. It says what happened, what it means, and where the data still is. // // AND IT MUST NOT SAY THE DATA IS RECOVERABLE EITHER — R-434, fixed 2026-09-01, hub v0.111.1. The // sentence shipped that morning promised "so this is recoverable file-by-file". Measured the same day // (R-433): no snapshot is reachable from a sub-account by ANY name — 777,600 exact names in the // vendor's own format over nine days, zero hits, with a passing control; `/home` and `/.zfs` are // different filesystems and `/home/.zfs` does not exist. So the promise named a route nobody can walk. // // THE FIX IS A DELETION, NOT A REPLACEMENT, AND THAT IS THE WHOLE POINT. R-434's row said the fix was // blocked on R-433 — on knowing what IS true. It is not, if the promise is simply withdrawn: a // sentence that asserts neither loss nor recovery is true under EVERY possible answer to the provider // question, so it never needs a second rewrite. An alarm rewritten twice in a week is worse than one // rewritten once, because the operator learns its words do not mean anything. func (oc *OffsiteChecker) emitSnapshotDrop(customerID string, off *offsiteReport, prev, cur int) { message := fmt.Sprintf( "Customer %s: off-site backup count fell from %d to %d snapshot(s) in one report — more than "+ "retention can explain. The daily Storage Box snapshots are read-only and still hold the "+ "older copy. The route back out of them is not yet established, so treat this as neither "+ "confirmed data loss nor confirmed recovery. Get in touch before restoring anything, and "+ "check whether a deletion ran on the box.", customerID, prev, cur) details, _ := json.Marshal(map[string]any{ "customer_id": customerID, "previous_count": prev, "current_count": cur, "drop": prev - cur, "last_success": off.LastSuccess, "last_status": off.LastStatus, }) oc.logger.Printf("[INFO] Offsite snapshot drop: %s %d -> %d (offsite_snapshots_dropped)", customerID, prev, cur) if _, err := oc.store.SaveEvent(customerID, "offsite_snapshots_dropped", "error", message, string(details), "hub"); err != nil { oc.logger.Printf("[WARN] Failed to save offsite snapshot-drop event for %s: %v", customerID, err) return } if oc.onEvent != nil { oc.onEvent(customerID, "offsite_snapshots_dropped", "error", message, string(details), "hub") } } // GetDropState exposes the latch for tests. func (oc *OffsiteChecker) GetDropState(customerID string) string { oc.mu.Lock() defer oc.mu.Unlock() if s := oc.dropStates[customerID]; s != "" { return s } return "unknown" } // warnLegacyOnce logs the legacy degrade a single time per customer. Once, because this is a // steady-state condition until the box upgrades — a per-cycle line would be pure noise — but it must be // logged at all, so a fleet silently running on the old anchor is visible rather than assumed. func (oc *OffsiteChecker) warnLegacyOnce(customerID string) { oc.legacyMu.Lock() defer oc.legacyMu.Unlock() if oc.legacyWarned == nil { oc.legacyWarned = map[string]bool{} } if oc.legacyWarned[customerID] { return } oc.legacyWarned[customerID] = true if oc.logger != nil { oc.logger.Printf("[WARN] [offsite] %s: controller sends no last_success — staleness degraded to the last-ATTEMPT anchor (pre-v0.181.0 controller; a persistently failing tier will not go stale here until it upgrades)", customerID) } } // neverRanAnchor returns the newest hub-held timestamp from which a never-ran-but-applied tier's // staleness may be counted (zero when the hub holds neither — the pre-v0.5x legacy shape). func (oc *OffsiteChecker) neverRanAnchor(customerID string) time.Time { var anchor time.Time if info, err := oc.store.GetOneTimeSecretInfo(customerID); err == nil && info != nil && info.ConsumedAt.After(anchor) { anchor = info.ConsumedAt } if t, err := oc.store.LatestEscrowTimeForCustomer(customerID); err == nil && t.After(anchor) { anchor = t } return anchor } // Check evaluates every customer's latest report. Escalation-only emits; recovery re-arms silently. func (oc *OffsiteChecker) Check() { customers, err := oc.store.GetCustomers() if err != nil { oc.logger.Printf("[WARN] Offsite check failed: %v", err) return } oc.mu.Lock() defer oc.mu.Unlock() seen := make(map[string]bool, len(customers)) for _, c := range customers { // GetCustomers can return the same customer twice when two reports tie on received_at // (second-resolution timestamps) — process each customer once per sweep. if seen[c.CustomerID] { continue } seen[c.CustomerID] = true off := parseOffsite(c.ReportJSON) if off == nil { delete(oc.fillStates, c.CustomerID) // vanished object (disabled / downgraded) → re-arm delete(oc.staleStates, c.CustomerID) delete(oc.dropStates, c.CustomerID) delete(oc.lastCounts, c.CustomerID) // no baseline survives a vanished object continue } if oc.store.IsCustomerBlocked(c.CustomerID) { delete(oc.fillStates, c.CustomerID) delete(oc.staleStates, c.CustomerID) delete(oc.dropStates, c.CustomerID) delete(oc.lastCounts, c.CustomerID) continue } // FILL (quota>0 only) newBand := oc.fillBand(off) if bandRank(newBand) > bandRank(oc.fillStates[c.CustomerID]) { oc.emitFill(c.CustomerID, off, newBand) } oc.fillStates[c.CustomerID] = newBand // STALENESS (binary, warn-severity) newStale := "ok" if oc.isStale(c.CustomerID, off) { newStale = "stale" } if newStale == "stale" && oc.staleStates[c.CustomerID] != "stale" { oc.emitStale(c.CustomerID, off) } // Part-7: make the anchored never-ran evaluation VISIBLE once (first observation of the // shape), so a live newborn tier's deferral is provable from the log without spamming // every sweep. if off.LastRun == "" && newStale == "ok" && off.Enabled && off.EscrowState == "escrowed" { if _, known := oc.staleStates[c.CustomerID]; !known { oc.logger.Printf("[INFO] Offsite staleness: %s never-ran within the anchored threshold (anchor %s) — newborn tier, not stale", c.CustomerID, oc.neverRanAnchor(c.CustomerID).UTC().Format(time.RFC3339)) } } oc.staleStates[c.CustomerID] = newStale // SNAPSHOT-DROP (R-431). Escalation-only, exactly like the two signals above: one deletion // produces ONE alarm, not one per report cycle. The baseline then moves to the new value, so a // SECOND deletion later is still caught — from the new floor. // // An UNTRUSTWORTHY report is a no-op in both directions: no alarm, and the baseline is left // alone rather than being overwritten with a number nobody could measure. That is what stops // a `needs_credential` zero from becoming the baseline and making the RECOVERY look like a rise. if oc.countIsTrustworthy(off) { dropped, prev, cur := oc.snapshotDropped(c.CustomerID, off) if dropped && oc.dropStates[c.CustomerID] != "dropped" { oc.emitSnapshotDrop(c.CustomerID, off, prev, cur) oc.dropStates[c.CustomerID] = "dropped" } else if !dropped { oc.dropStates[c.CustomerID] = "ok" } oc.lastCounts[c.CustomerID] = cur } } for k := range oc.fillStates { if !seen[k] { delete(oc.fillStates, k) } } for k := range oc.staleStates { if !seen[k] { delete(oc.staleStates, k) } } for k := range oc.dropStates { if !seen[k] { delete(oc.dropStates, k) delete(oc.lastCounts, k) } } } // GetFillState / GetStaleState expose current states for tests. func (oc *OffsiteChecker) GetFillState(customerID string) string { oc.mu.Lock() defer oc.mu.Unlock() if s := oc.fillStates[customerID]; s != "" { return s } return "unknown" } func (oc *OffsiteChecker) GetStaleState(customerID string) string { oc.mu.Lock() defer oc.mu.Unlock() if s := oc.staleStates[customerID]; s != "" { return s } return "unknown" } func (oc *OffsiteChecker) emitFill(customerID string, off *offsiteReport, band string) { usedGB := off.RepoSizeBytes >> 30 pct := float64(off.RepoSizeBytes) * 100 / float64(int64(off.QuotaGB)<<30) var eventType, severity, message string switch band { case bandCritical: eventType, severity = "offsite_fill_critical", "critical" message = fmt.Sprintf("Customer %s: offsite backup at %.0f%% of its %d GB quota (%d GB used) — at 100%% new offsite runs are refused; consider the freeze lever or a bigger quota", customerID, pct, off.QuotaGB, usedGB) case bandWarning: eventType, severity = "offsite_fill_warning", "warning" message = fmt.Sprintf("Customer %s: offsite backup at %.0f%% of its %d GB quota (%d GB used)", customerID, pct, off.QuotaGB, usedGB) default: return } details, _ := json.Marshal(map[string]any{ "customer_id": customerID, "quota_gb": off.QuotaGB, "repo_size_bytes": off.RepoSizeBytes, "percent": pct, }) oc.logger.Printf("[INFO] Offsite fill: %s %.0f%% (%s)", customerID, pct, eventType) if _, err := oc.store.SaveEvent(customerID, eventType, severity, message, string(details), "hub"); err != nil { oc.logger.Printf("[WARN] Failed to save offsite fill event for %s: %v", customerID, err) return } if oc.onEvent != nil { oc.onEvent(customerID, eventType, severity, message, string(details), "hub") } } // staleAge describes WHY the tier is stale, in the terms the verdict actually used. // // R-100: this must not say "last run 8h ago" while alarming, which is what it did when the verdict // moved to the success anchor — a tier that RUNS nightly and FAILS nightly would have alarmed with a // fresh-looking timestamp, and the operator would have read a true alarm as a false one. The two cases // are genuinely different diagnoses and the message now separates them: // - runs are not happening at all → check the schedule/controller // - runs happen and FAIL → check the error; this is the case backup_failed also reports func (oc *OffsiteChecker) staleAge(off *offsiteReport) (age, hint string) { parse := func(v string) (time.Duration, bool) { t, err := time.Parse(time.RFC3339, v) if err != nil { return 0, false } return oc.now().Sub(t).Round(time.Hour), true } if off.LastSuccess == "" { if d, ok := parse(off.LastRun); ok { return fmt.Sprintf("no run has EVER succeeded (last attempt %s ago, status %q)", d, off.LastStatus), " Runs are happening and failing — check the error, not the schedule" } return "never ran", " The offsite leg is silently not running; check the controller/schedule" } d, ok := parse(off.LastSuccess) if !ok { return "last successful run at an unparseable time", " Check the controller/schedule" } if la, ok := parse(off.LastRun); ok && off.LastRun != off.LastSuccess { return fmt.Sprintf("last SUCCESSFUL run %s ago (it last attempted %s ago, status %q)", d, la, off.LastStatus), " Runs are happening and failing — check the error, not the schedule" } return fmt.Sprintf("last successful run %s ago", d), " The offsite leg is silently not running; check the controller/schedule" } func (oc *OffsiteChecker) emitStale(customerID string, off *offsiteReport) { age, hint := oc.staleAge(off) message := fmt.Sprintf("Customer %s: offsite backup is STALE — enabled + escrowed but %s (threshold %s).%s", customerID, age, oc.staleAfter, hint) details, _ := json.Marshal(map[string]any{ "customer_id": customerID, "last_run": off.LastRun, "last_success": off.LastSuccess, "last_status": off.LastStatus, "stale_after": oc.staleAfter.String(), }) oc.logger.Printf("[INFO] Offsite staleness: %s (%s)", customerID, age) if _, err := oc.store.SaveEvent(customerID, "offsite_stale", "warning", message, string(details), "hub"); err != nil { oc.logger.Printf("[WARN] Failed to save offsite staleness event for %s: %v", customerID, err) return } if oc.onEvent != nil { oc.onEvent(customerID, "offsite_stale", "warning", message, string(details), "hub") } }