b119301c6f
gates / gates (push) Successful in 2m48s
Hub code unreleased; ships with tomorrow's hub release. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
576 lines
28 KiB
Go
576 lines
28 KiB
Go
package monitor
|
||
|
||
import (
|
||
"encoding/json"
|
||
"fmt"
|
||
"log"
|
||
"sync"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
||
)
|
||
|
||
// OffsiteChecker (SLICE 4) watches each customer's offsite backup health from the controller report's
|
||
// `offsite` status object. Two independent signals, one checker (sibling of StorageFillChecker — same
|
||
// born/persistent, escalation-only emit, recovery re-arm shape; NOT bolted onto the disk checkers —
|
||
// different data source, different remedy text):
|
||
//
|
||
// - FILL: repo_size_bytes vs the shared-model soft quota (quota_gb>0) at warn 90% / crit 95% — the
|
||
// operator's early warning before the controller's own 100% run-refusal bites the customer.
|
||
// - SNAPSHOT-DROP (R-431): the reported snapshot count falls by more than retention can explain —
|
||
// the "something deleted this customer's off-site history" detector. See snapshotDrop below for
|
||
// why it lives HERE and not on the box, and for the measurement behind its threshold.
|
||
// - STALENESS: enabled + escrowed but no run in >48h (or never) — the silently-STUCK detector. A
|
||
// RECENTLY-failing offsite is NOT stale (backup_failed already alerts it); staleness is the
|
||
// complement: nothing is even trying. Pending/disabled targets are normal onboarding, never stale.
|
||
//
|
||
// Reports without an `offsite` object (pre-v0.109 controllers, offbox not enabled) are skipped nil-safe.
|
||
type OffsiteChecker struct {
|
||
store *store.Store
|
||
logger *log.Logger
|
||
onEvent EventNotifyFunc
|
||
staleAfter time.Duration
|
||
|
||
// legacyWarned tracks the one-shot R-100 legacy-degrade log, per customer.
|
||
legacyMu sync.Mutex
|
||
legacyWarned map[string]bool
|
||
now func() time.Time // injectable clock (tests)
|
||
|
||
mu sync.Mutex
|
||
fillStates map[string]string // customerID → fill band
|
||
staleStates map[string]string // customerID → "ok" | "stale"
|
||
// R-431. dropStates is the escalation latch ("ok" | "dropped"); lastCounts is the baseline the
|
||
// next sweep compares against. Only a TRUSTWORTHY report updates lastCounts — see snapshotDrop.
|
||
dropStates map[string]string
|
||
lastCounts map[string]int
|
||
}
|
||
|
||
const defaultOffsiteStaleAfter = 48 * time.Hour
|
||
|
||
// offsiteReport mirrors the controller report's `offsite` object (v0.109.0).
|
||
type offsiteReport struct {
|
||
Enabled bool `json:"enabled"`
|
||
EscrowState string `json:"escrow_state"`
|
||
LastRun string `json:"last_run"`
|
||
LastStatus string `json:"last_status"`
|
||
// LastSuccess (R-100) is the last run that actually SUCCEEDED — the staleness anchor. Absent on a
|
||
// controller older than v0.181.0; isStale degrades explicitly in that case, see there.
|
||
LastSuccess string `json:"last_success"`
|
||
SnapshotCount int `json:"snapshot_count"`
|
||
RepoSizeBytes int64 `json:"repo_size_bytes"`
|
||
QuotaGB int `json:"quota_gb"`
|
||
// StatsKnown (R-331/R-225) — the ONLY thing separating "this repository holds nothing" from
|
||
// "nobody has ever measured this repository". Both are `snapshot_count: 0` on the wire and they
|
||
// are opposite news. ABSENT on a pre-v0.225.0 controller, and absence means the box CANNOT
|
||
// ANSWER, never that the answer is no. snapshotDrop refuses to judge without it.
|
||
StatsKnown bool `json:"stats_known,omitempty"`
|
||
// State (R-204/R-193) — a DECLARED condition, empty on every healthy box. A box that has
|
||
// re-initialised, lost its credential or been abandoned reports a real, correct, large drop.
|
||
// The box knows its own situation; believe it rather than alarming on it.
|
||
State string `json:"state,omitempty"`
|
||
}
|
||
|
||
// NewOffsiteChecker builds the checker. Same seeding philosophy as StorageFillChecker: already-breached
|
||
// customers are left UNSEEDED so their first Check emits (born/persistent); the dispatcher's cooldown
|
||
// dedups a hub restart.
|
||
func NewOffsiteChecker(s *store.Store, staleAfter time.Duration, onEvent EventNotifyFunc, logger *log.Logger) *OffsiteChecker {
|
||
if staleAfter <= 0 {
|
||
staleAfter = defaultOffsiteStaleAfter
|
||
}
|
||
oc := &OffsiteChecker{
|
||
store: s, logger: logger, onEvent: onEvent, staleAfter: staleAfter, now: time.Now,
|
||
fillStates: make(map[string]string), staleStates: make(map[string]string),
|
||
dropStates: make(map[string]string), lastCounts: make(map[string]int),
|
||
}
|
||
customers, err := s.GetCustomers()
|
||
if err != nil {
|
||
logger.Printf("[WARN] Offsite checker: failed to seed states: %v", err)
|
||
return oc
|
||
}
|
||
var seeded int
|
||
for _, c := range customers {
|
||
off := parseOffsite(c.ReportJSON)
|
||
if off == nil || s.IsCustomerBlocked(c.CustomerID) {
|
||
continue
|
||
}
|
||
if band := oc.fillBand(off); band == bandOK {
|
||
oc.fillStates[c.CustomerID] = bandOK
|
||
seeded++
|
||
}
|
||
if !oc.isStale(c.CustomerID, off) {
|
||
oc.staleStates[c.CustomerID] = "ok"
|
||
}
|
||
// R-431: seed the baseline from the current report so a hub restart does not read the first
|
||
// sweep as a drop from nothing. Only a trustworthy report seeds — an untrustworthy one leaves
|
||
// no baseline, and no baseline means no verdict.
|
||
if oc.countIsTrustworthy(off) {
|
||
oc.lastCounts[c.CustomerID] = off.SnapshotCount
|
||
oc.dropStates[c.CustomerID] = "ok"
|
||
}
|
||
}
|
||
logger.Printf("[INFO] Offsite checker initialized: fill warn=90%% crit=95%%, stale after %s, %d ok-seeded", staleAfter, seeded)
|
||
return oc
|
||
}
|
||
|
||
func parseOffsite(reportJSON string) *offsiteReport {
|
||
var r struct {
|
||
Offsite *offsiteReport `json:"offsite"`
|
||
}
|
||
if json.Unmarshal([]byte(reportJSON), &r) != nil {
|
||
return nil
|
||
}
|
||
return r.Offsite // nil when absent (old controller / offbox not enabled) — the caller skips
|
||
}
|
||
|
||
// fillBand maps the quota usage to a band. quota<=0 (dedicated/unset) never alerts.
|
||
func (oc *OffsiteChecker) fillBand(off *offsiteReport) string {
|
||
if off.QuotaGB <= 0 || off.RepoSizeBytes <= 0 {
|
||
return bandOK
|
||
}
|
||
pct := float64(off.RepoSizeBytes) * 100 / float64(int64(off.QuotaGB)<<30)
|
||
return bandForPercent(pct, 90, 95)
|
||
}
|
||
|
||
// isStale: enabled + ESCROWED (the only state where runs are expected) with no SUCCESSFUL run in
|
||
// >staleAfter (or never ran, ANCHORED — see below). Pending/disabled = normal onboarding, never stale.
|
||
//
|
||
// R-100 — PRESENCE IS NOT SUCCESS. This counted from `LastRun`, which the controller writes
|
||
// unconditionally at the end of every run INCLUDING failures. So it asked "how long since we last
|
||
// TRIED", and a tier failing on every single run refreshed the clock nightly and read as perfectly
|
||
// fresh forever. It now counts from `LastSuccess`, which only a successful run advances.
|
||
//
|
||
// The old comment here said "a recent-but-failing run is NOT stale (backup_failed owns that signal)",
|
||
// and that WAS true — `backup_failed` does fire, nightly, and reaches the operator. The defect is not
|
||
// silence, it is DEFEATED DEFENCE IN DEPTH: this checker is the hub-side, pull-based net that exists to
|
||
// be independent of controller-PUSHED events, and anchoring it on a field the failing controller keeps
|
||
// refreshing made it depend on the very thing it backs up. F-HUB (the hub dropping an event under
|
||
// SQLITE_BUSY, no retry) is exactly that loss. INVARIANT pinned by TestIsStale_* in offsite_r100_test.go.
|
||
//
|
||
// Part-7 (v0.73.0) — the never-ran branch no longer fires on sight. The 2026-07-23 cry-wolf:
|
||
// demo-hp's tier was repaired and escrowed at 10:01Z and offsite_stale fired MINUTES later
|
||
// (`last_run:"" … threshold 48h`), because "enabled + escrowed + never ran" had no time anchor.
|
||
// Boundary: reaching this code at all means the latest report CARRIES the offsite object — i.e.
|
||
// the v0.72.0 delivery state is `applied` (Check nil-skips everything else); pre-applied
|
||
// never-ran shapes are offsite_delivery_stuck's alone — ONE STATE, ONE OWNER, never both.
|
||
// Anchor: the newest of one_time_secrets.consumed_at (delivery completed) and the customer's
|
||
// escrow-blob timestamp (host_escrow.updated_at/created_at — runs become POSSIBLE only at the
|
||
// ceremony). The EXISTING staleAfter threshold, anchored there, IS the grace — no new knob.
|
||
func (oc *OffsiteChecker) isStale(customerID string, off *offsiteReport) bool {
|
||
if !off.Enabled || off.EscrowState != "escrowed" {
|
||
return false
|
||
}
|
||
// NEVER RAN AT ALL — unchanged v0.73.0 behaviour, and it must stay unchanged. A newborn box has no
|
||
// LastRun and no LastSuccess; alarming it on sight is the 2026-07-23 cry-wolf. Checked on LastRun
|
||
// (not LastSuccess) deliberately: LastRun is the "has anything ever happened here" signal, and a box
|
||
// whose first run FAILED has a LastRun but no LastSuccess — that is a run, not a newborn, and it
|
||
// belongs on the anchored path below.
|
||
if off.LastRun == "" {
|
||
anchor := oc.neverRanAnchor(customerID)
|
||
if anchor.IsZero() {
|
||
return true // legacy shape (no secret timestamps, no escrow row) — fail toward visibility, as before
|
||
}
|
||
return oc.now().Sub(anchor) > oc.staleAfter
|
||
}
|
||
|
||
// LEGACY CONTROLLER — it has run, but sends no last_success (pre-v0.181.0). Handle it EXPLICITLY,
|
||
// in the direction that preserves known behaviour: count from LastRun exactly as before.
|
||
// - treating absence as FAILURE would alarm every un-upgraded box in the fleet at once;
|
||
// - treating it as SUCCESS is the bug.
|
||
// Preserving today's behaviour is correct here; the controller version floor drives the upgrade.
|
||
// Same degrade direction as R-88 Part 2's age_state, and for the same reason.
|
||
if off.LastSuccess == "" {
|
||
oc.warnLegacyOnce(customerID)
|
||
t, err := time.Parse(time.RFC3339, off.LastRun)
|
||
if err != nil {
|
||
return true // unparseable = unknown-old — fail toward visibility
|
||
}
|
||
return oc.now().Sub(t) > oc.staleAfter
|
||
}
|
||
|
||
// THE ANCHORED VERDICT. Note what is NOT consulted: LastStatus. A single failed night does not make
|
||
// a tier stale — the threshold simply keeps running from the last good run, so one blip is tolerated
|
||
// and a persistent failure is caught. Reading LastStatus here ("error ⇒ stale") would alarm on every
|
||
// transient blip, which is the F-A1 noise path that trains an operator to ignore the alarm.
|
||
// `running` is a real wire value (a report captured mid-run) and is likewise none of our business.
|
||
t, err := time.Parse(time.RFC3339, off.LastSuccess)
|
||
if err != nil {
|
||
return true // unparseable = unknown-old — fail toward visibility
|
||
}
|
||
return oc.now().Sub(t) > oc.staleAfter
|
||
}
|
||
|
||
// ── SNAPSHOT-DROP (R-431) ───────────────────────────────────────────────────────────────────────
|
||
//
|
||
// WHY THIS LIVES ON THE HUB AND NOT ON THE BOX. The thing being detected is a box deleting its own
|
||
// off-site backups — so a detector living on that box is a detector the same event can silence. The
|
||
// hub already receives the count in every report and already keeps the history, so it can notice
|
||
// without trusting the box's judgement. The box still REPORTS the number, and a sophisticated
|
||
// attacker could lie about it; that is a far higher bar than deleting files, and the weekly integrity
|
||
// check (R-359) would then disagree with the lie.
|
||
//
|
||
// WHAT IT IS FOR. R-95: the restic credential can delete its own repository, and the sub-account API
|
||
// has no append-only axis, so prevention needs a transport change. The Storage Box's daily ZFS
|
||
// snapshots bound the loss — MEASURED 2026-09-01: a write into `/.zfs/snapshot` is REFUSED
|
||
// (`dest open …: Failure`) while the same write to the account home succeeds. So the remedy exists;
|
||
// what was missing was NOTICING, and this is that.
|
||
//
|
||
// THE THRESHOLD, AND THE MEASUREMENT IT CAME FROM. Measured over the hub's own `reports` table on
|
||
// 2026-09-01: 12 898 reports, 4 customers, 2026-06-05 → 2026-09-01.
|
||
//
|
||
// * In the whole history there are NINE decreases, and EVERY ONE of them lands exactly on ZERO
|
||
// (36→0, 18→0 ×2, 15→0, 12→0, 8→0, 3→0). There is not one gradual retention decrease anywhere.
|
||
// * Every one of those nine predates `stats_known`, i.e. they are the R-331 shape — a zero that
|
||
// means "could not measure", not "nothing is there". Several carry a declared State
|
||
// (`needs_credential`, `awaiting_recovery_key`), which says so outright.
|
||
// * In the window where `stats_known` is TRUE (380 reports across both live boxes) there are ZERO
|
||
// decreases: demo-felhom sat flat at 10, demo-hp moved 67→68→69. Only rises.
|
||
//
|
||
// So observed retention churn gives NOTHING to calibrate against, and saying so is the honest answer
|
||
// rather than inventing a number (R-401's lesson). The threshold is therefore reasoned from what
|
||
// retention CAN do, not from what it was seen to do:
|
||
//
|
||
// the box runs `forget --keep-daily 7 --keep-weekly 4 --keep-monthly 6 --group-by host,tags`
|
||
// over ~8 apps. On a boundary day several groups can expire at once, so a legitimate pass can
|
||
// plausibly remove low double digits. **What it can NEVER do is halve the total**: keeping 7 daily
|
||
// + 4 weekly + 6 monthly per group is a floor, and a mass deletion goes to ~0.
|
||
//
|
||
// Hence: **a fall of MORE THAN HALF the previous count, and at least 5 snapshots.** The 50% cannot be
|
||
// reached by retention; the floor of 5 stops a tiny-count box alarming on ordinary ageing. It is
|
||
// deliberately NOT sensitive — a detector that cries wolf is switched off within a fortnight, and
|
||
// this project has proved that twice in a week.
|
||
// WHAT THIS DETECTOR DOES NOT SEE — R-435, and it must be read wherever "an unexplained fall is
|
||
// noticed within a day" is claimed, because that claim is true only of falls above the fraction.
|
||
//
|
||
// **It sees a MASS deletion. It does not see ONE APP being wiped.** Worked on the live fleet
|
||
// 2026-09-01: demo-hp's baseline is 69 snapshots across 9 apps, so ~35 must go before this speaks;
|
||
// one app's tag is ~9 and is invisible. And `offbox.go:1388` runs `forget --prune` **grouped by
|
||
// host,tags** — a per-tag wipe is exactly the shape a faulty retention or a targeted deletion
|
||
// produces, so the blind spot sits on the most likely single-app failure, not an exotic one.
|
||
//
|
||
// THIS IS DELIBERATE AND THE THRESHOLD SHOULD NOT BE LOWERED TO "FIX" IT. The reasoning is below: a
|
||
// detector that cries wolf is switched off within a fortnight, and this project has proved that
|
||
// twice in a week. What is NOT acceptable is claiming coverage this does not have. Anyone adding
|
||
// per-app detection should add a SECOND signal keyed on the per-tag count, not move these numbers.
|
||
const (
|
||
snapshotDropFraction = 0.5 // more than half the history gone in one step
|
||
snapshotDropFloor = 5 // and at least this many, so small counts do not twitch
|
||
)
|
||
|
||
// countIsTrustworthy — the three pre-conditions, each with a scar behind it. A report failing ANY of
|
||
// them is not evidence of anything: it neither alarms NOR updates the baseline, because comparing
|
||
// against a number nobody could measure is how a detector invents its own findings.
|
||
func (oc *OffsiteChecker) countIsTrustworthy(off *offsiteReport) bool {
|
||
// 1. R-331 — a zero is not a zero when stats are unknown. Absent `stats_known` means the box
|
||
// cannot answer. Every historical decrease in this hub's data is this shape.
|
||
if !off.StatsKnown {
|
||
return false
|
||
}
|
||
// 2. R-204/R-193 — the box DECLARES its own situation. A re-initialised, credential-less or
|
||
// abandoned box reports a real, correct drop; alarming on it would be blaming the box for
|
||
// telling the truth.
|
||
if off.State != "" {
|
||
return false
|
||
}
|
||
// 3. R-100 — PRESENCE IS NOT SUCCESS, and this file is where that lesson was learned. A count
|
||
// from a failed or half-finished run is not a measurement of the store. "incomplete" (R-203)
|
||
// is deliberately excluded too: a partial run legitimately counts less.
|
||
if off.LastStatus != "ok" || off.LastSuccess == "" {
|
||
return false
|
||
}
|
||
return true
|
||
}
|
||
|
||
// snapshotDropped reports whether `off` is a drop worth alarming about, against the remembered
|
||
// baseline. Caller holds oc.mu.
|
||
func (oc *OffsiteChecker) snapshotDropped(customerID string, off *offsiteReport) (bool, int, int) {
|
||
prev, had := oc.lastCounts[customerID]
|
||
if !had || prev <= 0 {
|
||
return false, prev, off.SnapshotCount
|
||
}
|
||
drop := prev - off.SnapshotCount
|
||
if drop < snapshotDropFloor {
|
||
return false, prev, off.SnapshotCount
|
||
}
|
||
return float64(drop) > float64(prev)*snapshotDropFraction, prev, off.SnapshotCount
|
||
}
|
||
|
||
// emitSnapshotDrop — one event, at a severity inside the hub's exact vocabulary (08 §6.1).
|
||
//
|
||
// THE MESSAGE MUST NOT SAY THE DATA IS LOST, because after the 2026-09-01 measurement that is usually
|
||
// false: the daily Storage Box snapshots are read-only to every account (proven, not cited) and hold
|
||
// the older copy. It says what happened, what it means, and where the data still is.
|
||
//
|
||
// AND IT MUST NOT SAY THE DATA IS RECOVERABLE EITHER — R-434, fixed 2026-09-01, hub v0.111.1. The
|
||
// sentence shipped that morning promised "so this is recoverable file-by-file". Measured the same day
|
||
// (R-433): no snapshot is reachable from a sub-account by ANY name — 777,600 exact names in the
|
||
// vendor's own format over nine days, zero hits, with a passing control; `/home` and `/.zfs` are
|
||
// different filesystems and `/home/.zfs` does not exist. So the promise named a route nobody can walk.
|
||
//
|
||
// THE FIX IS A DELETION, NOT A REPLACEMENT, AND THAT IS THE WHOLE POINT. R-434's row said the fix was
|
||
// blocked on R-433 — on knowing what IS true. It is not, if the promise is simply withdrawn: a
|
||
// sentence that asserts neither loss nor recovery is true under EVERY possible answer to the provider
|
||
// question, so it never needs a second rewrite. An alarm rewritten twice in a week is worse than one
|
||
// rewritten once, because the operator learns its words do not mean anything.
|
||
func (oc *OffsiteChecker) emitSnapshotDrop(customerID string, off *offsiteReport, prev, cur int) {
|
||
message := fmt.Sprintf(
|
||
"Customer %s: off-site backup count fell from %d to %d snapshot(s) in one report — more than "+
|
||
"retention can explain. The daily Storage Box snapshots are read-only and still hold the "+
|
||
"older copy. The route back out of them is not yet established, so treat this as neither "+
|
||
"confirmed data loss nor confirmed recovery. Get in touch before restoring anything, and "+
|
||
"check whether a deletion ran on the box.",
|
||
customerID, prev, cur)
|
||
details, _ := json.Marshal(map[string]any{
|
||
"customer_id": customerID, "previous_count": prev, "current_count": cur,
|
||
"drop": prev - cur, "last_success": off.LastSuccess, "last_status": off.LastStatus,
|
||
})
|
||
oc.logger.Printf("[INFO] Offsite snapshot drop: %s %d -> %d (offsite_snapshots_dropped)", customerID, prev, cur)
|
||
if _, err := oc.store.SaveEvent(customerID, "offsite_snapshots_dropped", "error", message, string(details), "hub"); err != nil {
|
||
oc.logger.Printf("[WARN] Failed to save offsite snapshot-drop event for %s: %v", customerID, err)
|
||
return
|
||
}
|
||
if oc.onEvent != nil {
|
||
oc.onEvent(customerID, "offsite_snapshots_dropped", "error", message, string(details), "hub")
|
||
}
|
||
}
|
||
|
||
// GetDropState exposes the latch for tests.
|
||
func (oc *OffsiteChecker) GetDropState(customerID string) string {
|
||
oc.mu.Lock()
|
||
defer oc.mu.Unlock()
|
||
if s := oc.dropStates[customerID]; s != "" {
|
||
return s
|
||
}
|
||
return "unknown"
|
||
}
|
||
|
||
// warnLegacyOnce logs the legacy degrade a single time per customer. Once, because this is a
|
||
// steady-state condition until the box upgrades — a per-cycle line would be pure noise — but it must be
|
||
// logged at all, so a fleet silently running on the old anchor is visible rather than assumed.
|
||
func (oc *OffsiteChecker) warnLegacyOnce(customerID string) {
|
||
oc.legacyMu.Lock()
|
||
defer oc.legacyMu.Unlock()
|
||
if oc.legacyWarned == nil {
|
||
oc.legacyWarned = map[string]bool{}
|
||
}
|
||
if oc.legacyWarned[customerID] {
|
||
return
|
||
}
|
||
oc.legacyWarned[customerID] = true
|
||
if oc.logger != nil {
|
||
oc.logger.Printf("[WARN] [offsite] %s: controller sends no last_success — staleness degraded to the last-ATTEMPT anchor (pre-v0.181.0 controller; a persistently failing tier will not go stale here until it upgrades)", customerID)
|
||
}
|
||
}
|
||
|
||
// neverRanAnchor returns the newest hub-held timestamp from which a never-ran-but-applied tier's
|
||
// staleness may be counted (zero when the hub holds neither — the pre-v0.5x legacy shape).
|
||
func (oc *OffsiteChecker) neverRanAnchor(customerID string) time.Time {
|
||
var anchor time.Time
|
||
if info, err := oc.store.GetOneTimeSecretInfo(customerID); err == nil && info != nil && info.ConsumedAt.After(anchor) {
|
||
anchor = info.ConsumedAt
|
||
}
|
||
if t, err := oc.store.LatestEscrowTimeForCustomer(customerID); err == nil && t.After(anchor) {
|
||
anchor = t
|
||
}
|
||
return anchor
|
||
}
|
||
|
||
// Check evaluates every customer's latest report. Escalation-only emits; recovery re-arms silently.
|
||
func (oc *OffsiteChecker) Check() {
|
||
customers, err := oc.store.GetCustomers()
|
||
if err != nil {
|
||
oc.logger.Printf("[WARN] Offsite check failed: %v", err)
|
||
return
|
||
}
|
||
oc.mu.Lock()
|
||
defer oc.mu.Unlock()
|
||
|
||
seen := make(map[string]bool, len(customers))
|
||
for _, c := range customers {
|
||
// GetCustomers can return the same customer twice when two reports tie on received_at
|
||
// (second-resolution timestamps) — process each customer once per sweep.
|
||
if seen[c.CustomerID] {
|
||
continue
|
||
}
|
||
seen[c.CustomerID] = true
|
||
off := parseOffsite(c.ReportJSON)
|
||
if off == nil {
|
||
// R-243: off-site gone → a raised escrow-pending alarm clears.
|
||
oc.checkEscrowPending(c.CustomerID, nil)
|
||
delete(oc.fillStates, c.CustomerID) // vanished object (disabled / downgraded) → re-arm
|
||
delete(oc.staleStates, c.CustomerID)
|
||
delete(oc.dropStates, c.CustomerID)
|
||
delete(oc.lastCounts, c.CustomerID) // no baseline survives a vanished object
|
||
continue
|
||
}
|
||
if oc.store.IsCustomerBlocked(c.CustomerID) {
|
||
delete(oc.fillStates, c.CustomerID)
|
||
delete(oc.staleStates, c.CustomerID)
|
||
delete(oc.dropStates, c.CustomerID)
|
||
delete(oc.lastCounts, c.CustomerID)
|
||
continue
|
||
}
|
||
|
||
// FILL (quota>0 only)
|
||
newBand := oc.fillBand(off)
|
||
if bandRank(newBand) > bandRank(oc.fillStates[c.CustomerID]) {
|
||
oc.emitFill(c.CustomerID, off, newBand)
|
||
}
|
||
oc.fillStates[c.CustomerID] = newBand
|
||
|
||
// STALENESS (binary, warn-severity)
|
||
newStale := "ok"
|
||
if oc.isStale(c.CustomerID, off) {
|
||
newStale = "stale"
|
||
}
|
||
if newStale == "stale" && oc.staleStates[c.CustomerID] != "stale" {
|
||
oc.emitStale(c.CustomerID, off)
|
||
}
|
||
// Part-7: make the anchored never-ran evaluation VISIBLE once (first observation of the
|
||
// shape), so a live newborn tier's deferral is provable from the log without spamming
|
||
// every sweep.
|
||
if off.LastRun == "" && newStale == "ok" && off.Enabled && off.EscrowState == "escrowed" {
|
||
if _, known := oc.staleStates[c.CustomerID]; !known {
|
||
oc.logger.Printf("[INFO] Offsite staleness: %s never-ran within the anchored threshold (anchor %s) — newborn tier, not stale",
|
||
c.CustomerID, oc.neverRanAnchor(c.CustomerID).UTC().Format(time.RFC3339))
|
||
}
|
||
}
|
||
oc.staleStates[c.CustomerID] = newStale
|
||
|
||
// R-243 (offsite_escrow_pending.go): the state the staleness signal deliberately leaves out — off-site ON,
|
||
// escrow never done, so no run ever starts. Persisted, weekly, operator-only.
|
||
oc.checkEscrowPending(c.CustomerID, off)
|
||
|
||
// SNAPSHOT-DROP (R-431). Escalation-only, exactly like the two signals above: one deletion
|
||
// produces ONE alarm, not one per report cycle. The baseline then moves to the new value, so a
|
||
// SECOND deletion later is still caught — from the new floor.
|
||
//
|
||
// An UNTRUSTWORTHY report is a no-op in both directions: no alarm, and the baseline is left
|
||
// alone rather than being overwritten with a number nobody could measure. That is what stops
|
||
// a `needs_credential` zero from becoming the baseline and making the RECOVERY look like a rise.
|
||
if oc.countIsTrustworthy(off) {
|
||
dropped, prev, cur := oc.snapshotDropped(c.CustomerID, off)
|
||
if dropped && oc.dropStates[c.CustomerID] != "dropped" {
|
||
oc.emitSnapshotDrop(c.CustomerID, off, prev, cur)
|
||
oc.dropStates[c.CustomerID] = "dropped"
|
||
} else if !dropped {
|
||
oc.dropStates[c.CustomerID] = "ok"
|
||
}
|
||
oc.lastCounts[c.CustomerID] = cur
|
||
}
|
||
}
|
||
for k := range oc.fillStates {
|
||
if !seen[k] {
|
||
delete(oc.fillStates, k)
|
||
}
|
||
}
|
||
for k := range oc.staleStates {
|
||
if !seen[k] {
|
||
delete(oc.staleStates, k)
|
||
}
|
||
}
|
||
for k := range oc.dropStates {
|
||
if !seen[k] {
|
||
delete(oc.dropStates, k)
|
||
delete(oc.lastCounts, k)
|
||
}
|
||
}
|
||
}
|
||
|
||
// GetFillState / GetStaleState expose current states for tests.
|
||
func (oc *OffsiteChecker) GetFillState(customerID string) string {
|
||
oc.mu.Lock()
|
||
defer oc.mu.Unlock()
|
||
if s := oc.fillStates[customerID]; s != "" {
|
||
return s
|
||
}
|
||
return "unknown"
|
||
}
|
||
|
||
func (oc *OffsiteChecker) GetStaleState(customerID string) string {
|
||
oc.mu.Lock()
|
||
defer oc.mu.Unlock()
|
||
if s := oc.staleStates[customerID]; s != "" {
|
||
return s
|
||
}
|
||
return "unknown"
|
||
}
|
||
|
||
func (oc *OffsiteChecker) emitFill(customerID string, off *offsiteReport, band string) {
|
||
usedGB := off.RepoSizeBytes >> 30
|
||
pct := float64(off.RepoSizeBytes) * 100 / float64(int64(off.QuotaGB)<<30)
|
||
var eventType, severity, message string
|
||
switch band {
|
||
case bandCritical:
|
||
eventType, severity = "offsite_fill_critical", "critical"
|
||
message = fmt.Sprintf("Customer %s: offsite backup at %.0f%% of its %d GB quota (%d GB used) — at 100%% new offsite runs are refused; consider the freeze lever or a bigger quota", customerID, pct, off.QuotaGB, usedGB)
|
||
case bandWarning:
|
||
eventType, severity = "offsite_fill_warning", "warning"
|
||
message = fmt.Sprintf("Customer %s: offsite backup at %.0f%% of its %d GB quota (%d GB used)", customerID, pct, off.QuotaGB, usedGB)
|
||
default:
|
||
return
|
||
}
|
||
details, _ := json.Marshal(map[string]any{
|
||
"customer_id": customerID, "quota_gb": off.QuotaGB, "repo_size_bytes": off.RepoSizeBytes, "percent": pct,
|
||
})
|
||
oc.logger.Printf("[INFO] Offsite fill: %s %.0f%% (%s)", customerID, pct, eventType)
|
||
if _, err := oc.store.SaveEvent(customerID, eventType, severity, message, string(details), "hub"); err != nil {
|
||
oc.logger.Printf("[WARN] Failed to save offsite fill event for %s: %v", customerID, err)
|
||
return
|
||
}
|
||
if oc.onEvent != nil {
|
||
oc.onEvent(customerID, eventType, severity, message, string(details), "hub")
|
||
}
|
||
}
|
||
|
||
// staleAge describes WHY the tier is stale, in the terms the verdict actually used.
|
||
//
|
||
// R-100: this must not say "last run 8h ago" while alarming, which is what it did when the verdict
|
||
// moved to the success anchor — a tier that RUNS nightly and FAILS nightly would have alarmed with a
|
||
// fresh-looking timestamp, and the operator would have read a true alarm as a false one. The two cases
|
||
// are genuinely different diagnoses and the message now separates them:
|
||
// - runs are not happening at all → check the schedule/controller
|
||
// - runs happen and FAIL → check the error; this is the case backup_failed also reports
|
||
func (oc *OffsiteChecker) staleAge(off *offsiteReport) (age, hint string) {
|
||
parse := func(v string) (time.Duration, bool) {
|
||
t, err := time.Parse(time.RFC3339, v)
|
||
if err != nil {
|
||
return 0, false
|
||
}
|
||
return oc.now().Sub(t).Round(time.Hour), true
|
||
}
|
||
if off.LastSuccess == "" {
|
||
if d, ok := parse(off.LastRun); ok {
|
||
return fmt.Sprintf("no run has EVER succeeded (last attempt %s ago, status %q)", d, off.LastStatus),
|
||
" Runs are happening and failing — check the error, not the schedule"
|
||
}
|
||
return "never ran", " The offsite leg is silently not running; check the controller/schedule"
|
||
}
|
||
d, ok := parse(off.LastSuccess)
|
||
if !ok {
|
||
return "last successful run at an unparseable time", " Check the controller/schedule"
|
||
}
|
||
if la, ok := parse(off.LastRun); ok && off.LastRun != off.LastSuccess {
|
||
return fmt.Sprintf("last SUCCESSFUL run %s ago (it last attempted %s ago, status %q)", d, la, off.LastStatus),
|
||
" Runs are happening and failing — check the error, not the schedule"
|
||
}
|
||
return fmt.Sprintf("last successful run %s ago", d),
|
||
" The offsite leg is silently not running; check the controller/schedule"
|
||
}
|
||
|
||
func (oc *OffsiteChecker) emitStale(customerID string, off *offsiteReport) {
|
||
age, hint := oc.staleAge(off)
|
||
message := fmt.Sprintf("Customer %s: offsite backup is STALE — enabled + escrowed but %s (threshold %s).%s", customerID, age, oc.staleAfter, hint)
|
||
details, _ := json.Marshal(map[string]any{
|
||
"customer_id": customerID, "last_run": off.LastRun, "last_success": off.LastSuccess,
|
||
"last_status": off.LastStatus, "stale_after": oc.staleAfter.String(),
|
||
})
|
||
oc.logger.Printf("[INFO] Offsite staleness: %s (%s)", customerID, age)
|
||
if _, err := oc.store.SaveEvent(customerID, "offsite_stale", "warning", message, string(details), "hub"); err != nil {
|
||
oc.logger.Printf("[WARN] Failed to save offsite staleness event for %s: %v", customerID, err)
|
||
return
|
||
}
|
||
if oc.onEvent != nil {
|
||
oc.onEvent(customerID, "offsite_stale", "warning", message, string(details), "hub")
|
||
}
|
||
}
|