package monitor import ( "context" "encoding/json" "errors" "fmt" "log" "sync" "time" "gitea.dooplex.hu/admin/felhom-hub/internal/tenantsync" ) // usageReader is the ep0 datastore-usage seam (satisfied by *tenantsync.Client). Narrow so tests inject // a fake and the checker never depends on the full tenantsync surface. type usageReader interface { Usage(ctx context.Context) (tenantsync.BoxUsage, error) } // PBSDRBoxChecker (v0.65.0) watches the PBS DR datastore (felhom-offsite on ep0) fill — the sibling of // OffsiteBoxChecker over a DIFFERENT source: the ep0 read-only `usage` op (df on the datastore path), NOT // the Hetzner API. FILL ONLY (PBS DR uses namespaces, not soft quotas — no oversubscription concept). Same // shape: 15-min throttle, cached snapshot, escalation-only emit + silent recovery re-arm, degraded-on- // failure honesty. THREE snapshot states: // - "ok" → bands drive alerts. // - "unavailable" → the endpoint script predates the usage op (≤ v1.1.0) → ErrUsageUnsupported. An // EXPECTED pre-update condition: neutral, NO alert, logged once. The gauge shows n/a. // - "degraded" → exec failed/timed out → keep the last snapshot, NO band transition (missing ≠ 0%). // // Events carry the customer-less scope "pbsdr-box" → operator channel ONLY (processCustomer no-ops on it); // NO SaveEvent (no customer row to key it to). Deliberately NOT bolted onto OffsiteBoxChecker. type PBSDRBoxChecker struct { reader usageReader fillWarn float64 fillCrit float64 onEvent EventNotifyFunc logger *log.Logger now func() time.Time // injectable clock (tests) reachFailThreshold int // consecutive failed fetch windows before reporting unreachable mu sync.Mutex snap PBSBoxSnapshot haveSnap bool lastFetch time.Time fillBand string // "" until first ok eval → born/persistent; only an "ok" fetch updates it unavailLogged bool // log the "unavailable" state ONCE, not per sweep // REACHABILITY (v0.106.0, R-339) — a SECOND, independent signal, deliberately not part of the // fill state above. Fill answers "how full is the store"; this answers "can the hub see the store // at all". They are separate because a missing reading must never move a fill band (missing ≠ 0%), // which is exactly what made a 9 h 37 m ep0 outage silent on 2026-08-18. reachFails int // consecutive failed fetch windows (NOT sweeps — lastFetch throttles both) reachAlerted bool // an unreachable event has been emitted for the CURRENT outage reachFirstFail time.Time // start of the current outage → the honest blind_for duration lastOK time.Time // last successful read; zero = never (born blind — do NOT fabricate one) } // pbsdrBoxScope is the customer-less cooldown/display key for PBS-DR box events. const pbsdrBoxScope = "pbsdr-box" // PBS-DR snapshot states. const ( PBSStateOK = "ok" PBSStateUnavailable = "unavailable" PBSStateDegraded = "degraded" ) const ( defaultPBSDRBoxFillWarnPercent = 80.0 defaultPBSDRBoxFillCritPercent = 90.0 // defaultBoxUnreachableWindows is the consecutive failed 15-minute fetch windows before the hub // reports a box unreachable. Three windows is ≈30–45 min depending on where the first failure // lands relative to the last success — long enough that a single blip or a brief API wobble says // nothing, short enough that a real outage is reported inside the hour. defaultBoxUnreachableWindows = 3 ) // PBSBoxSnapshot is the cached PBS-DR datastore aggregate the web layer renders. Sizes in BYTES. type PBSBoxSnapshot struct { CapacityBytes int64 UsedBytes int64 FillPercent float64 FillBand string // "" | ok | warning | critical State string // ok | unavailable | degraded FetchedAt time.Time } // NewPBSDRBoxChecker builds the checker (defaults 80/90 on invalid thresholds). Nothing seeded → the // first ok Check that finds a breach emits (born/persistent; the dispatcher's 1 h cooldown dedups a restart). func NewPBSDRBoxChecker(reader usageReader, fillWarn, fillCrit float64, reachWindows int, onEvent EventNotifyFunc, logger *log.Logger) *PBSDRBoxChecker { if fillWarn <= 0 || fillWarn >= 100 { fillWarn = defaultPBSDRBoxFillWarnPercent } if fillCrit <= 0 || fillCrit > 100 || fillCrit <= fillWarn { fillCrit = defaultPBSDRBoxFillCritPercent } if fillCrit <= fillWarn { fillWarn, fillCrit = defaultPBSDRBoxFillWarnPercent, defaultPBSDRBoxFillCritPercent } if reachWindows <= 0 { reachWindows = defaultBoxUnreachableWindows } c := &PBSDRBoxChecker{ reader: reader, fillWarn: fillWarn, fillCrit: fillCrit, reachFailThreshold: reachWindows, onEvent: onEvent, logger: logger, now: time.Now, } // The reachability threshold is in this line deliberately: a running hub can be asked what it is // configured to do without anyone reading the config file, and its ABSENCE from the pod log is how // you would notice the parameter never reached the checker. logger.Printf("[INFO] PBS-DR box checker initialized: fill warn=%.0f%% crit=%.0f%%, unreachable after %d consecutive failed reads, refresh %s", fillWarn, fillCrit, reachWindows, boxFetchInterval) return c } // Check runs on the 60 s sweep; fetches (throttled) and emits on each fill escalation. func (c *PBSDRBoxChecker) Check() { c.mu.Lock() defer c.mu.Unlock() if c.haveSnap && c.now().Sub(c.lastFetch) < boxFetchInterval { return // throttle — serve the cache } c.lastFetch = c.now() // updated even on failure → retried once per window, not per sweep usage, err := c.reader.Usage(context.Background()) if err != nil { if errors.Is(err, tenantsync.ErrUsageUnsupported) { // EXPECTED pre-update condition — a distinct "unavailable" state, NOT degraded, NO alert. c.snap = PBSBoxSnapshot{State: PBSStateUnavailable, FetchedAt: c.now()} c.haveSnap = true if !c.unavailLogged { c.logger.Printf("[INFO] PBS-DR box: usage op unavailable — endpoint tenantsync update (v1.2.0) pending; gauge shows n/a until then") c.unavailLogged = true } // REACHABILITY, Scenario D: the counter is deliberately NOT advanced here. This is an // EXPECTED pre-update condition (endpoint tenantsync ≤ v1.1.0), not blindness — we reached // the endpoint and it told us the op does not exist. Alerting on it would fire for days on // a healthy box and train the operator to ignore the alert, which costs more than it buys. return // fillBand untouched — an expected data gap never re-arms/transitions } c.logger.Printf("[WARN] PBS-DR box: usage read failed (keeping last snapshot): %v", err) if c.haveSnap { c.snap.State = PBSStateDegraded // serve last-known, visibly stale; NO band transition } // REACHABILITY: count the failed window and report once past the threshold. NOT gated on // haveSnap — a hub restarted INTO an outage has no snapshot to mark degraded and is precisely // the case the pre-v0.106.0 code handled worst: it would have stayed silent forever. if c.reachFails == 0 { c.reachFirstFail = c.now() } c.reachFails++ if c.reachFails >= c.reachFailThreshold { c.emitUnreachable(err) } return } c.unavailLogged = false // recovered from unavailable // REACHABILITY: we reached the endpoint and it answered — that is a success for this signal even // if the answer is unusable for a fill percentage (zero capacity below). Handled HERE, before the // `snap.State != PBSStateOK` early return further down, or a zero-capacity read would leave an // outstanding blindness alert un-cleared forever. if c.reachAlerted { c.emitRecovered() c.reachAlerted = false } c.reachFails = 0 c.reachFirstFail = time.Time{} c.lastOK = c.now() snap := PBSBoxSnapshot{CapacityBytes: usage.Total, UsedBytes: usage.Used, State: PBSStateOK, FetchedAt: c.now()} if usage.Total > 0 { snap.FillPercent = float64(usage.Used) * 100 / float64(usage.Total) snap.FillBand = bandForPercent(snap.FillPercent, c.fillWarn, c.fillCrit) } else { snap.State = PBSStateDegraded // zero total — guard the division } c.snap = snap c.haveSnap = true if snap.State != PBSStateOK { return } c.logger.Printf("[INFO] PBS-DR box refreshed: %.1f%% full (%s of %s)", snap.FillPercent, fmtSize(snap.UsedBytes), fmtSize(snap.CapacityBytes)) // FILL band (escalation-only; recovery re-arms because bandOK has rank 0). if bandRank(snap.FillBand) > bandRank(c.fillBand) { c.emitFill(snap, snap.FillBand) } c.fillBand = snap.FillBand } // Snapshot returns the cached aggregate + whether one exists (mutex copy-out). The web layer's only // source — it NEVER polls ep0 in the request path. func (c *PBSDRBoxChecker) Snapshot() (PBSBoxSnapshot, bool) { c.mu.Lock() defer c.mu.Unlock() return c.snap, c.haveSnap } // FillState exposes the current band (tests). func (c *PBSDRBoxChecker) FillState() string { c.mu.Lock() defer c.mu.Unlock() return orUnknown(c.fillBand) } // emitUnreachable reports that the hub cannot see the PBS-DR datastore. // // ⚠ CADENCE — THIS REPEATS, AND THAT IS DELIBERATE. Do not "fix" it back to the band shape. // emitFill/emitOversub are ESCALATION-ONLY: they fire once on a band rise and stay silent while the // condition persists, which is right for a fill trend. Applied here it would produce exactly ONE mail // at roughly minute 30 of a nine-hour outage, and one mail is missable — that is the failure this // whole signal exists to remove. So this fires on EVERY failed window past the threshold and leans on // the dispatcher's 1-hour per-type operator cooldown to throttle the mail down to an hourly "still // blind" heartbeat, which is itself the escalation. The cost is a `suppressed` notification row about // every 15 minutes during an outage — honest bookkeeping, and cheaper than a missed outage. func (c *PBSDRBoxChecker) emitUnreachable(cause error) { c.reachAlerted = true lastOK := "never" if !c.lastOK.IsZero() { lastOK = c.lastOK.Format(time.RFC3339) } d := map[string]any{ "scope": pbsdrBoxScope, "consecutive_failures": c.reachFails, "error": cause.Error(), } // last_ok is OMITTED rather than zero-valued when there has never been a successful read — a // fabricated timestamp reads as "it was fine until then", which would be a lie (Scenario E). if !c.lastOK.IsZero() { d["last_ok"] = c.lastOK.Format(time.RFC3339) } details, _ := json.Marshal(d) msg := fmt.Sprintf("PBS DR endpoint unreachable — %d consecutive 15-minute checks failed (last successful read: %s); the off-site DR tier cannot be seen from the hub", c.reachFails, lastOK) c.logger.Printf("[WARN] PBS-DR box UNREACHABLE: %d consecutive failed reads (last ok: %s)", c.reachFails, lastOK) if c.onEvent != nil { c.onEvent(pbsdrBoxScope, "pbsdr_box_unreachable", "warning", msg, string(details), "hub") } } // emitRecovered is the paired all-clear. Severity "info" — it reaches the operator only because // `pbsdr_box_recovered` is registered in the dispatcher's recoveredPairedDownTypes, which short- // circuits the severity gate. Without that entry this is stored and never mailed. func (c *PBSDRBoxChecker) emitRecovered() { blindFor := "unknown" if !c.reachFirstFail.IsZero() { blindFor = c.now().Sub(c.reachFirstFail).Round(time.Second).String() } details, _ := json.Marshal(map[string]any{ "scope": pbsdrBoxScope, "blind_for": blindFor, "consecutive_failures": c.reachFails, }) msg := fmt.Sprintf("PBS DR endpoint reachable again after %s — the off-site DR tier is visible to the hub", blindFor) c.logger.Printf("[INFO] PBS-DR box RECOVERED after %s (%d failed reads)", blindFor, c.reachFails) if c.onEvent != nil { c.onEvent(pbsdrBoxScope, "pbsdr_box_recovered", "info", msg, string(details), "hub") } } func (c *PBSDRBoxChecker) emitFill(snap PBSBoxSnapshot, band string) { var severity, message string switch band { case bandCritical: severity = "critical" message = fmt.Sprintf("PBS DR datastore %.0f%% full (%s of %s) — approaching capacity; the offsite DR tier's retention is at risk; free space or grow the datastore", snap.FillPercent, fmtSize(snap.UsedBytes), fmtSize(snap.CapacityBytes)) case bandWarning: severity = "warning" message = fmt.Sprintf("PBS DR datastore %.0f%% full (%s of %s) — the offsite DR datastore is filling", snap.FillPercent, fmtSize(snap.UsedBytes), fmtSize(snap.CapacityBytes)) default: return } details, _ := json.Marshal(map[string]any{ "scope": pbsdrBoxScope, "capacity_bytes": snap.CapacityBytes, "used_bytes": snap.UsedBytes, "fill_percent": snap.FillPercent, "warn_percent": c.fillWarn, "crit_percent": c.fillCrit, }) c.logger.Printf("[INFO] PBS-DR box fill: %.0f%% (%s)", snap.FillPercent, band) // NO SaveEvent — customer-less scope; straight to the operator dispatcher. if c.onEvent != nil { c.onEvent(pbsdrBoxScope, "pbsdr_box_fill", severity, message, string(details), "hub") } }