From 2d249315973ed8610139af303505e8059fcb0505 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Thu, 24 Sep 2026 16:26:17 +0200 Subject: [PATCH] hub v0.124.0: a thin pool is critical at 90% (data or metadata), one alarm per pool per 6 h (R-672) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- .../audits/night-2026-09-24/tools/walk.py | 2 +- hub/CHANGELOG.md | 16 +++++ hub/internal/monitor/r672_thinpool_test.go | 69 +++++++++++++++++++ hub/internal/monitor/storage_fill.go | 58 +++++++++++++--- hub/internal/notify/dispatcher.go | 32 ++++++++- .../notify/r672_storage_cooldown_test.go | 53 ++++++++++++++ hub/internal/store/store.go | 11 +++ 7 files changed, 231 insertions(+), 10 deletions(-) create mode 100644 hub/internal/monitor/r672_thinpool_test.go create mode 100644 hub/internal/notify/r672_storage_cooldown_test.go diff --git a/documentation/audits/night-2026-09-24/tools/walk.py b/documentation/audits/night-2026-09-24/tools/walk.py index fa85d055..bbaab28c 100644 --- a/documentation/audits/night-2026-09-24/tools/walk.py +++ b/documentation/audits/night-2026-09-24/tools/walk.py @@ -27,7 +27,7 @@ EV = "/mnt/5_hdd/felhom.eu/git/felhom.eu/documentation/audits/night-2026-09-24" DRILL = "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill" # GUEST=9201 selects demo-hp's hub-enabled guest (the mail proof); default 9202, the scratch guest. GUEST = os.environ.get("GUEST", "9202") -BASE = {"9202": "https://192.168.0.114", "9201": "https://192.168.0.138"}[GUEST] +BASE = os.environ.get("BASE") or {"9202": "https://192.168.0.114", "9201": "https://192.168.0.138"}[GUEST] DOMAIN = os.environ.get("DOMAIN", "enkisfelhom.hu") HOSTHDR = f"Host: felhom.{DOMAIN}" HP = "demo-hp" diff --git a/hub/CHANGELOG.md b/hub/CHANGELOG.md index ac05a48c..c3428f0d 100644 --- a/hub/CHANGELOG.md +++ b/hub/CHANGELOG.md @@ -1,3 +1,19 @@ +## v0.124.0 — a thin pool is critical at 90 %, one alarm per pool per 6 hours (2026-09-24, R-672) + +- **Storage fill, thin pools** (`monitor/storage_fill.go`): an `lvmthin` target is judged on the WORSE of data + and metadata fill (`thin_pool.metadata_used_fraction`, now read by `store.GetHostStorageTargets` as + `MetaPercent`, -1 when absent) with its own bands — warning 85 %, **critical 90 %** (other storages keep + 90/95). Measured 2026-09-24 on demo-hp: `local-lvm` went from 95 % to 100 % in about a minute and 9201's disks + remounted read-only five minutes later; the alarm fired only at the next 15-minute report. The message says a + full thin pool turns every guest on it read-only; details carry `data_percent` and `meta_percent`. +- **Per pool, 6 hours** (`notify/dispatcher.go`): `storage_fill_warning` / `storage_fill_critical` operator + cooldown keyed `customer:type:host/storage` with a 6-hour window (was `customer:type`, 1 hour — one pool could + silence another filling in the same hour). Every other type's key is unchanged. +- Agent v0.133.0 requests an immediate report when a thin pool crosses 90 %, so this fires in seconds; an older + agent's pool is judged at its next report. +- Red-proofs: the generic data-only bands → a 91 % pool raises only a warning and a metadata-full pool nothing; + the old key → a second pool in the same hour is silenced. Suite green. + ## v0.123.0 — `app_stopped_unhealthy`: the household is told when the box stops a crashing app (2026-09-24, `09` §3 decision 28) Controller v0.269.0 stops an app in a crash loop (≥ 6 restarts in 10 min) or an out-of-memory storm (≥ 20 diff --git a/hub/internal/monitor/r672_thinpool_test.go b/hub/internal/monitor/r672_thinpool_test.go new file mode 100644 index 00000000..232df0a1 --- /dev/null +++ b/hub/internal/monitor/r672_thinpool_test.go @@ -0,0 +1,69 @@ +package monitor + +import ( + "strconv" + "testing" + + "gitea.dooplex.hu/admin/felhom-hub/internal/store" +) + +// R-672 (hub v0.124.0): an lvmthin pool is CRITICAL at 90 % of data OR metadata. The consequence is the +// event the operator receives, asserted per case. +// +// COMPANION RED-PROOF (REPORT): `judged` returning the generic bands and data only (hub v0.123.0) → a +// thin pool at 91 % raises only storage_fill_warning, and a metadata-full pool raises nothing. +func saveThin(t *testing.T, st *store.Store, dataPct, metaPct float64) { + t.Helper() + body := []byte(`{"host_id":"h1","storage_targets":[{"name":"local-lvm","type":"lvmthin","mount_path":"","used_fraction":` + + ftoa(dataPct/100) + `,"total_bytes":1000,"used_bytes":1,"thin_pool":{"data_used_fraction":` + ftoa(dataPct/100) + + `,"metadata_used_fraction":` + ftoa(metaPct/100) + `}}]}`) + if err := st.SaveHostReport("h1", "c1", body, store.HostReportDenorm{}); err != nil { + t.Fatal(err) + } +} + +func ftoa(f float64) string { return strconv.FormatFloat(f, 'g', -1, 64) } + +func TestR672_ThinPoolCriticalAt90(t *testing.T) { + for _, tc := range []struct { + name string + data, meta float64 + wantType, sev string + }{ + {"data 91 %", 91, 3, "storage_fill_critical", "critical"}, + {"metadata 92 %, data 50 %", 50, 92, "storage_fill_critical", "critical"}, + {"data 87 %", 87, 3, "storage_fill_warning", "warning"}, + {"data 60 %", 60, 3, "", ""}, + } { + t.Run(tc.name, func(t *testing.T) { + st := newDiskStore(t) + saveThin(t, st, 10, 1) + var types, sevs []string + fc := NewStorageFillChecker(st, 90, 95, func(_, et, sev, _, _, _ string) { types = append(types, et); sevs = append(sevs, sev) }, quietLog()) + saveThin(t, st, tc.data, tc.meta) + fc.Check() + if tc.wantType == "" { + if len(types) != 0 { + t.Fatalf("a healthy pool raised %v", types) + } + return + } + if len(types) != 1 || types[0] != tc.wantType || sevs[0] != tc.sev { + t.Fatalf("events = %v %v — want one %s (%s)", types, sevs, tc.wantType, tc.sev) + } + }) + } +} + +// A non-thin storage keeps the generic bands: 91 % is a warning. +func TestR672_NonThinUnchanged(t *testing.T) { + st := newDiskStore(t) + saveStorageReport(t, st, stTarget{"dumpvol", "local-dir", "/mnt/backup", 10}) + var types []string + fc := NewStorageFillChecker(st, 90, 95, func(_, et, _, _, _, _ string) { types = append(types, et) }, quietLog()) + saveStorageReport(t, st, stTarget{"dumpvol", "local-dir", "/mnt/backup", 91}) + fc.Check() + if len(types) != 1 || types[0] != "storage_fill_warning" { + t.Fatalf("a dir storage at 91 %% must stay a warning: %v", types) + } +} diff --git a/hub/internal/monitor/storage_fill.go b/hub/internal/monitor/storage_fill.go index 1ddc5db5..10a44acf 100644 --- a/hub/internal/monitor/storage_fill.go +++ b/hub/internal/monitor/storage_fill.go @@ -35,8 +35,34 @@ type StorageFillChecker struct { const ( defaultStorageFillWarnPercent = 90.0 defaultStorageFillCritPercent = 95.0 + + // R-672 (hub v0.124.0): an lvmthin pool is CRITICAL at 90 % of data OR metadata, warned at 85 %. A full + // thin pool does not merely refuse writes — every guest on it gets I/O errors and remounts read-only + // (measured on demo-hp 2026-09-24: 9201's disks went read-only 5 minutes after the pool reached 100 %, + // and it went from 95 % to 100 % in about a minute). The generic 90/95 bands leave no time to act. + thinPoolWarnPercent = 85.0 + thinPoolCritPercent = 90.0 ) +// judged is what the bands are applied to: a thin pool's worse of data and metadata, with its own bands. +func (fc *StorageFillChecker) judged(row store.HostStorageTargetRow) (pct, warn, crit float64) { + if row.Type != "lvmthin" { + return row.Percent, fc.warn, fc.crit + } + pct = row.Percent + if row.MetaPercent > pct { + pct = row.MetaPercent + } + return pct, minF(fc.warn, thinPoolWarnPercent), minF(fc.crit, thinPoolCritPercent) +} + +func minF(a, b float64) float64 { + if a < b { + return a + } + return b +} + // fillKey is the per-(host,target) state key. A NUL separator can't appear in a host id / storage name. func fillKey(hostID, target string) string { return hostID + "\x00" + target } @@ -87,7 +113,8 @@ func NewStorageFillChecker(s *store.Store, warnPercent, critPercent float64, onE continue } fc.customerOf[row.HostID] = row.CustomerID - band := bandForPercent(row.Percent, fc.warn, fc.crit) + pct, w, c := fc.judged(row) + band := bandForPercent(pct, w, c) if band != bandOK { breachedCount++ continue // leave UNSEEDED → first Check emits (the dispatcher's 1h cooldown dedups a restart) @@ -123,7 +150,8 @@ func (fc *StorageFillChecker) Check() { } fc.customerOf[row.HostID] = row.CustomerID - newBand := bandForPercent(row.Percent, fc.warn, fc.crit) + pct, w, c := fc.judged(row) + newBand := bandForPercent(pct, w, c) oldBand := fc.states[key] // "" (rank 0) for an unseen / breached-at-init key if bandRank(newBand) > bandRank(oldBand) { fc.emit(row, oldBand, newBand) @@ -153,15 +181,20 @@ func (fc *StorageFillChecker) GetState(hostID, target string) string { func (fc *StorageFillChecker) emit(row store.HostStorageTargetRow, oldBand, newBand string) { var eventType, severity, message string + pct, warn, crit := fc.judged(row) + thin := "" + if row.Type == "lvmthin" { + thin = fmt.Sprintf(" (thin pool: data %.0f%%, metadata %s) — a full thin pool turns EVERY guest on it read-only", row.Percent, metaLabel(row.MetaPercent)) + } switch newBand { case bandCritical: eventType = "storage_fill_critical" severity = "critical" // natural critical — hub v0.24.0 routes it; the operator email styles it 🔴 - message = fmt.Sprintf("Host %s: storage %q CRITICALLY full at %.0f%% (threshold %.0f%%) — backups/writes to it will fail; free space immediately", row.HostID, row.Name, row.Percent, fc.crit) + message = fmt.Sprintf("Host %s: storage %q CRITICALLY full at %.0f%% (threshold %.0f%%)%s — backups/writes to it will fail; free space immediately", row.HostID, row.Name, pct, crit, thin) case bandWarning: eventType = "storage_fill_warning" severity = "warning" - message = fmt.Sprintf("Host %s: storage %q high at %.0f%% (threshold %.0f%%) — free space before it fills", row.HostID, row.Name, row.Percent, fc.warn) + message = fmt.Sprintf("Host %s: storage %q high at %.0f%% (threshold %.0f%%)%s — free space before it fills", row.HostID, row.Name, pct, warn, thin) default: return } @@ -170,14 +203,16 @@ func (fc *StorageFillChecker) emit(row store.HostStorageTargetRow, oldBand, newB "host_id": row.HostID, "storage": row.Name, "storage_type": row.Type, - "percent": row.Percent, + "percent": pct, + "data_percent": row.Percent, + "meta_percent": row.MetaPercent, "total_bytes": row.TotalBytes, "used_bytes": row.UsedBytes, - "warn_percent": fc.warn, - "crit_percent": fc.crit, + "warn_percent": warn, + "crit_percent": crit, }) - fc.logger.Printf("[INFO] Storage fill: %s %q %.0f%% %s→%s (%s)", row.HostID, row.Name, row.Percent, bandLabel(oldBand), newBand, eventType) + fc.logger.Printf("[INFO] Storage fill: %s %q %.0f%% %s→%s (%s)", row.HostID, row.Name, pct, bandLabel(oldBand), newBand, eventType) if _, err := fc.store.SaveEvent(row.CustomerID, eventType, severity, message, string(details), "hub"); err != nil { fc.logger.Printf("[WARN] Failed to save storage fill event for %s/%s: %v", row.HostID, row.Name, err) @@ -199,3 +234,10 @@ func bandForPercent(pct, warn, crit float64) string { return bandOK } } + +func metaLabel(p float64) string { + if p < 0 { + return "unknown" + } + return fmt.Sprintf("%.0f%%", p) +} diff --git a/hub/internal/notify/dispatcher.go b/hub/internal/notify/dispatcher.go index 3795433f..44e5bab4 100644 --- a/hub/internal/notify/dispatcher.go +++ b/hub/internal/notify/dispatcher.go @@ -457,6 +457,33 @@ func cooldownStackSuffixFor(detailsJSON string) string { return ":" + d.StackName } +// perStorageCooldownEvents (R-672, hub v0.124.0): the storage-fill alarms are keyed PER POOL (host + +// storage), with a 6-hour window — "one operator event per pool per 6 hours". The old `customer:type` key +// let one host's pool take the hour and silence another pool filling in it, and a 1-hour window re-mailed +// a pool that sat full for a day. The checker itself emits only on a band ESCALATION, so the window only +// ever collapses a flapping pool. +var perStorageCooldownEvents = map[string]bool{ + "storage_fill_warning": true, + "storage_fill_critical": true, +} + +const storageFillCooldown = 6 * time.Hour + +// cooldownStorageSuffix returns ":"+host_id+"/"+storage for a per-storage type, else "". +func cooldownStorageSuffix(eventType, detailsJSON string) string { + if !perStorageCooldownEvents[eventType] || detailsJSON == "" { + return "" + } + var d struct { + HostID string `json:"host_id"` + Storage string `json:"storage"` + } + if err := json.Unmarshal([]byte(detailsJSON), &d); err != nil || d.Storage == "" { + return "" + } + return ":" + d.HostID + "/" + d.Storage +} + // nodeLivenessEvents skip the 1-hour operator cooldown (OPERATOR RULING 2026-09-15, decision A; // 08-alarm-ladder.md §5). BIGNIGHT F9: the box was dead for 33 minutes and the `node_stale` mail was // suppressed because F8's `node_stale` had used the hour 39 minutes earlier; the `node_recovered` @@ -487,6 +514,9 @@ func operatorCooldownFor(eventType string) time.Duration { if nodeLivenessEvents[eventType] { return nodeLivenessDedupeWindow } + if perStorageCooldownEvents[eventType] { + return storageFillCooldown + } return operatorCooldown } @@ -499,7 +529,7 @@ func (d *Dispatcher) processOperator(customerID, eventType, severity, message, d // is byte-identical to v0.107.0's — pinned by TestR389_NoOtherEventTypeKeyChanges. cooldownKey := customerID + ":" + eventType + cooldownTierSuffix(detailsJSON) + cooldownRunSuffix(detailsJSON) + - cooldownStackSuffix(eventType, detailsJSON) + cooldownStackSuffix(eventType, detailsJSON) + cooldownStorageSuffix(eventType, detailsJSON) window := operatorCooldownFor(eventType) d.mu.Lock() if last, ok := d.opCooldowns[cooldownKey]; ok && time.Since(last) < window { diff --git a/hub/internal/notify/r672_storage_cooldown_test.go b/hub/internal/notify/r672_storage_cooldown_test.go new file mode 100644 index 00000000..c4eb60d9 --- /dev/null +++ b/hub/internal/notify/r672_storage_cooldown_test.go @@ -0,0 +1,53 @@ +package notify + +import ( + "io" + "log" + "sync" + "testing" + "time" +) + +// R-672 (hub v0.124.0): storage-fill alarms are "one operator event per POOL per 6 hours". The +// consequence asserted is the mail count. +// +// COMPANION RED-PROOF (REPORT): drop cooldownStorageSuffix from the key and the 6-hour window (hub +// v0.123.0: `customer:type`, 1 hour) → "a second pool filling in the same hour was silenced". +func TestR672_StorageFillIsPerPoolPerSixHours(t *testing.T) { + st := newDispStore(t) + d := NewDispatcher(st, "test-key", "from@felhom.eu", "op@felhom.eu", true, log.New(io.Discard, "", 0)) + var mu sync.Mutex + sent := 0 + d.sendEmailFn = func(string, string, string, map[string]string) error { mu.Lock(); sent++; mu.Unlock(); return nil } + count := func() int { mu.Lock(); defer mu.Unlock(); return sent } + pool := func(host, s string) string { return `{"host_id":"` + host + `","storage":"` + s + `","percent":91}` } + + d.processOperator("c1", "storage_fill_critical", "critical", "pool A full", pool("h1", "local-lvm"), "hub") + d.processOperator("c1", "storage_fill_critical", "critical", "pool B full", pool("h1", "fast-lvm"), "hub") + if count() != 2 { + t.Fatalf("a second pool filling in the same hour was silenced (sent=%d)", count()) + } + d.processOperator("c1", "storage_fill_critical", "critical", "pool A again", pool("h1", "local-lvm"), "hub") + if count() != 2 { + t.Fatalf("the same pool mailed twice at once (sent=%d)", count()) + } + // Two hours later the SAME pool is still inside its 6-hour window (the old 1-hour window re-mailed it). + d.mu.Lock() + for k := range d.opCooldowns { + d.opCooldowns[k] = time.Now().Add(-2 * time.Hour) + } + d.mu.Unlock() + d.processOperator("c1", "storage_fill_critical", "critical", "pool A 2h later", pool("h1", "local-lvm"), "hub") + if count() != 2 { + t.Fatalf("the same pool was mailed again after 2 hours — want one per 6 hours (sent=%d)", count()) + } + d.mu.Lock() + for k := range d.opCooldowns { + d.opCooldowns[k] = time.Now().Add(-7 * time.Hour) + } + d.mu.Unlock() + d.processOperator("c1", "storage_fill_critical", "critical", "pool A 7h later", pool("h1", "local-lvm"), "hub") + if count() != 3 { + t.Fatalf("after 6 hours the pool must alarm again (sent=%d)", count()) + } +} diff --git a/hub/internal/store/store.go b/hub/internal/store/store.go index e1def2c0..513279c1 100644 --- a/hub/internal/store/store.go +++ b/hub/internal/store/store.go @@ -3730,6 +3730,9 @@ type HostStorageTargetRow struct { Percent float64 // 0..100 (used_fraction × 100) TotalBytes int64 UsedBytes int64 + // MetaPercent is an lvmthin pool's METADATA fill, 0..100; -1 when the report carries none (R-672, + // hub v0.124.0 — a thin pool is corrupted by metadata exhaustion exactly like data exhaustion). + MetaPercent float64 } // GetHostStorageTargets returns the per-storage fill of every host's LATEST report (MAX(id) per host), @@ -3759,14 +3762,22 @@ func (s *Store) GetHostStorageTargets() ([]HostStorageTargetRow, error) { UsedFraction float64 `json:"used_fraction"` TotalBytes int64 `json:"total_bytes"` UsedBytes int64 `json:"used_bytes"` + ThinPool *struct { + MetadataUsedFraction *float64 `json:"metadata_used_fraction"` + } `json:"thin_pool"` } `json:"storage_targets"` } _ = json.Unmarshal([]byte(reportJSON), &body) // malformed/old body → no targets (never a false alert) for _, t := range body.StorageTargets { + meta := -1.0 + if t.ThinPool != nil && t.ThinPool.MetadataUsedFraction != nil { + meta = *t.ThinPool.MetadataUsedFraction * 100 + } out = append(out, HostStorageTargetRow{ HostID: hostID, CustomerID: customerID, Name: t.Name, Type: t.Type, MountPath: t.MountPath, Percent: t.UsedFraction * 100, TotalBytes: t.TotalBytes, UsedBytes: t.UsedBytes, + MetaPercent: meta, }) } }