hub v0.124.0: a thin pool is critical at 90% (data or metadata), one alarm per pool per 6 h (R-672)
gates / gates (push) Successful in 24s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-24 16:26:17 +02:00
parent 1a72fef87f
commit 2d24931597
7 changed files with 231 additions and 10 deletions
@@ -27,7 +27,7 @@ EV = "/mnt/5_hdd/felhom.eu/git/felhom.eu/documentation/audits/night-2026-09-24"
DRILL = "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill"
# GUEST=9201 selects demo-hp's hub-enabled guest (the mail proof); default 9202, the scratch guest.
GUEST = os.environ.get("GUEST", "9202")
BASE = {"9202": "https://192.168.0.114", "9201": "https://192.168.0.138"}[GUEST]
BASE = os.environ.get("BASE") or {"9202": "https://192.168.0.114", "9201": "https://192.168.0.138"}[GUEST]
DOMAIN = os.environ.get("DOMAIN", "enkisfelhom.hu")
HOSTHDR = f"Host: felhom.{DOMAIN}"
HP = "demo-hp"
+16
View File
@@ -1,3 +1,19 @@
## v0.124.0 — a thin pool is critical at 90 %, one alarm per pool per 6 hours (2026-09-24, R-672)
- **Storage fill, thin pools** (`monitor/storage_fill.go`): an `lvmthin` target is judged on the WORSE of data
and metadata fill (`thin_pool.metadata_used_fraction`, now read by `store.GetHostStorageTargets` as
`MetaPercent`, -1 when absent) with its own bands — warning 85 %, **critical 90 %** (other storages keep
90/95). Measured 2026-09-24 on demo-hp: `local-lvm` went from 95 % to 100 % in about a minute and 9201's disks
remounted read-only five minutes later; the alarm fired only at the next 15-minute report. The message says a
full thin pool turns every guest on it read-only; details carry `data_percent` and `meta_percent`.
- **Per pool, 6 hours** (`notify/dispatcher.go`): `storage_fill_warning` / `storage_fill_critical` operator
cooldown keyed `customer:type:host/storage` with a 6-hour window (was `customer:type`, 1 hour — one pool could
silence another filling in the same hour). Every other type's key is unchanged.
- Agent v0.133.0 requests an immediate report when a thin pool crosses 90 %, so this fires in seconds; an older
agent's pool is judged at its next report.
- Red-proofs: the generic data-only bands → a 91 % pool raises only a warning and a metadata-full pool nothing;
the old key → a second pool in the same hour is silenced. Suite green.
## v0.123.0 — `app_stopped_unhealthy`: the household is told when the box stops a crashing app (2026-09-24, `09` §3 decision 28)
Controller v0.269.0 stops an app in a crash loop (≥ 6 restarts in 10 min) or an out-of-memory storm (≥ 20
@@ -0,0 +1,69 @@
package monitor
import (
"strconv"
"testing"
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
)
// R-672 (hub v0.124.0): an lvmthin pool is CRITICAL at 90 % of data OR metadata. The consequence is the
// event the operator receives, asserted per case.
//
// COMPANION RED-PROOF (REPORT): `judged` returning the generic bands and data only (hub v0.123.0) → a
// thin pool at 91 % raises only storage_fill_warning, and a metadata-full pool raises nothing.
func saveThin(t *testing.T, st *store.Store, dataPct, metaPct float64) {
t.Helper()
body := []byte(`{"host_id":"h1","storage_targets":[{"name":"local-lvm","type":"lvmthin","mount_path":"","used_fraction":` +
ftoa(dataPct/100) + `,"total_bytes":1000,"used_bytes":1,"thin_pool":{"data_used_fraction":` + ftoa(dataPct/100) +
`,"metadata_used_fraction":` + ftoa(metaPct/100) + `}}]}`)
if err := st.SaveHostReport("h1", "c1", body, store.HostReportDenorm{}); err != nil {
t.Fatal(err)
}
}
func ftoa(f float64) string { return strconv.FormatFloat(f, 'g', -1, 64) }
func TestR672_ThinPoolCriticalAt90(t *testing.T) {
for _, tc := range []struct {
name string
data, meta float64
wantType, sev string
}{
{"data 91 %", 91, 3, "storage_fill_critical", "critical"},
{"metadata 92 %, data 50 %", 50, 92, "storage_fill_critical", "critical"},
{"data 87 %", 87, 3, "storage_fill_warning", "warning"},
{"data 60 %", 60, 3, "", ""},
} {
t.Run(tc.name, func(t *testing.T) {
st := newDiskStore(t)
saveThin(t, st, 10, 1)
var types, sevs []string
fc := NewStorageFillChecker(st, 90, 95, func(_, et, sev, _, _, _ string) { types = append(types, et); sevs = append(sevs, sev) }, quietLog())
saveThin(t, st, tc.data, tc.meta)
fc.Check()
if tc.wantType == "" {
if len(types) != 0 {
t.Fatalf("a healthy pool raised %v", types)
}
return
}
if len(types) != 1 || types[0] != tc.wantType || sevs[0] != tc.sev {
t.Fatalf("events = %v %v — want one %s (%s)", types, sevs, tc.wantType, tc.sev)
}
})
}
}
// A non-thin storage keeps the generic bands: 91 % is a warning.
func TestR672_NonThinUnchanged(t *testing.T) {
st := newDiskStore(t)
saveStorageReport(t, st, stTarget{"dumpvol", "local-dir", "/mnt/backup", 10})
var types []string
fc := NewStorageFillChecker(st, 90, 95, func(_, et, _, _, _, _ string) { types = append(types, et) }, quietLog())
saveStorageReport(t, st, stTarget{"dumpvol", "local-dir", "/mnt/backup", 91})
fc.Check()
if len(types) != 1 || types[0] != "storage_fill_warning" {
t.Fatalf("a dir storage at 91 %% must stay a warning: %v", types)
}
}
+50 -8
View File
@@ -35,8 +35,34 @@ type StorageFillChecker struct {
const (
defaultStorageFillWarnPercent = 90.0
defaultStorageFillCritPercent = 95.0
// R-672 (hub v0.124.0): an lvmthin pool is CRITICAL at 90 % of data OR metadata, warned at 85 %. A full
// thin pool does not merely refuse writes — every guest on it gets I/O errors and remounts read-only
// (measured on demo-hp 2026-09-24: 9201's disks went read-only 5 minutes after the pool reached 100 %,
// and it went from 95 % to 100 % in about a minute). The generic 90/95 bands leave no time to act.
thinPoolWarnPercent = 85.0
thinPoolCritPercent = 90.0
)
// judged is what the bands are applied to: a thin pool's worse of data and metadata, with its own bands.
func (fc *StorageFillChecker) judged(row store.HostStorageTargetRow) (pct, warn, crit float64) {
if row.Type != "lvmthin" {
return row.Percent, fc.warn, fc.crit
}
pct = row.Percent
if row.MetaPercent > pct {
pct = row.MetaPercent
}
return pct, minF(fc.warn, thinPoolWarnPercent), minF(fc.crit, thinPoolCritPercent)
}
func minF(a, b float64) float64 {
if a < b {
return a
}
return b
}
// fillKey is the per-(host,target) state key. A NUL separator can't appear in a host id / storage name.
func fillKey(hostID, target string) string { return hostID + "\x00" + target }
@@ -87,7 +113,8 @@ func NewStorageFillChecker(s *store.Store, warnPercent, critPercent float64, onE
continue
}
fc.customerOf[row.HostID] = row.CustomerID
band := bandForPercent(row.Percent, fc.warn, fc.crit)
pct, w, c := fc.judged(row)
band := bandForPercent(pct, w, c)
if band != bandOK {
breachedCount++
continue // leave UNSEEDED → first Check emits (the dispatcher's 1h cooldown dedups a restart)
@@ -123,7 +150,8 @@ func (fc *StorageFillChecker) Check() {
}
fc.customerOf[row.HostID] = row.CustomerID
newBand := bandForPercent(row.Percent, fc.warn, fc.crit)
pct, w, c := fc.judged(row)
newBand := bandForPercent(pct, w, c)
oldBand := fc.states[key] // "" (rank 0) for an unseen / breached-at-init key
if bandRank(newBand) > bandRank(oldBand) {
fc.emit(row, oldBand, newBand)
@@ -153,15 +181,20 @@ func (fc *StorageFillChecker) GetState(hostID, target string) string {
func (fc *StorageFillChecker) emit(row store.HostStorageTargetRow, oldBand, newBand string) {
var eventType, severity, message string
pct, warn, crit := fc.judged(row)
thin := ""
if row.Type == "lvmthin" {
thin = fmt.Sprintf(" (thin pool: data %.0f%%, metadata %s) — a full thin pool turns EVERY guest on it read-only", row.Percent, metaLabel(row.MetaPercent))
}
switch newBand {
case bandCritical:
eventType = "storage_fill_critical"
severity = "critical" // natural critical — hub v0.24.0 routes it; the operator email styles it 🔴
message = fmt.Sprintf("Host %s: storage %q CRITICALLY full at %.0f%% (threshold %.0f%%) — backups/writes to it will fail; free space immediately", row.HostID, row.Name, row.Percent, fc.crit)
message = fmt.Sprintf("Host %s: storage %q CRITICALLY full at %.0f%% (threshold %.0f%%)%s — backups/writes to it will fail; free space immediately", row.HostID, row.Name, pct, crit, thin)
case bandWarning:
eventType = "storage_fill_warning"
severity = "warning"
message = fmt.Sprintf("Host %s: storage %q high at %.0f%% (threshold %.0f%%) — free space before it fills", row.HostID, row.Name, row.Percent, fc.warn)
message = fmt.Sprintf("Host %s: storage %q high at %.0f%% (threshold %.0f%%)%s — free space before it fills", row.HostID, row.Name, pct, warn, thin)
default:
return
}
@@ -170,14 +203,16 @@ func (fc *StorageFillChecker) emit(row store.HostStorageTargetRow, oldBand, newB
"host_id": row.HostID,
"storage": row.Name,
"storage_type": row.Type,
"percent": row.Percent,
"percent": pct,
"data_percent": row.Percent,
"meta_percent": row.MetaPercent,
"total_bytes": row.TotalBytes,
"used_bytes": row.UsedBytes,
"warn_percent": fc.warn,
"crit_percent": fc.crit,
"warn_percent": warn,
"crit_percent": crit,
})
fc.logger.Printf("[INFO] Storage fill: %s %q %.0f%% %s→%s (%s)", row.HostID, row.Name, row.Percent, bandLabel(oldBand), newBand, eventType)
fc.logger.Printf("[INFO] Storage fill: %s %q %.0f%% %s→%s (%s)", row.HostID, row.Name, pct, bandLabel(oldBand), newBand, eventType)
if _, err := fc.store.SaveEvent(row.CustomerID, eventType, severity, message, string(details), "hub"); err != nil {
fc.logger.Printf("[WARN] Failed to save storage fill event for %s/%s: %v", row.HostID, row.Name, err)
@@ -199,3 +234,10 @@ func bandForPercent(pct, warn, crit float64) string {
return bandOK
}
}
func metaLabel(p float64) string {
if p < 0 {
return "unknown"
}
return fmt.Sprintf("%.0f%%", p)
}
+31 -1
View File
@@ -457,6 +457,33 @@ func cooldownStackSuffixFor(detailsJSON string) string {
return ":" + d.StackName
}
// perStorageCooldownEvents (R-672, hub v0.124.0): the storage-fill alarms are keyed PER POOL (host +
// storage), with a 6-hour window — "one operator event per pool per 6 hours". The old `customer:type` key
// let one host's pool take the hour and silence another pool filling in it, and a 1-hour window re-mailed
// a pool that sat full for a day. The checker itself emits only on a band ESCALATION, so the window only
// ever collapses a flapping pool.
var perStorageCooldownEvents = map[string]bool{
"storage_fill_warning": true,
"storage_fill_critical": true,
}
const storageFillCooldown = 6 * time.Hour
// cooldownStorageSuffix returns ":"+host_id+"/"+storage for a per-storage type, else "".
func cooldownStorageSuffix(eventType, detailsJSON string) string {
if !perStorageCooldownEvents[eventType] || detailsJSON == "" {
return ""
}
var d struct {
HostID string `json:"host_id"`
Storage string `json:"storage"`
}
if err := json.Unmarshal([]byte(detailsJSON), &d); err != nil || d.Storage == "" {
return ""
}
return ":" + d.HostID + "/" + d.Storage
}
// nodeLivenessEvents skip the 1-hour operator cooldown (OPERATOR RULING 2026-09-15, decision A;
// 08-alarm-ladder.md §5). BIGNIGHT F9: the box was dead for 33 minutes and the `node_stale` mail was
// suppressed because F8's `node_stale` had used the hour 39 minutes earlier; the `node_recovered`
@@ -487,6 +514,9 @@ func operatorCooldownFor(eventType string) time.Duration {
if nodeLivenessEvents[eventType] {
return nodeLivenessDedupeWindow
}
if perStorageCooldownEvents[eventType] {
return storageFillCooldown
}
return operatorCooldown
}
@@ -499,7 +529,7 @@ func (d *Dispatcher) processOperator(customerID, eventType, severity, message, d
// is byte-identical to v0.107.0's — pinned by TestR389_NoOtherEventTypeKeyChanges.
cooldownKey := customerID + ":" + eventType +
cooldownTierSuffix(detailsJSON) + cooldownRunSuffix(detailsJSON) +
cooldownStackSuffix(eventType, detailsJSON)
cooldownStackSuffix(eventType, detailsJSON) + cooldownStorageSuffix(eventType, detailsJSON)
window := operatorCooldownFor(eventType)
d.mu.Lock()
if last, ok := d.opCooldowns[cooldownKey]; ok && time.Since(last) < window {
@@ -0,0 +1,53 @@
package notify
import (
"io"
"log"
"sync"
"testing"
"time"
)
// R-672 (hub v0.124.0): storage-fill alarms are "one operator event per POOL per 6 hours". The
// consequence asserted is the mail count.
//
// COMPANION RED-PROOF (REPORT): drop cooldownStorageSuffix from the key and the 6-hour window (hub
// v0.123.0: `customer:type`, 1 hour) → "a second pool filling in the same hour was silenced".
func TestR672_StorageFillIsPerPoolPerSixHours(t *testing.T) {
st := newDispStore(t)
d := NewDispatcher(st, "test-key", "from@felhom.eu", "op@felhom.eu", true, log.New(io.Discard, "", 0))
var mu sync.Mutex
sent := 0
d.sendEmailFn = func(string, string, string, map[string]string) error { mu.Lock(); sent++; mu.Unlock(); return nil }
count := func() int { mu.Lock(); defer mu.Unlock(); return sent }
pool := func(host, s string) string { return `{"host_id":"` + host + `","storage":"` + s + `","percent":91}` }
d.processOperator("c1", "storage_fill_critical", "critical", "pool A full", pool("h1", "local-lvm"), "hub")
d.processOperator("c1", "storage_fill_critical", "critical", "pool B full", pool("h1", "fast-lvm"), "hub")
if count() != 2 {
t.Fatalf("a second pool filling in the same hour was silenced (sent=%d)", count())
}
d.processOperator("c1", "storage_fill_critical", "critical", "pool A again", pool("h1", "local-lvm"), "hub")
if count() != 2 {
t.Fatalf("the same pool mailed twice at once (sent=%d)", count())
}
// Two hours later the SAME pool is still inside its 6-hour window (the old 1-hour window re-mailed it).
d.mu.Lock()
for k := range d.opCooldowns {
d.opCooldowns[k] = time.Now().Add(-2 * time.Hour)
}
d.mu.Unlock()
d.processOperator("c1", "storage_fill_critical", "critical", "pool A 2h later", pool("h1", "local-lvm"), "hub")
if count() != 2 {
t.Fatalf("the same pool was mailed again after 2 hours — want one per 6 hours (sent=%d)", count())
}
d.mu.Lock()
for k := range d.opCooldowns {
d.opCooldowns[k] = time.Now().Add(-7 * time.Hour)
}
d.mu.Unlock()
d.processOperator("c1", "storage_fill_critical", "critical", "pool A 7h later", pool("h1", "local-lvm"), "hub")
if count() != 3 {
t.Fatalf("after 6 hours the pool must alarm again (sent=%d)", count())
}
}
+11
View File
@@ -3730,6 +3730,9 @@ type HostStorageTargetRow struct {
Percent float64 // 0..100 (used_fraction × 100)
TotalBytes int64
UsedBytes int64
// MetaPercent is an lvmthin pool's METADATA fill, 0..100; -1 when the report carries none (R-672,
// hub v0.124.0 — a thin pool is corrupted by metadata exhaustion exactly like data exhaustion).
MetaPercent float64
}
// GetHostStorageTargets returns the per-storage fill of every host's LATEST report (MAX(id) per host),
@@ -3759,14 +3762,22 @@ func (s *Store) GetHostStorageTargets() ([]HostStorageTargetRow, error) {
UsedFraction float64 `json:"used_fraction"`
TotalBytes int64 `json:"total_bytes"`
UsedBytes int64 `json:"used_bytes"`
ThinPool *struct {
MetadataUsedFraction *float64 `json:"metadata_used_fraction"`
} `json:"thin_pool"`
} `json:"storage_targets"`
}
_ = json.Unmarshal([]byte(reportJSON), &body) // malformed/old body → no targets (never a false alert)
for _, t := range body.StorageTargets {
meta := -1.0
if t.ThinPool != nil && t.ThinPool.MetadataUsedFraction != nil {
meta = *t.ThinPool.MetadataUsedFraction * 100
}
out = append(out, HostStorageTargetRow{
HostID: hostID, CustomerID: customerID,
Name: t.Name, Type: t.Type, MountPath: t.MountPath,
Percent: t.UsedFraction * 100, TotalBytes: t.TotalBytes, UsedBytes: t.UsedBytes,
MetaPercent: meta,
})
}
}