hub v0.124.0: a thin pool is critical at 90% (data or metadata), one alarm per pool per 6 h (R-672)
gates / gates (push) Successful in 24s
gates / gates (push) Successful in 24s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -27,7 +27,7 @@ EV = "/mnt/5_hdd/felhom.eu/git/felhom.eu/documentation/audits/night-2026-09-24"
|
||||
DRILL = "/mnt/5_hdd/felhom.eu/drill/app-catalog-drill"
|
||||
# GUEST=9201 selects demo-hp's hub-enabled guest (the mail proof); default 9202, the scratch guest.
|
||||
GUEST = os.environ.get("GUEST", "9202")
|
||||
BASE = {"9202": "https://192.168.0.114", "9201": "https://192.168.0.138"}[GUEST]
|
||||
BASE = os.environ.get("BASE") or {"9202": "https://192.168.0.114", "9201": "https://192.168.0.138"}[GUEST]
|
||||
DOMAIN = os.environ.get("DOMAIN", "enkisfelhom.hu")
|
||||
HOSTHDR = f"Host: felhom.{DOMAIN}"
|
||||
HP = "demo-hp"
|
||||
|
||||
@@ -1,3 +1,19 @@
|
||||
## v0.124.0 — a thin pool is critical at 90 %, one alarm per pool per 6 hours (2026-09-24, R-672)
|
||||
|
||||
- **Storage fill, thin pools** (`monitor/storage_fill.go`): an `lvmthin` target is judged on the WORSE of data
|
||||
and metadata fill (`thin_pool.metadata_used_fraction`, now read by `store.GetHostStorageTargets` as
|
||||
`MetaPercent`, -1 when absent) with its own bands — warning 85 %, **critical 90 %** (other storages keep
|
||||
90/95). Measured 2026-09-24 on demo-hp: `local-lvm` went from 95 % to 100 % in about a minute and 9201's disks
|
||||
remounted read-only five minutes later; the alarm fired only at the next 15-minute report. The message says a
|
||||
full thin pool turns every guest on it read-only; details carry `data_percent` and `meta_percent`.
|
||||
- **Per pool, 6 hours** (`notify/dispatcher.go`): `storage_fill_warning` / `storage_fill_critical` operator
|
||||
cooldown keyed `customer:type:host/storage` with a 6-hour window (was `customer:type`, 1 hour — one pool could
|
||||
silence another filling in the same hour). Every other type's key is unchanged.
|
||||
- Agent v0.133.0 requests an immediate report when a thin pool crosses 90 %, so this fires in seconds; an older
|
||||
agent's pool is judged at its next report.
|
||||
- Red-proofs: the generic data-only bands → a 91 % pool raises only a warning and a metadata-full pool nothing;
|
||||
the old key → a second pool in the same hour is silenced. Suite green.
|
||||
|
||||
## v0.123.0 — `app_stopped_unhealthy`: the household is told when the box stops a crashing app (2026-09-24, `09` §3 decision 28)
|
||||
|
||||
Controller v0.269.0 stops an app in a crash loop (≥ 6 restarts in 10 min) or an out-of-memory storm (≥ 20
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
package monitor
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
||||
)
|
||||
|
||||
// R-672 (hub v0.124.0): an lvmthin pool is CRITICAL at 90 % of data OR metadata. The consequence is the
|
||||
// event the operator receives, asserted per case.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT): `judged` returning the generic bands and data only (hub v0.123.0) → a
|
||||
// thin pool at 91 % raises only storage_fill_warning, and a metadata-full pool raises nothing.
|
||||
func saveThin(t *testing.T, st *store.Store, dataPct, metaPct float64) {
|
||||
t.Helper()
|
||||
body := []byte(`{"host_id":"h1","storage_targets":[{"name":"local-lvm","type":"lvmthin","mount_path":"","used_fraction":` +
|
||||
ftoa(dataPct/100) + `,"total_bytes":1000,"used_bytes":1,"thin_pool":{"data_used_fraction":` + ftoa(dataPct/100) +
|
||||
`,"metadata_used_fraction":` + ftoa(metaPct/100) + `}}]}`)
|
||||
if err := st.SaveHostReport("h1", "c1", body, store.HostReportDenorm{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func ftoa(f float64) string { return strconv.FormatFloat(f, 'g', -1, 64) }
|
||||
|
||||
func TestR672_ThinPoolCriticalAt90(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
data, meta float64
|
||||
wantType, sev string
|
||||
}{
|
||||
{"data 91 %", 91, 3, "storage_fill_critical", "critical"},
|
||||
{"metadata 92 %, data 50 %", 50, 92, "storage_fill_critical", "critical"},
|
||||
{"data 87 %", 87, 3, "storage_fill_warning", "warning"},
|
||||
{"data 60 %", 60, 3, "", ""},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
st := newDiskStore(t)
|
||||
saveThin(t, st, 10, 1)
|
||||
var types, sevs []string
|
||||
fc := NewStorageFillChecker(st, 90, 95, func(_, et, sev, _, _, _ string) { types = append(types, et); sevs = append(sevs, sev) }, quietLog())
|
||||
saveThin(t, st, tc.data, tc.meta)
|
||||
fc.Check()
|
||||
if tc.wantType == "" {
|
||||
if len(types) != 0 {
|
||||
t.Fatalf("a healthy pool raised %v", types)
|
||||
}
|
||||
return
|
||||
}
|
||||
if len(types) != 1 || types[0] != tc.wantType || sevs[0] != tc.sev {
|
||||
t.Fatalf("events = %v %v — want one %s (%s)", types, sevs, tc.wantType, tc.sev)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A non-thin storage keeps the generic bands: 91 % is a warning.
|
||||
func TestR672_NonThinUnchanged(t *testing.T) {
|
||||
st := newDiskStore(t)
|
||||
saveStorageReport(t, st, stTarget{"dumpvol", "local-dir", "/mnt/backup", 10})
|
||||
var types []string
|
||||
fc := NewStorageFillChecker(st, 90, 95, func(_, et, _, _, _, _ string) { types = append(types, et) }, quietLog())
|
||||
saveStorageReport(t, st, stTarget{"dumpvol", "local-dir", "/mnt/backup", 91})
|
||||
fc.Check()
|
||||
if len(types) != 1 || types[0] != "storage_fill_warning" {
|
||||
t.Fatalf("a dir storage at 91 %% must stay a warning: %v", types)
|
||||
}
|
||||
}
|
||||
@@ -35,8 +35,34 @@ type StorageFillChecker struct {
|
||||
const (
|
||||
defaultStorageFillWarnPercent = 90.0
|
||||
defaultStorageFillCritPercent = 95.0
|
||||
|
||||
// R-672 (hub v0.124.0): an lvmthin pool is CRITICAL at 90 % of data OR metadata, warned at 85 %. A full
|
||||
// thin pool does not merely refuse writes — every guest on it gets I/O errors and remounts read-only
|
||||
// (measured on demo-hp 2026-09-24: 9201's disks went read-only 5 minutes after the pool reached 100 %,
|
||||
// and it went from 95 % to 100 % in about a minute). The generic 90/95 bands leave no time to act.
|
||||
thinPoolWarnPercent = 85.0
|
||||
thinPoolCritPercent = 90.0
|
||||
)
|
||||
|
||||
// judged is what the bands are applied to: a thin pool's worse of data and metadata, with its own bands.
|
||||
func (fc *StorageFillChecker) judged(row store.HostStorageTargetRow) (pct, warn, crit float64) {
|
||||
if row.Type != "lvmthin" {
|
||||
return row.Percent, fc.warn, fc.crit
|
||||
}
|
||||
pct = row.Percent
|
||||
if row.MetaPercent > pct {
|
||||
pct = row.MetaPercent
|
||||
}
|
||||
return pct, minF(fc.warn, thinPoolWarnPercent), minF(fc.crit, thinPoolCritPercent)
|
||||
}
|
||||
|
||||
func minF(a, b float64) float64 {
|
||||
if a < b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// fillKey is the per-(host,target) state key. A NUL separator can't appear in a host id / storage name.
|
||||
func fillKey(hostID, target string) string { return hostID + "\x00" + target }
|
||||
|
||||
@@ -87,7 +113,8 @@ func NewStorageFillChecker(s *store.Store, warnPercent, critPercent float64, onE
|
||||
continue
|
||||
}
|
||||
fc.customerOf[row.HostID] = row.CustomerID
|
||||
band := bandForPercent(row.Percent, fc.warn, fc.crit)
|
||||
pct, w, c := fc.judged(row)
|
||||
band := bandForPercent(pct, w, c)
|
||||
if band != bandOK {
|
||||
breachedCount++
|
||||
continue // leave UNSEEDED → first Check emits (the dispatcher's 1h cooldown dedups a restart)
|
||||
@@ -123,7 +150,8 @@ func (fc *StorageFillChecker) Check() {
|
||||
}
|
||||
fc.customerOf[row.HostID] = row.CustomerID
|
||||
|
||||
newBand := bandForPercent(row.Percent, fc.warn, fc.crit)
|
||||
pct, w, c := fc.judged(row)
|
||||
newBand := bandForPercent(pct, w, c)
|
||||
oldBand := fc.states[key] // "" (rank 0) for an unseen / breached-at-init key
|
||||
if bandRank(newBand) > bandRank(oldBand) {
|
||||
fc.emit(row, oldBand, newBand)
|
||||
@@ -153,15 +181,20 @@ func (fc *StorageFillChecker) GetState(hostID, target string) string {
|
||||
|
||||
func (fc *StorageFillChecker) emit(row store.HostStorageTargetRow, oldBand, newBand string) {
|
||||
var eventType, severity, message string
|
||||
pct, warn, crit := fc.judged(row)
|
||||
thin := ""
|
||||
if row.Type == "lvmthin" {
|
||||
thin = fmt.Sprintf(" (thin pool: data %.0f%%, metadata %s) — a full thin pool turns EVERY guest on it read-only", row.Percent, metaLabel(row.MetaPercent))
|
||||
}
|
||||
switch newBand {
|
||||
case bandCritical:
|
||||
eventType = "storage_fill_critical"
|
||||
severity = "critical" // natural critical — hub v0.24.0 routes it; the operator email styles it 🔴
|
||||
message = fmt.Sprintf("Host %s: storage %q CRITICALLY full at %.0f%% (threshold %.0f%%) — backups/writes to it will fail; free space immediately", row.HostID, row.Name, row.Percent, fc.crit)
|
||||
message = fmt.Sprintf("Host %s: storage %q CRITICALLY full at %.0f%% (threshold %.0f%%)%s — backups/writes to it will fail; free space immediately", row.HostID, row.Name, pct, crit, thin)
|
||||
case bandWarning:
|
||||
eventType = "storage_fill_warning"
|
||||
severity = "warning"
|
||||
message = fmt.Sprintf("Host %s: storage %q high at %.0f%% (threshold %.0f%%) — free space before it fills", row.HostID, row.Name, row.Percent, fc.warn)
|
||||
message = fmt.Sprintf("Host %s: storage %q high at %.0f%% (threshold %.0f%%)%s — free space before it fills", row.HostID, row.Name, pct, warn, thin)
|
||||
default:
|
||||
return
|
||||
}
|
||||
@@ -170,14 +203,16 @@ func (fc *StorageFillChecker) emit(row store.HostStorageTargetRow, oldBand, newB
|
||||
"host_id": row.HostID,
|
||||
"storage": row.Name,
|
||||
"storage_type": row.Type,
|
||||
"percent": row.Percent,
|
||||
"percent": pct,
|
||||
"data_percent": row.Percent,
|
||||
"meta_percent": row.MetaPercent,
|
||||
"total_bytes": row.TotalBytes,
|
||||
"used_bytes": row.UsedBytes,
|
||||
"warn_percent": fc.warn,
|
||||
"crit_percent": fc.crit,
|
||||
"warn_percent": warn,
|
||||
"crit_percent": crit,
|
||||
})
|
||||
|
||||
fc.logger.Printf("[INFO] Storage fill: %s %q %.0f%% %s→%s (%s)", row.HostID, row.Name, row.Percent, bandLabel(oldBand), newBand, eventType)
|
||||
fc.logger.Printf("[INFO] Storage fill: %s %q %.0f%% %s→%s (%s)", row.HostID, row.Name, pct, bandLabel(oldBand), newBand, eventType)
|
||||
|
||||
if _, err := fc.store.SaveEvent(row.CustomerID, eventType, severity, message, string(details), "hub"); err != nil {
|
||||
fc.logger.Printf("[WARN] Failed to save storage fill event for %s/%s: %v", row.HostID, row.Name, err)
|
||||
@@ -199,3 +234,10 @@ func bandForPercent(pct, warn, crit float64) string {
|
||||
return bandOK
|
||||
}
|
||||
}
|
||||
|
||||
func metaLabel(p float64) string {
|
||||
if p < 0 {
|
||||
return "unknown"
|
||||
}
|
||||
return fmt.Sprintf("%.0f%%", p)
|
||||
}
|
||||
|
||||
@@ -457,6 +457,33 @@ func cooldownStackSuffixFor(detailsJSON string) string {
|
||||
return ":" + d.StackName
|
||||
}
|
||||
|
||||
// perStorageCooldownEvents (R-672, hub v0.124.0): the storage-fill alarms are keyed PER POOL (host +
|
||||
// storage), with a 6-hour window — "one operator event per pool per 6 hours". The old `customer:type` key
|
||||
// let one host's pool take the hour and silence another pool filling in it, and a 1-hour window re-mailed
|
||||
// a pool that sat full for a day. The checker itself emits only on a band ESCALATION, so the window only
|
||||
// ever collapses a flapping pool.
|
||||
var perStorageCooldownEvents = map[string]bool{
|
||||
"storage_fill_warning": true,
|
||||
"storage_fill_critical": true,
|
||||
}
|
||||
|
||||
const storageFillCooldown = 6 * time.Hour
|
||||
|
||||
// cooldownStorageSuffix returns ":"+host_id+"/"+storage for a per-storage type, else "".
|
||||
func cooldownStorageSuffix(eventType, detailsJSON string) string {
|
||||
if !perStorageCooldownEvents[eventType] || detailsJSON == "" {
|
||||
return ""
|
||||
}
|
||||
var d struct {
|
||||
HostID string `json:"host_id"`
|
||||
Storage string `json:"storage"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(detailsJSON), &d); err != nil || d.Storage == "" {
|
||||
return ""
|
||||
}
|
||||
return ":" + d.HostID + "/" + d.Storage
|
||||
}
|
||||
|
||||
// nodeLivenessEvents skip the 1-hour operator cooldown (OPERATOR RULING 2026-09-15, decision A;
|
||||
// 08-alarm-ladder.md §5). BIGNIGHT F9: the box was dead for 33 minutes and the `node_stale` mail was
|
||||
// suppressed because F8's `node_stale` had used the hour 39 minutes earlier; the `node_recovered`
|
||||
@@ -487,6 +514,9 @@ func operatorCooldownFor(eventType string) time.Duration {
|
||||
if nodeLivenessEvents[eventType] {
|
||||
return nodeLivenessDedupeWindow
|
||||
}
|
||||
if perStorageCooldownEvents[eventType] {
|
||||
return storageFillCooldown
|
||||
}
|
||||
return operatorCooldown
|
||||
}
|
||||
|
||||
@@ -499,7 +529,7 @@ func (d *Dispatcher) processOperator(customerID, eventType, severity, message, d
|
||||
// is byte-identical to v0.107.0's — pinned by TestR389_NoOtherEventTypeKeyChanges.
|
||||
cooldownKey := customerID + ":" + eventType +
|
||||
cooldownTierSuffix(detailsJSON) + cooldownRunSuffix(detailsJSON) +
|
||||
cooldownStackSuffix(eventType, detailsJSON)
|
||||
cooldownStackSuffix(eventType, detailsJSON) + cooldownStorageSuffix(eventType, detailsJSON)
|
||||
window := operatorCooldownFor(eventType)
|
||||
d.mu.Lock()
|
||||
if last, ok := d.opCooldowns[cooldownKey]; ok && time.Since(last) < window {
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
package notify
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-672 (hub v0.124.0): storage-fill alarms are "one operator event per POOL per 6 hours". The
|
||||
// consequence asserted is the mail count.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT): drop cooldownStorageSuffix from the key and the 6-hour window (hub
|
||||
// v0.123.0: `customer:type`, 1 hour) → "a second pool filling in the same hour was silenced".
|
||||
func TestR672_StorageFillIsPerPoolPerSixHours(t *testing.T) {
|
||||
st := newDispStore(t)
|
||||
d := NewDispatcher(st, "test-key", "from@felhom.eu", "op@felhom.eu", true, log.New(io.Discard, "", 0))
|
||||
var mu sync.Mutex
|
||||
sent := 0
|
||||
d.sendEmailFn = func(string, string, string, map[string]string) error { mu.Lock(); sent++; mu.Unlock(); return nil }
|
||||
count := func() int { mu.Lock(); defer mu.Unlock(); return sent }
|
||||
pool := func(host, s string) string { return `{"host_id":"` + host + `","storage":"` + s + `","percent":91}` }
|
||||
|
||||
d.processOperator("c1", "storage_fill_critical", "critical", "pool A full", pool("h1", "local-lvm"), "hub")
|
||||
d.processOperator("c1", "storage_fill_critical", "critical", "pool B full", pool("h1", "fast-lvm"), "hub")
|
||||
if count() != 2 {
|
||||
t.Fatalf("a second pool filling in the same hour was silenced (sent=%d)", count())
|
||||
}
|
||||
d.processOperator("c1", "storage_fill_critical", "critical", "pool A again", pool("h1", "local-lvm"), "hub")
|
||||
if count() != 2 {
|
||||
t.Fatalf("the same pool mailed twice at once (sent=%d)", count())
|
||||
}
|
||||
// Two hours later the SAME pool is still inside its 6-hour window (the old 1-hour window re-mailed it).
|
||||
d.mu.Lock()
|
||||
for k := range d.opCooldowns {
|
||||
d.opCooldowns[k] = time.Now().Add(-2 * time.Hour)
|
||||
}
|
||||
d.mu.Unlock()
|
||||
d.processOperator("c1", "storage_fill_critical", "critical", "pool A 2h later", pool("h1", "local-lvm"), "hub")
|
||||
if count() != 2 {
|
||||
t.Fatalf("the same pool was mailed again after 2 hours — want one per 6 hours (sent=%d)", count())
|
||||
}
|
||||
d.mu.Lock()
|
||||
for k := range d.opCooldowns {
|
||||
d.opCooldowns[k] = time.Now().Add(-7 * time.Hour)
|
||||
}
|
||||
d.mu.Unlock()
|
||||
d.processOperator("c1", "storage_fill_critical", "critical", "pool A 7h later", pool("h1", "local-lvm"), "hub")
|
||||
if count() != 3 {
|
||||
t.Fatalf("after 6 hours the pool must alarm again (sent=%d)", count())
|
||||
}
|
||||
}
|
||||
@@ -3730,6 +3730,9 @@ type HostStorageTargetRow struct {
|
||||
Percent float64 // 0..100 (used_fraction × 100)
|
||||
TotalBytes int64
|
||||
UsedBytes int64
|
||||
// MetaPercent is an lvmthin pool's METADATA fill, 0..100; -1 when the report carries none (R-672,
|
||||
// hub v0.124.0 — a thin pool is corrupted by metadata exhaustion exactly like data exhaustion).
|
||||
MetaPercent float64
|
||||
}
|
||||
|
||||
// GetHostStorageTargets returns the per-storage fill of every host's LATEST report (MAX(id) per host),
|
||||
@@ -3759,14 +3762,22 @@ func (s *Store) GetHostStorageTargets() ([]HostStorageTargetRow, error) {
|
||||
UsedFraction float64 `json:"used_fraction"`
|
||||
TotalBytes int64 `json:"total_bytes"`
|
||||
UsedBytes int64 `json:"used_bytes"`
|
||||
ThinPool *struct {
|
||||
MetadataUsedFraction *float64 `json:"metadata_used_fraction"`
|
||||
} `json:"thin_pool"`
|
||||
} `json:"storage_targets"`
|
||||
}
|
||||
_ = json.Unmarshal([]byte(reportJSON), &body) // malformed/old body → no targets (never a false alert)
|
||||
for _, t := range body.StorageTargets {
|
||||
meta := -1.0
|
||||
if t.ThinPool != nil && t.ThinPool.MetadataUsedFraction != nil {
|
||||
meta = *t.ThinPool.MetadataUsedFraction * 100
|
||||
}
|
||||
out = append(out, HostStorageTargetRow{
|
||||
HostID: hostID, CustomerID: customerID,
|
||||
Name: t.Name, Type: t.Type, MountPath: t.MountPath,
|
||||
Percent: t.UsedFraction * 100, TotalBytes: t.TotalBytes, UsedBytes: t.UsedBytes,
|
||||
MetaPercent: meta,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user