fix(disk-health): one physical disk must be evaluated once per run (R-335)
gates / gates (push) Successful in 9s
gates / gates (push) Successful in 9s
Found on live hardware two hours after the v0.215.0 deploy, by noticing the release's own positive observable disagreed with its own persisted artefact: the check logged '3 disk(s) evaluated' while disk-health-state.json held two records. demo-hp's c11-scratch and felhom-backup are the same NVMe and share a durable id, so one disk was walked twice per run. Not cosmetic. The loop writes a disk's record before the next entry reads it, so the second copy of an aliased disk consumed the FIRST copy's write as its prior: the disk sustained against ITSELF and reached Hiba on a first sighting, defeating truth-table row 6 — the rule that separates a one-hour benign excursion from a false critical. It would also have emitted two identical events for one drive. Latent on demo-hp only because all counters are zero. Each diskKey is now evaluated once per run. Both entries stay marked seen so neither looks like a disappeared disk, and the card still renders both rows — the dedup is about state and alerts, not display. Red-proof run and reverted: deleting the guard makes the first sighting emit Kind:2 (Hiba-from-sectors) at 8 sectors.
This commit is contained in:
@@ -554,6 +554,62 @@ func TestDiskCheck_DisappearedDiskIsForgotten(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// ── One physical disk, several storage entries ──────────────────────────────────────────────────
|
||||
|
||||
// FOUND ON LIVE HARDWARE (demo-hp, 2026-08-14): the check reported "3 disk(s) evaluated" while the
|
||||
// persisted state held only TWO records, because `c11-scratch` and `felhom-backup` are the same NVMe
|
||||
// and resolve to the same durable id.
|
||||
//
|
||||
// That aliasing was a real defect, not a cosmetic one. The loop writes this run's record before the
|
||||
// next entry reads it, so the second copy of a disk consumed the FIRST copy's write as its prior —
|
||||
// the disk sustained against ITSELF and reached Hiba on a first sighting, defeating truth-table row 6
|
||||
// entirely, and emitted two identical events while doing it.
|
||||
//
|
||||
// Red-proof: delete the `if seen[key] { continue }` guard in RunDiskHealthCheck → the first run
|
||||
// reaches Fail and emits, and both assertions below fail.
|
||||
func TestDiskCheck_SameDiskTwiceIsEvaluatedOnce(t *testing.T) {
|
||||
h := newDiskHarness(t)
|
||||
// Two storage entries, ONE physical device — identical durable id, as the agent really reports it.
|
||||
twice := func() []agentapi.DiskInfo {
|
||||
sm := &agentapi.SmartSummary{Health: agentapi.SmartPassed,
|
||||
PendingSectors: smartPtr(8), OfflineUncorrectable: smartPtr(8), ReallocatedSectors: smartPtr(0)}
|
||||
return []agentapi.DiskInfo{
|
||||
{Name: "c11-scratch", Type: "local-dir", BackingDevice: "/dev/nvme0n1", DurableID: "uuid:nvme-1", Smart: sm},
|
||||
{Name: "felhom-backup", Type: "local-dir", BackingDevice: "/dev/nvme0n1", DurableID: "uuid:nvme-1", Smart: sm},
|
||||
}
|
||||
}
|
||||
h.set(twice()...)
|
||||
h.run(t)
|
||||
|
||||
// FIRST sighting: silent, and Figyelmeztetés — a disk must not sustain against its own alias.
|
||||
if n := len(h.events()); n != 0 {
|
||||
t.Fatalf("first sighting of an aliased disk must be silent, got %d: %+v", n, h.events())
|
||||
}
|
||||
rec := h.record(t, "uuid:nvme-1")
|
||||
if rec == nil {
|
||||
t.Fatal("no record for the aliased disk")
|
||||
}
|
||||
if agentapi.DiskVerdict(rec.Verdict) != agentapi.DiskVerdictWarn {
|
||||
t.Errorf("first sighting verdict = %d, want Warn — the disk sustained against its own alias",
|
||||
rec.Verdict)
|
||||
}
|
||||
if rec.PriorSawUncorrectable {
|
||||
t.Error("the prior used on a first sighting must be empty, not this run's own write")
|
||||
}
|
||||
|
||||
// SECOND run: now genuinely sustained → exactly ONE event, not one per storage entry.
|
||||
h.set(twice()...)
|
||||
h.run(t)
|
||||
if n := len(h.events()); n != 1 {
|
||||
t.Errorf("one physical disk must produce ONE event, got %d: %+v", n, h.events())
|
||||
}
|
||||
|
||||
// Both entries still render on the card — dedup is about state and alerts, not display.
|
||||
if rows := h.s.diskHealthRows(context.Background()); len(rows) != 2 {
|
||||
t.Errorf("card must still show both storage entries, got %d rows", len(rows))
|
||||
}
|
||||
}
|
||||
|
||||
// An unreachable agent changes nothing at all — no state churn, no alarm, no error.
|
||||
func TestDiskCheck_UnreachableAgentIsInert(t *testing.T) {
|
||||
h := newDiskHarness(t)
|
||||
|
||||
Reference in New Issue
Block a user