fd50a73e65
Both are the system reporting healthy while the customer is not, and both live in the same status-derivation code. Neither is fixed by making the system quieter. C9-F1 (HIGH) — Tier-2 writes recovery-unit/ on EVERY run and RestoreTier2Files has never read it (tier2_restore.go:101-104 reads hdd/ + userdata/ only). Phase 0 enumerated all 53 catalog templates against both demo boxes: 43 apps have NO readable subtree, so the button stopped the app, restored 0 files, restarted it and said "Nincs hiányzó fájl — minden fájl megvan a helyén." — at the moment the customer pressed it because files were missing, with 156 MB of BookStack's data unread in the same copy. 9 apps have file legs but never their DB or volumes, so the same sentence was also a clean bill of health over data never opened (immich: 1.3 GB Postgres unit). Honesty half shipped: a pre-flight coverage check refuses UP FRONT without stopping the app and NAMES the action that works; a run that proceeds claims only what it EXAMINED and discloses that the database and volumes are not covered. Completeness is filed as C9-F1b — routing to the Tier-1 unit restore puts a destructive operation behind a non-destructive button, so its confirm copy has to carry that difference. C9-F4 filed: nothing reads the Tier-2 recovery-unit/ mirror, so the second local copy that exists for drive loss is unreachable by any customer action. C9-F2 (HIGH) — a crash loop was counted as working. StateRestarting is deliberately NOT added to IsDownState (that alarms on every deploy fleet-wide, the over-correction F-A1 nearly cost us); a sustained run becomes down after crashLoopAfter = 5m, set above the 120s deploy timeout, Mealie's 60s start_period and R-97b's 180s grace. The dashboard counter uses the same predicate, so it no longer contradicts the alarm on the same screen. README's claim that faults "still surface as restarting" was a wish with no test — corrected in place; it is the seventh such instance. Six red-proofs observed, including the one that matters most: adding StateRestarting to IsDownState fails the brief-restart test with "every deploy and update would page the operator". go test ./... rc=0, 27 packages, run and read separately from this commit.
126 lines
4.6 KiB
Go
126 lines
4.6 KiB
Go
package main
|
|
|
|
import (
|
|
"time"
|
|
|
|
"testing"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/web"
|
|
)
|
|
|
|
// v0.164.0: classifyRunStates is the single fix-3 derivation point. A deliberate user stop
|
|
// (StateStopped) must NOT alarm — it is excluded from both the banner dead-list and the notifier
|
|
// Down-set — while every genuine fault (StateExited / StateDegraded) keeps alerting byte-identically.
|
|
// Invariants behind the suppression are documented at classifyRunStates (I1: compose down ⇒ zero
|
|
// containers ⇒ StateStopped; I2: P2 census — all catalog services unless-stopped ⇒ faults never rest
|
|
// at stopped).
|
|
|
|
func stack(name string, st stacks.ContainerState, deployed, deploying bool) stacks.Stack {
|
|
return stacks.Stack{
|
|
Name: name,
|
|
Meta: stacks.Metadata{DisplayName: name},
|
|
State: st,
|
|
Deployed: deployed,
|
|
Deploying: deploying,
|
|
}
|
|
}
|
|
|
|
func downByName(states []notify.AppRunState) map[string]bool {
|
|
m := map[string]bool{}
|
|
for _, s := range states {
|
|
m[s.Name] = s.Down
|
|
}
|
|
return m
|
|
}
|
|
|
|
func deadNames(dead []web.DeadApp) map[string]bool {
|
|
m := map[string]bool{}
|
|
for _, d := range dead {
|
|
m[d.Name] = true
|
|
}
|
|
return m
|
|
}
|
|
|
|
// Group A (Scenario A) — suppression. Over a [running, stopped, exited, degraded] fixture, the dead
|
|
// list is EXACTLY {exited, degraded} and the Down flags are {false, false, true, true}: the stopped
|
|
// app is silent, the two faults still alarm.
|
|
//
|
|
// COMPANION red-proof: revert the filter to bare `stacks.IsDownState(st.State)` (drop the
|
|
// `&& st.State != stacks.StateStopped` guard) → stopped reports Down=true and enters the dead list →
|
|
// both the dead-set and the Down-flag assertions below fail. (Verified by hand-editing the seam.)
|
|
func TestClassifyRunStates_StoppedIsSuppressed(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("radarr", stacks.StateRunning, true, false),
|
|
stack("cwa", stacks.StateStopped, true, false),
|
|
stack("immich", stacks.StateExited, true, false),
|
|
stack("nextcloud", stacks.StateDegraded, true, false),
|
|
}
|
|
|
|
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
|
|
|
gotDead := deadNames(dead)
|
|
if len(gotDead) != 2 || !gotDead["immich"] || !gotDead["nextcloud"] {
|
|
t.Fatalf("dead list must be exactly {immich(exited), nextcloud(degraded)}, got %+v", dead)
|
|
}
|
|
if gotDead["cwa"] {
|
|
t.Errorf("a deliberately stopped app must NOT be in the dead list (no banner)")
|
|
}
|
|
if gotDead["radarr"] {
|
|
t.Errorf("a running app must never be in the dead list")
|
|
}
|
|
|
|
down := downByName(states)
|
|
want := map[string]bool{"radarr": false, "cwa": false, "immich": true, "nextcloud": true}
|
|
if len(down) != len(want) {
|
|
t.Fatalf("every deployed app must have a run state, got %+v", down)
|
|
}
|
|
for name, w := range want {
|
|
if down[name] != w {
|
|
t.Errorf("Down[%s] = %v, want %v (stopped ⇒ false ⇒ no app_start_failed event)", name, down[name], w)
|
|
}
|
|
}
|
|
}
|
|
|
|
// Group B (Scenario B) — fault parity. With only exited + degraded present, BOTH surface in the dead
|
|
// list AND both report Down=true — byte-identical to v0.163.1 for every non-stopped down state. The
|
|
// suppression touches stopped and nothing else.
|
|
func TestClassifyRunStates_FaultParity(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("immich", stacks.StateExited, true, false),
|
|
stack("nextcloud", stacks.StateDegraded, true, false),
|
|
}
|
|
|
|
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
|
|
|
gotDead := deadNames(dead)
|
|
if len(gotDead) != 2 || !gotDead["immich"] || !gotDead["nextcloud"] {
|
|
t.Fatalf("both faults must appear in the dead list, got %+v", dead)
|
|
}
|
|
down := downByName(states)
|
|
if !down["immich"] || !down["nextcloud"] {
|
|
t.Fatalf("both faults must report Down=true, got %+v", down)
|
|
}
|
|
// State strings must ride through to the banner unchanged (banner shows "(exited)"/"(degraded)").
|
|
byName := map[string]string{}
|
|
for _, d := range dead {
|
|
byName[d.Name] = d.State
|
|
}
|
|
if byName["immich"] != string(stacks.StateExited) || byName["nextcloud"] != string(stacks.StateDegraded) {
|
|
t.Errorf("dead-app State must carry the raw aggregate state, got %+v", byName)
|
|
}
|
|
}
|
|
|
|
// Deploying and undeployed stacks are skipped entirely (unchanged fix-3 behavior).
|
|
func TestClassifyRunStates_SkipsDeployingAndUndeployed(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("mid", stacks.StateDeploying, true, true), // mid-deploy → skipped
|
|
stack("gone", stacks.StateExited, false, false), // not deployed → skipped
|
|
}
|
|
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
|
if len(dead) != 0 || len(states) != 0 {
|
|
t.Fatalf("deploying and undeployed stacks must be skipped, got dead=%+v states=%+v", dead, states)
|
|
}
|
|
}
|