fd50a73e65
Both are the system reporting healthy while the customer is not, and both live in the same status-derivation code. Neither is fixed by making the system quieter. C9-F1 (HIGH) — Tier-2 writes recovery-unit/ on EVERY run and RestoreTier2Files has never read it (tier2_restore.go:101-104 reads hdd/ + userdata/ only). Phase 0 enumerated all 53 catalog templates against both demo boxes: 43 apps have NO readable subtree, so the button stopped the app, restored 0 files, restarted it and said "Nincs hiányzó fájl — minden fájl megvan a helyén." — at the moment the customer pressed it because files were missing, with 156 MB of BookStack's data unread in the same copy. 9 apps have file legs but never their DB or volumes, so the same sentence was also a clean bill of health over data never opened (immich: 1.3 GB Postgres unit). Honesty half shipped: a pre-flight coverage check refuses UP FRONT without stopping the app and NAMES the action that works; a run that proceeds claims only what it EXAMINED and discloses that the database and volumes are not covered. Completeness is filed as C9-F1b — routing to the Tier-1 unit restore puts a destructive operation behind a non-destructive button, so its confirm copy has to carry that difference. C9-F4 filed: nothing reads the Tier-2 recovery-unit/ mirror, so the second local copy that exists for drive loss is unreachable by any customer action. C9-F2 (HIGH) — a crash loop was counted as working. StateRestarting is deliberately NOT added to IsDownState (that alarms on every deploy fleet-wide, the over-correction F-A1 nearly cost us); a sustained run becomes down after crashLoopAfter = 5m, set above the 120s deploy timeout, Mealie's 60s start_period and R-97b's 180s grace. The dashboard counter uses the same predicate, so it no longer contradicts the alarm on the same screen. README's claim that faults "still surface as restarting" was a wish with no test — corrected in place; it is the seventh such instance. Six red-proofs observed, including the one that matters most: adding StateRestarting to IsDownState fails the brief-restart test with "every deploy and update would page the operator". go test ./... rc=0, 27 packages, run and read separately from this commit.
125 lines
5.2 KiB
Go
125 lines
5.2 KiB
Go
package main
|
|
|
|
import (
|
|
"time"
|
|
|
|
"testing"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
|
)
|
|
|
|
// F-CRIT-1 cause 2 (Campaign 8): `classifyRunStates` whitelisted StateStopped on invariant I1
|
|
// ("StateStopped means the USER stopped it"). The quiesce loop broke I1 by stopping stacks via the
|
|
// same `docker compose down` path, so a stack quiesce stopped and then FAILED to restart was also
|
|
// StateStopped — and was whitelisted into total silence. Live evidence: a customer app dead
|
|
// indefinitely, no banner, no event, no email, while the dead-app scanner ran 11 times over it.
|
|
//
|
|
// The two cases are byte-identical on the Docker side. The ONLY thing that separates them is that
|
|
// the quiesce loop knows it tried to restart and could not — `failedRestart` is that knowledge.
|
|
|
|
// Scenario A (cause 2) — a stack quiesce failed to restart MUST alarm, despite being StateStopped.
|
|
//
|
|
// RED-PROOF: restore the unconditional whitelist (`down := IsDownState(st.State) &&
|
|
// st.State != stacks.StateStopped && !quiesced[st.Name]`) → immich reports Down=false and stays out
|
|
// of the dead list, and this fails with "a stack that FAILED to restart is silent".
|
|
func TestClassifyRunStates_FailedRestartAlarmsDespiteStateStopped(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("bookstack", stacks.StateRunning, true, false),
|
|
stack("immich", stacks.StateStopped, true, false), // quiesce stopped it; restart FAILED
|
|
}
|
|
failed := map[string]bool{"immich": true}
|
|
|
|
dead, states := classifyRunStates(sts, nil, failed, time.Now())
|
|
|
|
if !downByName(states)["immich"] {
|
|
t.Error("a stack that FAILED to restart is silent (Down=false) — this is F-CRIT-1")
|
|
}
|
|
if !deadNames(dead)["immich"] {
|
|
t.Error("a stack that FAILED to restart is absent from the dashboard dead-list — this is F-CRIT-1")
|
|
}
|
|
if downByName(states)["bookstack"] {
|
|
t.Error("a healthy running stack was marked down")
|
|
}
|
|
}
|
|
|
|
// Scenario B — a DELIBERATE user stop must still be silent. This pins v0.164.0 and is what stops
|
|
// the fix above from becoming a regression.
|
|
//
|
|
// RED-PROOF: make the whitelist unconditional in the other direction (drop the `&& !failedRestart`
|
|
// term, i.e. treat every StateStopped as a failed restart) → cwa alarms and this fails with
|
|
// "a deliberate user stop alarmed".
|
|
func TestClassifyRunStates_UserStopStillSilent(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("cwa", stacks.StateStopped, true, false), // the user stopped this from the UI
|
|
stack("immich", stacks.StateStopped, true, false),
|
|
}
|
|
// only immich failed to restart; cwa was never touched by a quiesce
|
|
failed := map[string]bool{"immich": true}
|
|
|
|
dead, states := classifyRunStates(sts, nil, failed, time.Now())
|
|
down := downByName(states)
|
|
|
|
if down["cwa"] || deadNames(dead)["cwa"] {
|
|
t.Error("a deliberate user stop alarmed — that is the v0.164.0 regression this must not reintroduce")
|
|
}
|
|
if !down["immich"] {
|
|
t.Error("the failed restart went silent")
|
|
}
|
|
}
|
|
|
|
// Scenario B, stronger form — with NO failed restarts at all, behaviour is byte-identical to
|
|
// v0.164.0: every StateStopped is silent.
|
|
func TestClassifyRunStates_NoFailedRestartsIsV0164Behaviour(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("radarr", stacks.StateRunning, true, false),
|
|
stack("cwa", stacks.StateStopped, true, false),
|
|
stack("immich", stacks.StateExited, true, false),
|
|
stack("nextcloud", stacks.StateDegraded, true, false),
|
|
}
|
|
|
|
dead, states := classifyRunStates(sts, nil, nil, time.Now())
|
|
down := downByName(states)
|
|
|
|
if down["cwa"] {
|
|
t.Error("stopped alarmed with no failed restarts — v0.164.0 behaviour broken")
|
|
}
|
|
if !down["immich"] || !down["nextcloud"] {
|
|
t.Error("a genuine fault (exited/degraded) stopped alarming")
|
|
}
|
|
if got := len(deadNames(dead)); got != 2 {
|
|
t.Errorf("dead list has %d entries, want exactly {immich, nextcloud}", got)
|
|
}
|
|
}
|
|
|
|
// Scenario C — during the R-97b grace window the stack is suppressed even if its restart failed.
|
|
// The grace exists so a slow-starting app is not called dead; it EXPIRES, and the alarm follows.
|
|
//
|
|
// RED-PROOF: drop the `&& !quiesced[st.Name]` term → the app alarms mid-restart on every normal
|
|
// backup, which is the false-alarm R-97b was built to remove.
|
|
func TestClassifyRunStates_GraceWindowStillSuppresses(t *testing.T) {
|
|
sts := []stacks.Stack{stack("immich", stacks.StateStopped, true, false)}
|
|
quiesced := map[string]bool{"immich": true} // still inside quiesceAlarmGrace
|
|
failed := map[string]bool{"immich": true} // and we already know the restart failed
|
|
|
|
dead, states := classifyRunStates(sts, quiesced, failed, time.Now())
|
|
|
|
if downByName(states)["immich"] {
|
|
t.Error("alarmed while still inside the grace window — R-97b Scenario E broken")
|
|
}
|
|
if len(dead) != 0 {
|
|
t.Errorf("dead list not empty during grace: %v", deadNames(dead))
|
|
}
|
|
}
|
|
|
|
// An undeployed or mid-deploy stack is never classified, failed restart or not.
|
|
func TestClassifyRunStates_UndeployedIgnored(t *testing.T) {
|
|
sts := []stacks.Stack{
|
|
stack("ghost", stacks.StateStopped, false, false),
|
|
stack("deploying", stacks.StateStopped, true, true),
|
|
}
|
|
dead, states := classifyRunStates(sts, nil, map[string]bool{"ghost": true, "deploying": true}, time.Now())
|
|
if len(dead) != 0 || len(states) != 0 {
|
|
t.Errorf("undeployed/deploying stacks were classified: dead=%v states=%v", deadNames(dead), states)
|
|
}
|
|
}
|