package localapi import ( "context" "encoding/json" "errors" "io" "log/slog" "os" "path/filepath" "strconv" "sync" "testing" "time" "gitea.dooplex.hu/admin/felhom-agent/internal/proxmox" ) // R-523 — the controller supervisor. The consequence under test is "a dead controller is started // again", and each guard is pinned by the case where acting would be wrong. type supExec struct { mu sync.Mutex status map[int]string // docker .State.Status per vmid; "" = container absent inspectErr error // non-nil = pct exec itself failed (unknown) restarts map[int]int // onRestart, when set, is the status the container reaches after a restart (a crash-looper // stays "exited"). onRestart string } func (f *supExec) GuestExec(_ context.Context, vmid int, args ...string) (string, error) { f.mu.Lock() defer f.mu.Unlock() switch { case len(args) >= 5 && args[0] == "docker" && args[1] == "inspect" && args[3] == "{{.State.Status}}": if f.inspectErr != nil { return "", f.inspectErr } st, ok := f.status[vmid] if !ok || st == "" { return "", errors.New("pct exec: exit status 1: Error: No such object: felhom-controller") } return st + "\n", nil case len(args) == 3 && args[0] == "systemctl" && args[1] == "restart" && args[2] == bootstrapUnit: if f.restarts == nil { f.restarts = map[int]int{} } f.restarts[vmid]++ if f.onRestart != "" { f.status[vmid] = f.onRestart } return "", nil } return "", errors.New("supExec: unexpected args") } func (f *supExec) GuestExecStdin(context.Context, int, io.Reader, ...string) (string, error) { return "", errors.New("supExec: no stdin exec expected") } func (f *supExec) count(vmid int) int { f.mu.Lock() defer f.mu.Unlock() return f.restarts[vmid] } type supClock struct{ t time.Time } func (c *supClock) now() time.Time { return c.t } func supServer(t *testing.T, ex *supExec, ctl *fakeGuestPowerCtl, provisioned ...int) (*Server, *supClock, string) { t.Helper() dir := t.TempDir() for _, v := range provisioned { if err := os.MkdirAll(filepath.Join(dir, strconv.Itoa(v), "bootstrap"), 0o700); err != nil { t.Fatal(err) } } clk := &supClock{t: time.Date(2026, 9, 15, 10, 0, 0, 0, time.UTC)} s := &Server{ staleLock: ctl, guestExec: ex, guestsDir: dir, swapInFlight: map[int]bool{}, logger: slog.New(slog.NewTextHandler(discardW{}, nil)), now: clk.now, } return s, clk, dir } func runningGuest(vmid int) *fakeGuestPowerCtl { return &fakeGuestPowerCtl{ guests: []proxmox.Guest{{VMID: vmid, Status: "running"}}, locks: map[int]string{}, onboot: map[int]bool{vmid: true}, } } // The consequence: a killed controller IS restarted — on the second consecutive observation, not // the first (the bootstrap's own rm-f/run window must never be raced). // // RED-PROOF: delete the `systemctl restart` GuestExec call in superviseOneController → restarts // stays 0 → "the killed controller was NOT restarted — this is R-523". func TestControllerSupervisor_KilledControllerIsRestarted(t *testing.T) { ex := &supExec{status: map[int]string{9201: "exited"}, onRestart: "running"} s, _, _ := supServer(t, ex, runningGuest(9201), 9201) s.ControllerSupervisorTick(context.Background()) if n := ex.count(9201); n != 0 { t.Fatalf("restarted on the FIRST observation (restarts=%d) — must confirm on a second sweep", n) } s.ControllerSupervisorTick(context.Background()) if n := ex.count(9201); n != 1 { t.Fatalf("the killed controller was NOT restarted — this is R-523 (restarts=%d)", n) } st := s.ControllerSupervisorStatus(context.Background()) if len(st.Guests) != 1 || st.Guests[0].RestartsTotal != 1 || st.Guests[0].LastRestartAt == "" || st.Guests[0].LastReason == "" { t.Fatalf("report stanza did not record the restart: %+v", st.Guests) } // Healthy again → no further restart. s.ControllerSupervisorTick(context.Background()) s.ControllerSupervisorTick(context.Background()) if n := ex.count(9201); n != 1 { t.Fatalf("a running controller was restarted again (restarts=%d)", n) } } func TestControllerSupervisor_AbsentContainerIsRestarted(t *testing.T) { ex := &supExec{status: map[int]string{}, onRestart: "running"} s, _, _ := supServer(t, ex, runningGuest(9201), 9201) s.ControllerSupervisorTick(context.Background()) s.ControllerSupervisorTick(context.Background()) if n := ex.count(9201); n != 1 { t.Fatalf("a removed controller container was not restarted (restarts=%d)", n) } } func TestControllerSupervisor_Guards(t *testing.T) { cases := []struct { name string setup func(s *Server, ex *supExec, ctl *fakeGuestPowerCtl, dir string) }{ {"parked", func(s *Server, _ *supExec, _ *fakeGuestPowerCtl, dir string) { if err := os.WriteFile(filepath.Join(dir, "9201", ControllerParkedMarker), nil, 0o600); err != nil { panic(err) } }}, {"swap in flight", func(s *Server, _ *supExec, _ *fakeGuestPowerCtl, _ string) { s.swapInFlight[9201] = true }}, {"guest locked", func(_ *Server, _ *supExec, ctl *fakeGuestPowerCtl, _ string) { ctl.locks[9201] = "backup" }}, {"vzdump running", func(_ *Server, _ *supExec, ctl *fakeGuestPowerCtl, _ string) { ctl.backupRun = map[int]bool{9201: true} }}, {"vzdump state unknown", func(_ *Server, _ *supExec, ctl *fakeGuestPowerCtl, _ string) { ctl.backupErr = errors.New("tasks unreadable") }}, {"guest not running", func(_ *Server, _ *supExec, ctl *fakeGuestPowerCtl, _ string) { ctl.guests[0].Status = "stopped" }}, {"docker state unknown", func(_ *Server, ex *supExec, _ *fakeGuestPowerCtl, _ string) { ex.inspectErr = errors.New("pct exec 9201: exit status 255: container not running") }}, {"guest list unavailable", func(_ *Server, _ *supExec, ctl *fakeGuestPowerCtl, _ string) { ctl.guestsErr = errors.New("api down") }}, } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { ex := &supExec{status: map[int]string{9201: "exited"}, onRestart: "running"} ctl := runningGuest(9201) s, _, dir := supServer(t, ex, ctl, 9201) tc.setup(s, ex, ctl, dir) for i := 0; i < 4; i++ { s.ControllerSupervisorTick(context.Background()) } if n := ex.count(9201); n != 0 { t.Fatalf("guard %q did not hold: controller restarted %d time(s)", tc.name, n) } }) } } // A guest the agent did not provision (no //bootstrap) is never touched. func TestControllerSupervisor_UnprovisionedGuestIgnored(t *testing.T) { ex := &supExec{status: map[int]string{9202: "exited"}} s, _, _ := supServer(t, ex, runningGuest(9202) /* nothing provisioned */) for i := 0; i < 3; i++ { s.ControllerSupervisorTick(context.Background()) } if n := ex.count(9202); n != 0 { t.Fatalf("an unprovisioned guest's container was restarted (%d)", n) } } // No thrash: a controller that will not stay up is restarted at most controllerCrashloopMax times // inside the window, then the supervisor raises the crash-loop and pauses; after the pause it tries // again. // // RED-PROOF (run 2026-09-15, recorded in REPORT.md): with the `len(st.restarts) >= // controllerCrashloopMax` block removed, restarts reached 10 in the first 20 sweeps and the test // failed at "crash-looping controller restarted 10 times". func TestControllerSupervisor_CrashloopBackoff(t *testing.T) { ex := &supExec{status: map[int]string{9201: "exited"}, onRestart: "exited"} s, clk, _ := supServer(t, ex, runningGuest(9201), 9201) ctx := context.Background() for i := 0; i < 20; i++ { // 10 minutes of 30 s sweeps s.ControllerSupervisorTick(ctx) clk.t = clk.t.Add(controllerSupervisorInterval) } if n := ex.count(9201); n != controllerCrashloopMax { t.Fatalf("crash-looping controller restarted %d times in 10 minutes — want exactly %d then a pause", n, controllerCrashloopMax) } st := s.ControllerSupervisorStatus(ctx).Guests[0] if !st.Crashloop || st.CrashloopSince == "" { t.Fatalf("crash-loop not raised in the report stanza: %+v", st) } // Still paused 25 minutes later. clk.t = clk.t.Add(15 * time.Minute) s.ControllerSupervisorTick(ctx) s.ControllerSupervisorTick(ctx) if n := ex.count(9201); n != controllerCrashloopMax { t.Fatalf("restarted during the crash-loop pause (restarts=%d)", n) } // After the pause: tries again. clk.t = clk.t.Add(controllerCrashloopPause) s.ControllerSupervisorTick(ctx) s.ControllerSupervisorTick(ctx) if n := ex.count(9201); n != controllerCrashloopMax+1 { t.Fatalf("did not resume after the pause (restarts=%d, want %d)", n, controllerCrashloopMax+1) } if since := s.ControllerSupervisorStatus(ctx).Guests[0].CrashloopSince; since != st.CrashloopSince && since != "" { t.Fatalf("crashloop_since changed without a new crash-loop: %q → %q", st.CrashloopSince, since) } } // The wire shape the hub parses. The hub's controller_supervisor_test.go carries this exact JSON. func TestControllerSupervisorStanza_WireShape(t *testing.T) { ex := &supExec{status: map[int]string{9201: "exited"}, onRestart: "running"} s, _, _ := supServer(t, ex, runningGuest(9201), 9201) s.ControllerSupervisorTick(context.Background()) s.ControllerSupervisorTick(context.Background()) b, _ := json.Marshal(s.ControllerSupervisorStatus(context.Background())) var m map[string][]map[string]any if err := json.Unmarshal(b, &m); err != nil { t.Fatal(err) } g := m["guests"][0] for _, k := range []string{"vmid", "restarts_total", "last_restart_at", "last_reason", "crashloop", "parked", "restarts_24h", "slow_crashloop"} { if _, ok := g[k]; !ok { t.Fatalf("stanza lacks %q — the hub keys on it: %s", k, b) } } } // ---- R-539 (operator ruling 3 of 2026-09-16): the SLOW crash loop ---------------------------------- // supKillOnce kills the controller and lets the supervisor restart it (two confirming sweeps), then // moves the clock on by gap. The container comes back "running", so each restart is a separate act. func supKillOnce(t *testing.T, s *Server, ex *supExec, clk *supClock, gap time.Duration) { t.Helper() before := ex.count(9201) ex.mu.Lock() ex.status[9201] = "exited" ex.mu.Unlock() s.ControllerSupervisorTick(context.Background()) clk.t = clk.t.Add(controllerSupervisorInterval) s.ControllerSupervisorTick(context.Background()) if ex.count(9201) != before+1 { t.Fatalf("kill was not followed by exactly one restart (restarts %d → %d)", before, ex.count(9201)) } clk.t = clk.t.Add(gap) } // The consequence: a controller that dies every 20 minutes — never three times inside the 15-minute // brake — raises slow_crashloop on the FIFTH restart in 24 hours, and the raise does not move again on // the sixth (the hub mails on movement; once per 24 hours is the ruling). // // RED-PROOF: without the slow counter the stanza never sets slow_crashloop → "five restarts 20 minutes // apart did not raise slow_crashloop — this is R-539". func TestControllerSupervisor_SlowCrashloop(t *testing.T) { ex := &supExec{status: map[int]string{9201: "running"}, onRestart: "running"} s, clk, _ := supServer(t, ex, runningGuest(9201), 9201) ctx := context.Background() for i := 1; i <= 4; i++ { supKillOnce(t, s, ex, clk, 20*time.Minute) } g := s.ControllerSupervisorStatus(ctx).Guests[0] if g.Crashloop { t.Fatalf("the 15-minute brake fired on restarts 20 minutes apart — the fixture is wrong: %+v", g) } if g.SlowCrashloop || g.SlowCrashloopSince != "" { t.Fatalf("slow_crashloop raised after only 4 restarts: %+v", g) } supKillOnce(t, s, ex, clk, 20*time.Minute) g = s.ControllerSupervisorStatus(ctx).Guests[0] if !g.SlowCrashloop || g.SlowCrashloopSince == "" || g.Restarts24h != 5 { t.Fatalf("five restarts 20 minutes apart did not raise slow_crashloop — this is R-539: %+v", g) } first := g.SlowCrashloopSince supKillOnce(t, s, ex, clk, 20*time.Minute) g = s.ControllerSupervisorStatus(ctx).Guests[0] if g.SlowCrashloopSince != first { t.Fatalf("the raise moved again on the 6th restart (%q → %q) — the operator would be mailed per restart", first, g.SlowCrashloopSince) } if !g.SlowCrashloop { t.Fatalf("slow_crashloop cleared while the loop continues: %+v", g) } } // The negative control: restarts that never reach five inside any 24 hours never raise it. func TestControllerSupervisor_SpreadRestartsNeverSlowCrashloop(t *testing.T) { ex := &supExec{status: map[int]string{9201: "running"}, onRestart: "running"} s, clk, _ := supServer(t, ex, runningGuest(9201), 9201) for i := 0; i < 8; i++ { // eight restarts, 7 hours apart: at most 4 inside any 24 hours supKillOnce(t, s, ex, clk, 7*time.Hour) } g := s.ControllerSupervisorStatus(context.Background()).Guests[0] if g.SlowCrashloop || g.SlowCrashloopSince != "" { t.Fatalf("restarts 7 hours apart raised slow_crashloop: %+v", g) } if g.Restarts24h > 4 { t.Fatalf("restarts_24h=%d — the 24-hour window is not pruning", g.Restarts24h) } } // An agent restart must not reset the slow counter (the ruling; a box whose AGENT also restarts would // otherwise never reach five). The same state directory, a fresh Server. // // RED-PROOF: keep the counter in memory only → the second Server starts at 0 → "the agent restart // reset the slow counter". func TestControllerSupervisor_SlowCounterSurvivesAgentRestart(t *testing.T) { ex := &supExec{status: map[int]string{9201: "running"}, onRestart: "running"} s, clk, dir := supServer(t, ex, runningGuest(9201), 9201) for i := 0; i < 4; i++ { supKillOnce(t, s, ex, clk, 20*time.Minute) } s2 := &Server{ staleLock: s.staleLock, guestExec: ex, guestsDir: dir, swapInFlight: map[int]bool{}, logger: s.logger, now: clk.now, } supKillOnce(t, s2, ex, clk, 20*time.Minute) g := s2.ControllerSupervisorStatus(context.Background()).Guests[0] if g.Restarts24h != 5 || !g.SlowCrashloop { t.Fatalf("the agent restart reset the slow counter: %+v", g) } }