package localapi import ( "context" "os" "path/filepath" "sort" "strconv" "strings" "sync" "time" "gitea.dooplex.hu/admin/felhom-agent/internal/hub" ) // R-523 — the in-guest controller supervisor. // // THE OUTAGE THIS EXISTS TO KILL (BIGNIGHT F9, 2026-09-14). `docker kill felhom-controller` left the // container `Exited (137)`. Nothing restarted it: Docker never restarts a container whose stop it // records as deliberate — measured 2026-09-15 on Docker 29.8.0 for BOTH `unless-stopped` and `always` // (evidence-p1fixes-2026-09-15/A1) — and the golden's `felhom-controller-bootstrap.service` is a // oneshot (`RemainAfterExit=yes`) that ran once at boot and watches nothing. The household's // dashboard answered 502 for 33 minutes until the box was power-cycled. // // This is doc 03 §4's sentence made real: "Healing a crashed controller is non-destructive by // construction … redeploy = restart … inside the existing guest — never a guest destroy." The act // is exactly the swap's own restart (`systemctl restart felhom-controller-bootstrap.service`, which // does `docker rm -f` + `docker run` from the baked image and the guest's persistent volume), over // the same GuestExecutor and the same two sudoers grants (`docker inspect -f *`, the unit restart). // No new privilege. // // THE GUARDS, each because doing the act at the wrong moment is worse than not doing it: // - not during a swap (the swap stops the controller ON PURPOSE and owns its own rollback); // - not when the operator parked it (`//controller-parked` on the HOST); // - not on a guest that is not running, is locked (backup/restore/snapshot/migrate), or has a // vzdump in flight — a stopping or restoring guest is someone else's transaction; // - not on ONE observation: the container must be seen not-running on two consecutive sweeps, so // the bootstrap's own rm-f/run window (boot, path-unit hot-plug) is never raced; // - no thrash: 3 restarts inside 15 minutes → stop restarting, raise `controller_crashloop`, try // again after 30 minutes. // // THE EVENTS. The agent has no event channel of its own; its heartbeat IS the channel (the // capability/leaf precedent). The per-guest record rides the host report as `controller_supervisor`, // and the hub's ControllerSupervisorChecker mints `controller_restarted_by_agent` (info) when a // guest's `last_restart_at` moves and `controller_crashloop` (error, operator-only) when // `crashloop_since` moves. Timestamps, not counters, so an agent restart (which zeroes the in-memory // record) can never read as a new restart. const ( // controllerSupervisorInterval is the sweep cadence. Two not-running observations are required, // so a killed controller is restarted 30–60 s after it died. controllerSupervisorInterval = 30 * time.Second // controllerSupervisorConfirm is how many consecutive not-running observations license a restart. controllerSupervisorConfirm = 2 // Backoff: controllerCrashloopMax restarts inside controllerCrashloopWindow → give up for // controllerCrashloopPause. controllerCrashloopMax = 3 controllerCrashloopWindow = 15 * time.Minute controllerCrashloopPause = 30 * time.Minute // controllerSupervisorHeartbeatEvery: a liveness line every 20 sweeps (10 minutes) — a silent // watchdog is indistinguishable from a dead one (standing rule 3). controllerSupervisorHeartbeatEvery = 20 // ControllerParkedMarker is the host-side file that parks a guest's controller. The operator // creates it with `touch /var/lib/felhom-agent/guests//controller-parked` and removes it to // unpark. Host-side on purpose: it needs no in-guest exec grant, it survives a guest rebuild of // the controller container, and a customer inside the guest cannot park the supervisor. ControllerParkedMarker = "controller-parked" defaultGuestsStateDir = "/var/lib/felhom-agent/guests" ) // controllerSupState is one guest's supervisor record. In-memory on purpose (the guest-power // precedent): an agent restart forgets a crash-loop pause, which costs at most one more restart // attempt, whereas persisting it could carry a stale "give up" across the restart that fixed it. type controllerSupState struct { notRunningSeen int restarts []time.Time // restart times inside the crash-loop window (pruned) restartsTotal int lastRestartAt time.Time lastReason string crashloopSince time.Time // zero = not in a crash-loop pause parked bool } type controllerSupervisor struct { mu sync.Mutex guests map[int]*controllerSupState sweeps int } // WatchControllers runs the controller supervisor sweep until ctx is done. No-op when the guest list // (staleLock) or the guest executor is not wired. func (s *Server) WatchControllers(ctx context.Context) { if s.staleLock == nil || s.guestExec == nil { s.logger.Info("controller-supervisor: not wired (no guest list or no guest executor) — disabled") return } s.logger.Info("controller-supervisor: started", "interval", controllerSupervisorInterval.String(), "confirm_sweeps", controllerSupervisorConfirm, "crashloop_max", controllerCrashloopMax, "crashloop_window", controllerCrashloopWindow.String(), "guests_dir", s.guestsStateDir()) t := time.NewTicker(controllerSupervisorInterval) defer t.Stop() for { select { case <-ctx.Done(): return case <-t.C: s.ControllerSupervisorTick(ctx) } } } func (s *Server) guestsStateDir() string { if s.guestsDir != "" { return s.guestsDir } return defaultGuestsStateDir } // provisionedGuest reports whether the agent provisioned a controller into this guest: the // `//bootstrap` directory exists. The directory itself, not bootstrap.json inside it — // the directory is owned by the mapped guest root (0700), so the non-root agent can see the entry but // not stat the file within. func (s *Server) provisionedGuest(vmid int) bool { fi, err := os.Stat(filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), "bootstrap")) return err == nil && fi.IsDir() } func (s *Server) controllerParked(vmid int) bool { _, err := os.Stat(filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), ControllerParkedMarker)) return err == nil } func (s *Server) supState(vmid int) *controllerSupState { if s.ctrlSup.guests == nil { s.ctrlSup.guests = map[int]*controllerSupState{} } st := s.ctrlSup.guests[vmid] if st == nil { st = &controllerSupState{} s.ctrlSup.guests[vmid] = st } return st } // ControllerSupervisorTick performs one sweep. Exported so a test (and a live check) can drive one // cycle without waiting on the ticker. func (s *Server) ControllerSupervisorTick(ctx context.Context) { if s.staleLock == nil || s.guestExec == nil { return } guests, err := s.staleLock.Guests(ctx) if err != nil { // Ownership unproven ⇒ touch nothing (the guest-power rule). s.logger.Warn("controller-supervisor: guest list unavailable — skipping sweep (ownership unproven)", "err", err) return } var evaluated, down int for _, g := range guests { if ctx.Err() != nil { return } if !s.provisionedGuest(g.VMID) { continue } evaluated++ if !s.superviseOneController(ctx, g.VMID, g.Status) { down++ } } s.ctrlSup.mu.Lock() s.ctrlSup.sweeps++ sweeps := s.ctrlSup.sweeps s.ctrlSup.mu.Unlock() if sweeps%controllerSupervisorHeartbeatEvery == 0 { s.logger.Info("controller-supervisor: alive", "sweeps_since_boot", sweeps, "guests_evaluated", evaluated, "controllers_not_running", down) } } // controllerRunning asks the guest's Docker for the controller's state. Returns (running, known). // known=false means the question could not be answered (pct exec failed for a reason other than a // missing container) — the caller does nothing on unknown. An ABSENT container is a known "not // running": `docker rm` of the controller is the same outage as a kill. func (s *Server) controllerRunning(ctx context.Context, vmid int) (running, known bool, status string) { out, err := s.guestExec.GuestExec(ctx, vmid, "docker", "inspect", "-f", "{{.State.Status}}", controllerContainer) if err != nil { msg := strings.ToLower(err.Error() + " " + out) if strings.Contains(msg, "no such object") || strings.Contains(msg, "no such container") { return false, true, "absent" } return false, false, "" } status = strings.TrimSpace(out) // "restarting" is Docker's own restart loop at work — not ours to fight on this sweep. return status == "running" || status == "restarting", true, status } // superviseOneController evaluates one provisioned guest and restarts its controller when every guard // allows. Returns false when the controller was observed not running. func (s *Server) superviseOneController(ctx context.Context, vmid int, guestStatus string) bool { now := s.clock() if guestStatus != "running" { s.resetNotRunning(vmid) return true // the guest-power watchdog owns a stopped guest; its controller is not "down" } running, known, status := s.controllerRunning(ctx, vmid) if !known { s.logger.Debug("controller-supervisor: controller state unknown (guest exec failed) — no action", "vmid", vmid) s.resetNotRunning(vmid) return true } parked := s.controllerParked(vmid) s.ctrlSup.mu.Lock() st := s.supState(vmid) st.parked = parked if running { st.notRunningSeen = 0 s.ctrlSup.mu.Unlock() return true } st.notRunningSeen++ seen := st.notRunningSeen s.ctrlSup.mu.Unlock() if parked { s.logger.Info("controller-supervisor: controller is not running and the guest is PARKED — leaving it", "vmid", vmid, "status", status, "marker", filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), ControllerParkedMarker)) return false } s.swapMu.Lock() swapping := s.swapInFlight[vmid] s.swapMu.Unlock() if swapping { s.logger.Info("controller-supervisor: controller is not running during a controller SWAP — the swap owns it", "vmid", vmid, "status", status) s.resetNotRunning(vmid) return false } if seen < controllerSupervisorConfirm { s.logger.Info("controller-supervisor: controller observed not running — confirming on the next sweep", "vmid", vmid, "status", status, "seen", seen, "of", controllerSupervisorConfirm) return false } lock, _, err := s.staleLock.Lock(ctx, vmid) if err != nil { s.logger.Warn("controller-supervisor: could not read the guest lock — no action (fail-safe)", "vmid", vmid, "err", err) return false } if lock != "" { s.logger.Info("controller-supervisor: guest is LOCKED — another operation owns it, no action", "vmid", vmid, "lock", lock) return false } if busy, berr := s.staleLock.BackupRunning(ctx, vmid); berr != nil || busy { s.logger.Info("controller-supervisor: a vzdump may be in flight for the guest — no action", "vmid", vmid, "backup_running", busy, "err", berr) return false } // Backoff. s.ctrlSup.mu.Lock() st = s.supState(vmid) if !st.crashloopSince.IsZero() { if now.Sub(st.crashloopSince) < controllerCrashloopPause { s.ctrlSup.mu.Unlock() s.logger.Warn("controller-supervisor: crash-loop pause in force — not restarting", "vmid", vmid, "since", st.crashloopSince.Format(time.RFC3339), "resume_after", controllerCrashloopPause.String()) return false } // Pause over: resume with a clean window. crashloopSince stays as the record of the last // crash-loop (the hub keys on it moving, not on it clearing). st.restarts = nil st.crashloopSince = time.Time{} } st.restarts = pruneBefore(st.restarts, now.Add(-controllerCrashloopWindow)) if len(st.restarts) >= controllerCrashloopMax { st.crashloopSince = now n := len(st.restarts) s.ctrlSup.mu.Unlock() s.logger.Error("controller-supervisor: CRASH-LOOP — the controller would not stay up; stopping restarts and raising controller_crashloop", "vmid", vmid, "restarts_in_window", n, "window", controllerCrashloopWindow.String(), "pause", controllerCrashloopPause.String()) return false } s.ctrlSup.mu.Unlock() reason := "controller container " + status + " on " + strconv.Itoa(controllerSupervisorConfirm) + " consecutive sweeps" s.logger.Warn("controller-supervisor: controller is NOT running — restarting the bootstrap unit", "vmid", vmid, "status", status, "unit", bootstrapUnit) if _, err := s.guestExec.GuestExec(ctx, vmid, "systemctl", "restart", bootstrapUnit); err != nil { s.logger.Error("controller-supervisor: bootstrap restart failed", "vmid", vmid, "err", err) reason += "; restart FAILED: " + err.Error() } s.ctrlSup.mu.Lock() st = s.supState(vmid) st.restarts = append(st.restarts, now) st.restartsTotal++ st.lastRestartAt = now st.lastReason = reason st.notRunningSeen = 0 s.ctrlSup.mu.Unlock() s.logger.Warn("controller-supervisor: RESTARTED the controller", "vmid", vmid, "reason", reason) return false } func (s *Server) resetNotRunning(vmid int) { s.ctrlSup.mu.Lock() defer s.ctrlSup.mu.Unlock() if st := s.ctrlSup.guests[vmid]; st != nil { st.notRunningSeen = 0 } } func (s *Server) clock() time.Time { if s.now != nil { return s.now() } return time.Now().UTC() } func pruneBefore(ts []time.Time, cutoff time.Time) []time.Time { out := ts[:0] for _, t := range ts { if !t.Before(cutoff) { out = append(out, t) } } return out } // ControllerSupervisorStatus is the host-report stanza source (hub.ControllerSupervisorReporter). // Nil when the supervisor is not wired, so the stanza is omitted. func (s *Server) ControllerSupervisorStatus(_ context.Context) *hub.ControllerSupervisorStatus { if s.staleLock == nil || s.guestExec == nil { return nil } s.ctrlSup.mu.Lock() defer s.ctrlSup.mu.Unlock() out := &hub.ControllerSupervisorStatus{Guests: []hub.ControllerSupervisorGuest{}} for vmid, st := range s.ctrlSup.guests { g := hub.ControllerSupervisorGuest{ VMID: vmid, RestartsTotal: st.restartsTotal, LastReason: st.lastReason, Parked: st.parked, Crashloop: !st.crashloopSince.IsZero(), } if !st.lastRestartAt.IsZero() { g.LastRestartAt = st.lastRestartAt.UTC().Format(time.RFC3339) } if !st.crashloopSince.IsZero() { g.CrashloopSince = st.crashloopSince.UTC().Format(time.RFC3339) } out.Guests = append(out.Guests, g) } sort.Slice(out.Guests, func(i, j int) bool { return out.Guests[i].VMID < out.Guests[j].VMID }) return out }