Files
felhom-agent/internal/localapi/controllersupervisor.go
T

360 lines
14 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package localapi
import (
"context"
"os"
"path/filepath"
"sort"
"strconv"
"strings"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
)
// R-523 — the in-guest controller supervisor.
//
// THE OUTAGE THIS EXISTS TO KILL (BIGNIGHT F9, 2026-09-14). `docker kill felhom-controller` left the
// container `Exited (137)`. Nothing restarted it: Docker never restarts a container whose stop it
// records as deliberate — measured 2026-09-15 on Docker 29.8.0 for BOTH `unless-stopped` and `always`
// (evidence-p1fixes-2026-09-15/A1) — and the golden's `felhom-controller-bootstrap.service` is a
// oneshot (`RemainAfterExit=yes`) that ran once at boot and watches nothing. The household's
// dashboard answered 502 for 33 minutes until the box was power-cycled.
//
// This is doc 03 §4's sentence made real: "Healing a crashed controller is non-destructive by
// construction … redeploy = restart … inside the existing guest — never a guest destroy." The act
// is exactly the swap's own restart (`systemctl restart felhom-controller-bootstrap.service`, which
// does `docker rm -f` + `docker run` from the baked image and the guest's persistent volume), over
// the same GuestExecutor and the same two sudoers grants (`docker inspect -f *`, the unit restart).
// No new privilege.
//
// THE GUARDS, each because doing the act at the wrong moment is worse than not doing it:
// - not during a swap (the swap stops the controller ON PURPOSE and owns its own rollback);
// - not when the operator parked it (`<guests>/<vmid>/controller-parked` on the HOST);
// - not on a guest that is not running, is locked (backup/restore/snapshot/migrate), or has a
// vzdump in flight — a stopping or restoring guest is someone else's transaction;
// - not on ONE observation: the container must be seen not-running on two consecutive sweeps, so
// the bootstrap's own rm-f/run window (boot, path-unit hot-plug) is never raced;
// - no thrash: 3 restarts inside 15 minutes → stop restarting, raise `controller_crashloop`, try
// again after 30 minutes.
//
// THE EVENTS. The agent has no event channel of its own; its heartbeat IS the channel (the
// capability/leaf precedent). The per-guest record rides the host report as `controller_supervisor`,
// and the hub's ControllerSupervisorChecker mints `controller_restarted_by_agent` (info) when a
// guest's `last_restart_at` moves and `controller_crashloop` (error, operator-only) when
// `crashloop_since` moves. Timestamps, not counters, so an agent restart (which zeroes the in-memory
// record) can never read as a new restart.
const (
// controllerSupervisorInterval is the sweep cadence. Two not-running observations are required,
// so a killed controller is restarted 30–60 s after it died.
controllerSupervisorInterval = 30 * time.Second
// controllerSupervisorConfirm is how many consecutive not-running observations license a restart.
controllerSupervisorConfirm = 2
// Backoff: controllerCrashloopMax restarts inside controllerCrashloopWindow → give up for
// controllerCrashloopPause.
controllerCrashloopMax = 3
controllerCrashloopWindow = 15 * time.Minute
controllerCrashloopPause = 30 * time.Minute
// controllerSupervisorHeartbeatEvery: a liveness line every 20 sweeps (10 minutes) — a silent
// watchdog is indistinguishable from a dead one (standing rule 3).
controllerSupervisorHeartbeatEvery = 20
// ControllerParkedMarker is the host-side file that parks a guest's controller. The operator
// creates it with `touch /var/lib/felhom-agent/guests/<vmid>/controller-parked` and removes it to
// unpark. Host-side on purpose: it needs no in-guest exec grant, it survives a guest rebuild of
// the controller container, and a customer inside the guest cannot park the supervisor.
ControllerParkedMarker = "controller-parked"
defaultGuestsStateDir = "/var/lib/felhom-agent/guests"
)
// controllerSupState is one guest's supervisor record. In-memory on purpose (the guest-power
// precedent): an agent restart forgets a crash-loop pause, which costs at most one more restart
// attempt, whereas persisting it could carry a stale "give up" across the restart that fixed it.
type controllerSupState struct {
notRunningSeen int
restarts []time.Time // restart times inside the crash-loop window (pruned)
restartsTotal int
lastRestartAt time.Time
lastReason string
crashloopSince time.Time // zero = not in a crash-loop pause
parked bool
}
type controllerSupervisor struct {
mu sync.Mutex
guests map[int]*controllerSupState
sweeps int
}
// WatchControllers runs the controller supervisor sweep until ctx is done. No-op when the guest list
// (staleLock) or the guest executor is not wired.
func (s *Server) WatchControllers(ctx context.Context) {
if s.staleLock == nil || s.guestExec == nil {
s.logger.Info("controller-supervisor: not wired (no guest list or no guest executor) — disabled")
return
}
s.logger.Info("controller-supervisor: started", "interval", controllerSupervisorInterval.String(),
"confirm_sweeps", controllerSupervisorConfirm, "crashloop_max", controllerCrashloopMax,
"crashloop_window", controllerCrashloopWindow.String(), "guests_dir", s.guestsStateDir())
t := time.NewTicker(controllerSupervisorInterval)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
s.ControllerSupervisorTick(ctx)
}
}
}
func (s *Server) guestsStateDir() string {
if s.guestsDir != "" {
return s.guestsDir
}
return defaultGuestsStateDir
}
// provisionedGuest reports whether the agent provisioned a controller into this guest: the
// `<guests>/<vmid>/bootstrap` directory exists. The directory itself, not bootstrap.json inside it —
// the directory is owned by the mapped guest root (0700), so the non-root agent can see the entry but
// not stat the file within.
func (s *Server) provisionedGuest(vmid int) bool {
fi, err := os.Stat(filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), "bootstrap"))
return err == nil && fi.IsDir()
}
func (s *Server) controllerParked(vmid int) bool {
_, err := os.Stat(filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), ControllerParkedMarker))
return err == nil
}
func (s *Server) supState(vmid int) *controllerSupState {
if s.ctrlSup.guests == nil {
s.ctrlSup.guests = map[int]*controllerSupState{}
}
st := s.ctrlSup.guests[vmid]
if st == nil {
st = &controllerSupState{}
s.ctrlSup.guests[vmid] = st
}
return st
}
// ControllerSupervisorTick performs one sweep. Exported so a test (and a live check) can drive one
// cycle without waiting on the ticker.
func (s *Server) ControllerSupervisorTick(ctx context.Context) {
if s.staleLock == nil || s.guestExec == nil {
return
}
guests, err := s.staleLock.Guests(ctx)
if err != nil {
// Ownership unproven ⇒ touch nothing (the guest-power rule).
s.logger.Warn("controller-supervisor: guest list unavailable — skipping sweep (ownership unproven)", "err", err)
return
}
var evaluated, down int
for _, g := range guests {
if ctx.Err() != nil {
return
}
if !s.provisionedGuest(g.VMID) {
continue
}
evaluated++
if !s.superviseOneController(ctx, g.VMID, g.Status) {
down++
}
}
s.ctrlSup.mu.Lock()
s.ctrlSup.sweeps++
sweeps := s.ctrlSup.sweeps
s.ctrlSup.mu.Unlock()
if sweeps%controllerSupervisorHeartbeatEvery == 0 {
s.logger.Info("controller-supervisor: alive", "sweeps_since_boot", sweeps,
"guests_evaluated", evaluated, "controllers_not_running", down)
}
}
// controllerRunning asks the guest's Docker for the controller's state. Returns (running, known).
// known=false means the question could not be answered (pct exec failed for a reason other than a
// missing container) — the caller does nothing on unknown. An ABSENT container is a known "not
// running": `docker rm` of the controller is the same outage as a kill.
func (s *Server) controllerRunning(ctx context.Context, vmid int) (running, known bool, status string) {
out, err := s.guestExec.GuestExec(ctx, vmid, "docker", "inspect", "-f", "{{.State.Status}}", controllerContainer)
if err != nil {
msg := strings.ToLower(err.Error() + " " + out)
if strings.Contains(msg, "no such object") || strings.Contains(msg, "no such container") {
return false, true, "absent"
}
return false, false, ""
}
status = strings.TrimSpace(out)
// "restarting" is Docker's own restart loop at work — not ours to fight on this sweep.
return status == "running" || status == "restarting", true, status
}
// superviseOneController evaluates one provisioned guest and restarts its controller when every guard
// allows. Returns false when the controller was observed not running.
func (s *Server) superviseOneController(ctx context.Context, vmid int, guestStatus string) bool {
now := s.clock()
if guestStatus != "running" {
s.resetNotRunning(vmid)
return true // the guest-power watchdog owns a stopped guest; its controller is not "down"
}
running, known, status := s.controllerRunning(ctx, vmid)
if !known {
s.logger.Debug("controller-supervisor: controller state unknown (guest exec failed) — no action", "vmid", vmid)
s.resetNotRunning(vmid)
return true
}
parked := s.controllerParked(vmid)
s.ctrlSup.mu.Lock()
st := s.supState(vmid)
st.parked = parked
if running {
st.notRunningSeen = 0
s.ctrlSup.mu.Unlock()
return true
}
st.notRunningSeen++
seen := st.notRunningSeen
s.ctrlSup.mu.Unlock()
if parked {
s.logger.Info("controller-supervisor: controller is not running and the guest is PARKED — leaving it",
"vmid", vmid, "status", status, "marker", filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), ControllerParkedMarker))
return false
}
s.swapMu.Lock()
swapping := s.swapInFlight[vmid]
s.swapMu.Unlock()
if swapping {
s.logger.Info("controller-supervisor: controller is not running during a controller SWAP — the swap owns it",
"vmid", vmid, "status", status)
s.resetNotRunning(vmid)
return false
}
if seen < controllerSupervisorConfirm {
s.logger.Info("controller-supervisor: controller observed not running — confirming on the next sweep",
"vmid", vmid, "status", status, "seen", seen, "of", controllerSupervisorConfirm)
return false
}
lock, _, err := s.staleLock.Lock(ctx, vmid)
if err != nil {
s.logger.Warn("controller-supervisor: could not read the guest lock — no action (fail-safe)", "vmid", vmid, "err", err)
return false
}
if lock != "" {
s.logger.Info("controller-supervisor: guest is LOCKED — another operation owns it, no action", "vmid", vmid, "lock", lock)
return false
}
if busy, berr := s.staleLock.BackupRunning(ctx, vmid); berr != nil || busy {
s.logger.Info("controller-supervisor: a vzdump may be in flight for the guest — no action",
"vmid", vmid, "backup_running", busy, "err", berr)
return false
}
// Backoff.
s.ctrlSup.mu.Lock()
st = s.supState(vmid)
if !st.crashloopSince.IsZero() {
if now.Sub(st.crashloopSince) < controllerCrashloopPause {
s.ctrlSup.mu.Unlock()
s.logger.Warn("controller-supervisor: crash-loop pause in force — not restarting",
"vmid", vmid, "since", st.crashloopSince.Format(time.RFC3339), "resume_after", controllerCrashloopPause.String())
return false
}
// Pause over: resume with a clean window. crashloopSince stays as the record of the last
// crash-loop (the hub keys on it moving, not on it clearing).
st.restarts = nil
st.crashloopSince = time.Time{}
}
st.restarts = pruneBefore(st.restarts, now.Add(-controllerCrashloopWindow))
if len(st.restarts) >= controllerCrashloopMax {
st.crashloopSince = now
n := len(st.restarts)
s.ctrlSup.mu.Unlock()
s.logger.Error("controller-supervisor: CRASH-LOOP — the controller would not stay up; stopping restarts and raising controller_crashloop",
"vmid", vmid, "restarts_in_window", n, "window", controllerCrashloopWindow.String(), "pause", controllerCrashloopPause.String())
return false
}
s.ctrlSup.mu.Unlock()
reason := "controller container " + status + " on " + strconv.Itoa(controllerSupervisorConfirm) + " consecutive sweeps"
s.logger.Warn("controller-supervisor: controller is NOT running — restarting the bootstrap unit",
"vmid", vmid, "status", status, "unit", bootstrapUnit)
if _, err := s.guestExec.GuestExec(ctx, vmid, "systemctl", "restart", bootstrapUnit); err != nil {
s.logger.Error("controller-supervisor: bootstrap restart failed", "vmid", vmid, "err", err)
reason += "; restart FAILED: " + err.Error()
}
s.ctrlSup.mu.Lock()
st = s.supState(vmid)
st.restarts = append(st.restarts, now)
st.restartsTotal++
st.lastRestartAt = now
st.lastReason = reason
st.notRunningSeen = 0
s.ctrlSup.mu.Unlock()
s.logger.Warn("controller-supervisor: RESTARTED the controller", "vmid", vmid, "reason", reason)
return false
}
func (s *Server) resetNotRunning(vmid int) {
s.ctrlSup.mu.Lock()
defer s.ctrlSup.mu.Unlock()
if st := s.ctrlSup.guests[vmid]; st != nil {
st.notRunningSeen = 0
}
}
func (s *Server) clock() time.Time {
if s.now != nil {
return s.now()
}
return time.Now().UTC()
}
func pruneBefore(ts []time.Time, cutoff time.Time) []time.Time {
out := ts[:0]
for _, t := range ts {
if !t.Before(cutoff) {
out = append(out, t)
}
}
return out
}
// ControllerSupervisorStatus is the host-report stanza source (hub.ControllerSupervisorReporter).
// Nil when the supervisor is not wired, so the stanza is omitted.
func (s *Server) ControllerSupervisorStatus(_ context.Context) *hub.ControllerSupervisorStatus {
if s.staleLock == nil || s.guestExec == nil {
return nil
}
s.ctrlSup.mu.Lock()
defer s.ctrlSup.mu.Unlock()
out := &hub.ControllerSupervisorStatus{Guests: []hub.ControllerSupervisorGuest{}}
for vmid, st := range s.ctrlSup.guests {
g := hub.ControllerSupervisorGuest{
VMID: vmid,
RestartsTotal: st.restartsTotal,
LastReason: st.lastReason,
Parked: st.parked,
Crashloop: !st.crashloopSince.IsZero(),
}
if !st.lastRestartAt.IsZero() {
g.LastRestartAt = st.lastRestartAt.UTC().Format(time.RFC3339)
}
if !st.crashloopSince.IsZero() {
g.CrashloopSince = st.crashloopSince.UTC().Format(time.RFC3339)
}
out.Guests = append(out.Guests, g)
}
sort.Slice(out.Guests, func(i, j int) bool { return out.Guests[i].VMID < out.Guests[j].VMID })
return out
}