v0.132.0: the slow crash-loop counter (R-539, operator ruling 3 of 2026-09-16)
gates / gates (push) Successful in 12s
gates / gates (push) Successful in 12s
Beside the unchanged 3-in-15-minutes brake, a second counter: restarts the
supervisor performed in the last 24 hours. At the fifth the heartbeat stanza
sets slow_crashloop_since (moving at most once per 24 h), slow_crashloop and
restarts_24h; hub v0.117.0 mints controller_slow_crashloop (warning,
operator-only) when the timestamp moves. It never stops restarting.
Persisted per guest (tmp+rename, 0600) so an agent restart or reboot does not
reset it - unlike the fast record, whose reason for staying in memory (a
persisted give-up outliving the fix) does not apply to a counter that only
warns. Deliberate kills count. The startup line prints the new limits.
Red-proofs seen failing: no counter; the once-per-24h guard removed ('the
operator would be mailed per restart'); the save removed ('Restarts24h:1'
after an agent restart). Negative control: restarts 7 h apart never raise it.
go build/vet/test ./... green, 30 packages.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -2,6 +2,7 @@ package localapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
@@ -68,6 +69,20 @@ const (
|
||||
ControllerParkedMarker = "controller-parked"
|
||||
|
||||
defaultGuestsStateDir = "/var/lib/felhom-agent/guests"
|
||||
|
||||
// R-539 (operator ruling 3 of 2026-09-16) — the SLOW crash loop. The 3-in-15-minutes brake above
|
||||
// cannot see a controller that dies every 20 minutes: no two restarts share its window, so it is
|
||||
// restarted for ever and the only trace is an info event that mails nobody (measured 2026-09-16,
|
||||
// R-531). A second counter over 24 hours raises a WARNING at the fifth restart. It does NOT stop
|
||||
// restarting — the fast brake stays the only brake, unchanged. Every restart the supervisor
|
||||
// performs counts, including one that follows a deliberate operator `docker kill` (measured
|
||||
// 2026-09-15: the supervisor cannot tell a kill from a crash, and a controller that is killed five
|
||||
// times a day is worth a line to the operator either way).
|
||||
controllerSlowCrashloopWindow = 24 * time.Hour
|
||||
controllerSlowCrashloopMax = 5
|
||||
// controllerSlowCounterFile holds the 24-hour restart times and the last raise, per guest, beside
|
||||
// the parked marker.
|
||||
controllerSlowCounterFile = "controller-restarts-24h.json"
|
||||
)
|
||||
|
||||
// controllerSupState is one guest's supervisor record. In-memory on purpose (the guest-power
|
||||
@@ -81,6 +96,13 @@ type controllerSupState struct {
|
||||
lastReason string
|
||||
crashloopSince time.Time // zero = not in a crash-loop pause
|
||||
parked bool
|
||||
|
||||
// R-539 — the slow counter. PERSISTED, unlike everything above, and the precedent's reason does not
|
||||
// apply to it: persisting the fast record could carry a stale "give up" across the restart that
|
||||
// fixed it, but this record never gives anything up — it only warns. Losing it on an agent restart,
|
||||
// on the other hand, would hide exactly the box it exists for (one whose agent restarts too).
|
||||
restarts24h []time.Time
|
||||
slowCrashloopSince time.Time // the last raise; kept after it ages out, the hub keys on it MOVING
|
||||
}
|
||||
|
||||
type controllerSupervisor struct {
|
||||
@@ -98,7 +120,9 @@ func (s *Server) WatchControllers(ctx context.Context) {
|
||||
}
|
||||
s.logger.Info("controller-supervisor: started", "interval", controllerSupervisorInterval.String(),
|
||||
"confirm_sweeps", controllerSupervisorConfirm, "crashloop_max", controllerCrashloopMax,
|
||||
"crashloop_window", controllerCrashloopWindow.String(), "guests_dir", s.guestsStateDir())
|
||||
"crashloop_window", controllerCrashloopWindow.String(),
|
||||
"slow_crashloop_max", controllerSlowCrashloopMax, "slow_crashloop_window", controllerSlowCrashloopWindow.String(),
|
||||
"guests_dir", s.guestsStateDir())
|
||||
t := time.NewTicker(controllerSupervisorInterval)
|
||||
defer t.Stop()
|
||||
for {
|
||||
@@ -139,11 +163,61 @@ func (s *Server) supState(vmid int) *controllerSupState {
|
||||
st := s.ctrlSup.guests[vmid]
|
||||
if st == nil {
|
||||
st = &controllerSupState{}
|
||||
s.loadSlowCounter(vmid, st)
|
||||
s.ctrlSup.guests[vmid] = st
|
||||
}
|
||||
return st
|
||||
}
|
||||
|
||||
// slowCounterRecord is the on-disk shape of the R-539 counter.
|
||||
type slowCounterRecord struct {
|
||||
Restarts []time.Time `json:"restarts"`
|
||||
SlowCrashloopSince time.Time `json:"slow_crashloop_since,omitempty"`
|
||||
}
|
||||
|
||||
func (s *Server) slowCounterPath(vmid int) string {
|
||||
return filepath.Join(s.guestsStateDir(), strconv.Itoa(vmid), controllerSlowCounterFile)
|
||||
}
|
||||
|
||||
// loadSlowCounter restores the persisted counter into a fresh state. Absent = a clean start; unreadable
|
||||
// or corrupt = a clean start with a WARN (a warning counter must never block supervision).
|
||||
func (s *Server) loadSlowCounter(vmid int, st *controllerSupState) {
|
||||
b, err := os.ReadFile(s.slowCounterPath(vmid))
|
||||
if err != nil {
|
||||
if !os.IsNotExist(err) {
|
||||
s.logger.Warn("controller-supervisor: slow counter unreadable — starting it from zero", "vmid", vmid, "err", err)
|
||||
}
|
||||
return
|
||||
}
|
||||
var rec slowCounterRecord
|
||||
if err := json.Unmarshal(b, &rec); err != nil {
|
||||
s.logger.Warn("controller-supervisor: slow counter corrupt — starting it from zero", "vmid", vmid, "err", err)
|
||||
return
|
||||
}
|
||||
st.restarts24h = pruneBefore(rec.Restarts, s.clock().Add(-controllerSlowCrashloopWindow))
|
||||
st.slowCrashloopSince = rec.SlowCrashloopSince
|
||||
if len(st.restarts24h) > 0 || !st.slowCrashloopSince.IsZero() {
|
||||
s.logger.Info("controller-supervisor: slow counter restored from disk", "vmid", vmid,
|
||||
"restarts_24h", len(st.restarts24h), "slow_crashloop_since", st.slowCrashloopSince.Format(time.RFC3339))
|
||||
}
|
||||
}
|
||||
|
||||
// saveSlowCounter writes the counter atomically (tmp + rename, 0600). A failure is logged and the
|
||||
// in-memory counter carries on — the next restart retries the write.
|
||||
func (s *Server) saveSlowCounter(vmid int, rec slowCounterRecord) {
|
||||
path := s.slowCounterPath(vmid)
|
||||
b, err := json.Marshal(rec)
|
||||
if err == nil {
|
||||
tmp := path + ".tmp"
|
||||
if err = os.WriteFile(tmp, b, 0o600); err == nil {
|
||||
err = os.Rename(tmp, path)
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
s.logger.Warn("controller-supervisor: could not persist the slow counter (kept in memory)", "vmid", vmid, "path", path, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// ControllerSupervisorTick performs one sweep. Exported so a test (and a live check) can drive one
|
||||
// cycle without waiting on the ticker.
|
||||
func (s *Server) ControllerSupervisorTick(ctx context.Context) {
|
||||
@@ -299,8 +373,22 @@ func (s *Server) superviseOneController(ctx context.Context, vmid int, guestStat
|
||||
st.lastRestartAt = now
|
||||
st.lastReason = reason
|
||||
st.notRunningSeen = 0
|
||||
// R-539: the slow counter. Raise at most once per 24 hours — the hub mails on the raise MOVING.
|
||||
st.restarts24h = append(pruneBefore(st.restarts24h, now.Add(-controllerSlowCrashloopWindow)), now)
|
||||
n24 := len(st.restarts24h)
|
||||
raised := false
|
||||
if n24 >= controllerSlowCrashloopMax && (st.slowCrashloopSince.IsZero() || now.Sub(st.slowCrashloopSince) >= controllerSlowCrashloopWindow) {
|
||||
st.slowCrashloopSince = now
|
||||
raised = true
|
||||
}
|
||||
rec := slowCounterRecord{Restarts: append([]time.Time(nil), st.restarts24h...), SlowCrashloopSince: st.slowCrashloopSince}
|
||||
s.ctrlSup.mu.Unlock()
|
||||
s.logger.Warn("controller-supervisor: RESTARTED the controller", "vmid", vmid, "reason", reason)
|
||||
s.saveSlowCounter(vmid, rec)
|
||||
s.logger.Warn("controller-supervisor: RESTARTED the controller", "vmid", vmid, "reason", reason, "restarts_24h", n24)
|
||||
if raised {
|
||||
s.logger.Warn("controller-supervisor: SLOW CRASH-LOOP — the controller keeps dying; still restarting it, raising controller_slow_crashloop",
|
||||
"vmid", vmid, "restarts_24h", n24, "window", controllerSlowCrashloopWindow.String(), "threshold", controllerSlowCrashloopMax)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -352,6 +440,12 @@ func (s *Server) ControllerSupervisorStatus(_ context.Context) *hub.ControllerSu
|
||||
if !st.crashloopSince.IsZero() {
|
||||
g.CrashloopSince = st.crashloopSince.UTC().Format(time.RFC3339)
|
||||
}
|
||||
now := s.clock()
|
||||
g.Restarts24h = len(pruneBefore(append([]time.Time(nil), st.restarts24h...), now.Add(-controllerSlowCrashloopWindow)))
|
||||
if !st.slowCrashloopSince.IsZero() {
|
||||
g.SlowCrashloopSince = st.slowCrashloopSince.UTC().Format(time.RFC3339)
|
||||
g.SlowCrashloop = now.Sub(st.slowCrashloopSince) < controllerSlowCrashloopWindow
|
||||
}
|
||||
out.Guests = append(out.Guests, g)
|
||||
}
|
||||
sort.Slice(out.Guests, func(i, j int) bool { return out.Guests[i].VMID < out.Guests[j].VMID })
|
||||
|
||||
Reference in New Issue
Block a user