controller v0.269.0: whole restore from the second drive; crash loops stopped; exact image digests; steps judged by their own .felhom.yml (decisions 26-28, R-661 R-666 R-667 R-668 R-664 R-665 R-662, 09 6.4 part 6)
gates / gates (push) Successful in 27s
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,190 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ── A crash loop and an out-of-memory storm are stopped by the box (`09` §3 decision 28, R-667) ────
|
||||
//
|
||||
// WHY THE OLD DETECTOR MISSED IT. Stack.CrashLooping asked for crashLoopAfter (5 min) of UNINTERRUPTED
|
||||
// `restarting`, measured by RestartingSince — a clock the status pass resets whenever it catches the
|
||||
// container `running` between two crashes. Measured on 9202 2026-09-24: gokapi at 385 → 546 restarts,
|
||||
// `restarting_since` a minute old at every look, „0 currently down".
|
||||
//
|
||||
// SO THIS COUNTS WHAT DOCKER COUNTS: each container's RestartCount, summed per app, sampled every scan.
|
||||
// RestartCount only grows for one container run and resets when compose RECREATES the container — a
|
||||
// drop is a reset and starts the window again, never a negative count.
|
||||
//
|
||||
// THE THRESHOLDS, from evidence (`audits/night-2026-09-24/A3/`):
|
||||
// - crash loop: >= CrashLoopRestarts (6) restarts within CrashLoopWindow (10 min). Docker's restart
|
||||
// back-off caps a steady crash loop at about ONE restart a minute (gokapi: 539 → 546 in 7 min), so
|
||||
// the brief's "10 in 10 minutes" sits on the edge and can miss a steady loop. Across all 40
|
||||
// containers of both demo boxes and 9202 no healthy container had restarted more than ONCE.
|
||||
// - out-of-memory storm: the existing app_oom_storm rule — >= 20 kernel OOM kills within 30 min.
|
||||
|
||||
const (
|
||||
CrashLoopRestarts = 6
|
||||
CrashLoopWindow = 10 * time.Minute
|
||||
OOMStormKills = 20
|
||||
OOMStormWindow = 30 * time.Minute
|
||||
|
||||
UnhealthyCrashLoop = "crash_loop"
|
||||
UnhealthyOOMStorm = "oom_storm"
|
||||
)
|
||||
|
||||
type unhealthySample struct {
|
||||
at time.Time
|
||||
restarts int64
|
||||
kills int64 // -1: unknown
|
||||
}
|
||||
|
||||
// UnhealthyVerdict is one app that crossed a threshold in the latest observation.
|
||||
type UnhealthyVerdict struct {
|
||||
Stack string
|
||||
Kind string // UnhealthyCrashLoop / UnhealthyOOMStorm
|
||||
Count int64 // restarts or kills inside the window
|
||||
Window time.Duration
|
||||
}
|
||||
|
||||
type unhealthyWatch struct {
|
||||
mu sync.Mutex
|
||||
samples map[string][]unhealthySample
|
||||
}
|
||||
|
||||
// judgeUnhealthy is the pure verdict over one app's samples (oldest first, the last one = now).
|
||||
func judgeUnhealthy(samples []unhealthySample) (kind string, count int64, window time.Duration) {
|
||||
if len(samples) < 2 {
|
||||
return "", 0, 0
|
||||
}
|
||||
last := samples[len(samples)-1]
|
||||
// A drop in either counter is a reset (the container was recreated): only samples after it count.
|
||||
start := 0
|
||||
for i := 1; i < len(samples); i++ {
|
||||
if samples[i].restarts < samples[i-1].restarts {
|
||||
start = i
|
||||
}
|
||||
}
|
||||
for i := start; i < len(samples); i++ {
|
||||
if last.at.Sub(samples[i].at) <= CrashLoopWindow {
|
||||
if d := last.restarts - samples[i].restarts; d >= CrashLoopRestarts {
|
||||
return UnhealthyCrashLoop, d, CrashLoopWindow
|
||||
}
|
||||
break
|
||||
}
|
||||
}
|
||||
if last.kills >= 0 {
|
||||
kstart := 0
|
||||
for i := 1; i < len(samples); i++ {
|
||||
if samples[i].kills >= 0 && samples[i-1].kills >= 0 && samples[i].kills < samples[i-1].kills {
|
||||
kstart = i
|
||||
}
|
||||
}
|
||||
for i := kstart; i < len(samples); i++ {
|
||||
if samples[i].kills < 0 {
|
||||
continue
|
||||
}
|
||||
if last.at.Sub(samples[i].at) <= OOMStormWindow {
|
||||
if d := last.kills - samples[i].kills; d >= OOMStormKills {
|
||||
return UnhealthyOOMStorm, d, OOMStormWindow
|
||||
}
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
return "", 0, 0
|
||||
}
|
||||
|
||||
// ObserveUnhealthy takes one sample per deployed app — RestartCount summed over its containers (one
|
||||
// `docker inspect` for all of them) and the kernel OOM kills from the latest ScanOOMKilled — and returns
|
||||
// the apps whose window crossed a threshold. An app that is deploying, updating or already held is not
|
||||
// sampled and its history is dropped (a deploy or an update recreates containers on purpose).
|
||||
func (m *Manager) ObserveUnhealthy(now time.Time, ooms []OOMContainer) []UnhealthyVerdict {
|
||||
type appC struct{ names []string }
|
||||
m.mu.RLock()
|
||||
apps := map[string]*appC{}
|
||||
owner := map[string]string{}
|
||||
var all []string
|
||||
skip := map[string]bool{}
|
||||
for name, st := range m.stacks {
|
||||
if !st.Deployed || st.Protected {
|
||||
continue
|
||||
}
|
||||
if st.Deploying || st.Updating || st.HoldReason != "" || st.updateHeld {
|
||||
skip[name] = true
|
||||
continue
|
||||
}
|
||||
a := &appC{}
|
||||
for _, c := range st.Containers {
|
||||
a.names = append(a.names, c.Name)
|
||||
owner[c.Name] = name
|
||||
all = append(all, c.Name)
|
||||
}
|
||||
apps[name] = a
|
||||
}
|
||||
m.mu.RUnlock()
|
||||
|
||||
restarts := map[string]int64{}
|
||||
if len(all) > 0 {
|
||||
sort.Strings(all)
|
||||
args := append([]string{"inspect", "-f", "{{.Name}}|{{.RestartCount}}"}, all...)
|
||||
out, _ := m.execCommand("docker", args...) // a vanished container fails its own line only
|
||||
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
|
||||
f := strings.SplitN(strings.TrimSpace(line), "|", 2)
|
||||
if len(f) != 2 {
|
||||
continue
|
||||
}
|
||||
n, err := strconv.ParseInt(strings.TrimSpace(f[1]), 10, 64)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if app, ok := owner[strings.TrimPrefix(f[0], "/")]; ok {
|
||||
restarts[app] += n
|
||||
}
|
||||
}
|
||||
}
|
||||
kills := map[string]int64{}
|
||||
for _, o := range ooms {
|
||||
if o.Kills >= 0 {
|
||||
kills[o.Stack] += o.Kills
|
||||
}
|
||||
}
|
||||
|
||||
m.unhealthy.mu.Lock()
|
||||
defer m.unhealthy.mu.Unlock()
|
||||
if m.unhealthy.samples == nil {
|
||||
m.unhealthy.samples = map[string][]unhealthySample{}
|
||||
}
|
||||
for name := range m.unhealthy.samples {
|
||||
if _, live := apps[name]; !live {
|
||||
delete(m.unhealthy.samples, name) // removed, deploying, updating or held: history dropped
|
||||
}
|
||||
}
|
||||
var out []UnhealthyVerdict
|
||||
for name := range apps {
|
||||
if skip[name] {
|
||||
continue
|
||||
}
|
||||
k := int64(-1)
|
||||
if v, ok := kills[name]; ok {
|
||||
k = v
|
||||
} else {
|
||||
k = 0
|
||||
}
|
||||
ss := append(m.unhealthy.samples[name], unhealthySample{at: now, restarts: restarts[name], kills: k})
|
||||
// keep 35 minutes of history — the longer window plus a margin
|
||||
for len(ss) > 1 && now.Sub(ss[0].at) > OOMStormWindow+5*time.Minute {
|
||||
ss = ss[1:]
|
||||
}
|
||||
m.unhealthy.samples[name] = ss
|
||||
if kind, n, win := judgeUnhealthy(ss); kind != "" {
|
||||
out = append(out, UnhealthyVerdict{Stack: name, Kind: kind, Count: n, Window: win})
|
||||
delete(m.unhealthy.samples, name) // one verdict per episode
|
||||
}
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].Stack < out[j].Stack })
|
||||
return out
|
||||
}
|
||||
Reference in New Issue
Block a user