Files
admin 0054d4bd69
gates / gates (push) Successful in 27s
controller v0.265.0: R-634 cause fixed, held apps say so, OOM storm alarm, R-647 leftovers
R-634: a whole-box backup no longer stops/restarts a DEPLOYING app (the
measured cause of containers running under 'not deployed'); StopStack
and StartStack refuse a deploying stack for every caller.
R-625: held badge 'Stopped - restore needed', no Update button.
R-636: kernel oom_kill counter; 20+ in 30 min -> one app_oom_storm.
R-647: held error per reader, copy_holds key, two log wordings.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-23 17:16:57 +02:00

135 lines
4.7 KiB
Go

package stacks
import (
"fmt"
"sort"
"strconv"
"strings"
)
// R-514 (v0.243.0) — an OOM-killed worker inside a RUNNING container is invisible to the state model.
//
// BIGNIGHT: Paperless's celery worker was killed by the memory cgroup inside the webserver container;
// the container kept running (`docker inspect` → OOMKilled=true, RestartCount=0), the app read „Fut",
// 11 documents failed and 8 waited forever. Docker's State.OOMKilled is set when ANY process in the
// container's cgroup was OOM-killed and stays set until the container restarts — so it is the
// observable, read for the running containers of deployed stacks.
// OOMContainer is one container whose cgroup had a process OOM-killed since it started.
type OOMContainer struct {
Stack string
Container string
StartedAt string // identifies the container run — one event per run
// Kills is the kernel's own count of OOM kills in this container's cgroup (`memory.events`
// oom_kill) — R-636, v0.265.0. OOMKilled is a STICKY flag: it stays true for the container's whole
// life after ONE kill, so "the key re-fired" means nothing; this counter is what separates one
// hiccup from RomM's 4,530 kills in six hours. -1 when it could not be read.
Kills int64
// MemLimit / Peak are the cgroup's memory.max and memory.peak, as "<n>M" (or "max"); "" unread.
MemLimit, Peak string
}
// ScanOOMKilled inspects the containers of every deployed, running-ish stack in ONE docker call and
// returns those with State.OOMKilled=true. It also caches the per-stack result for the dashboard.
func (m *Manager) ScanOOMKilled() ([]OOMContainer, error) {
m.mu.RLock()
owner := map[string]string{}
var names []string
for name, st := range m.stacks {
if !st.Deployed {
continue
}
for _, c := range st.Containers {
if c.State == StateRunning || c.State == StateUnhealthy || c.State == StateRestarting {
owner[c.Name] = name
names = append(names, c.Name)
}
}
}
m.mu.RUnlock()
if len(names) == 0 {
m.setOOMCache(nil)
return nil, nil
}
sort.Strings(names)
args := append([]string{"inspect", "-f", "{{.Name}}|{{.State.OOMKilled}}|{{.State.StartedAt}}"}, names...)
out, err := m.execCommand("docker", args...)
if err != nil && strings.TrimSpace(out) == "" {
return nil, err
}
var found []OOMContainer
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
f := strings.SplitN(strings.TrimSpace(line), "|", 3)
if len(f) != 3 || f[1] != "true" {
continue
}
cname := strings.TrimPrefix(f[0], "/")
if st, ok := owner[cname]; ok {
found = append(found, OOMContainer{Stack: st, Container: cname, StartedAt: f[2], Kills: -1})
}
}
// R-636: only for the containers already flagged — one `docker exec` each, and the flagged set is
// empty on a healthy box. Read from INSIDE the container (its cgroup namespace makes
// /sys/fs/cgroup its own cgroup); the controller's own namespace cannot see the others.
for i := range found {
out, _ := m.execCommand("docker", "exec", found[i].Container, "cat",
"/sys/fs/cgroup/memory.events", "/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory.peak")
found[i].Kills, found[i].MemLimit, found[i].Peak = parseCgroupMemory(out)
}
m.setOOMCache(found)
return found, nil
}
// parseCgroupMemory reads `cat memory.events memory.max memory.peak`: the key/value lines give
// oom_kill; the first bare line is memory.max, the second memory.peak (absent on older kernels — cat
// then fails on that file and still prints the others). Unreadable → Kills -1.
func parseCgroupMemory(out string) (kills int64, limit, peak string) {
kills = -1
var bare []string
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
f := strings.Fields(line)
switch {
case len(f) == 2 && f[0] == "oom_kill":
if n, err := strconv.ParseInt(f[1], 10, 64); err == nil {
kills = n
}
case len(f) == 1:
bare = append(bare, f[0])
}
}
mb := func(v string) string {
if n, err := strconv.ParseInt(v, 10, 64); err == nil {
return fmt.Sprintf("%dM", n/(1024*1024))
}
return v // "max"
}
if len(bare) > 0 {
limit = mb(bare[0])
}
if len(bare) > 1 {
peak = mb(bare[1])
}
return kills, limit, peak
}
func (m *Manager) setOOMCache(found []OOMContainer) {
cache := map[string][]string{}
for _, o := range found {
cache[o.Stack] = append(cache[o.Stack], o.Container)
}
m.oomMu.Lock()
m.oomCache = cache
m.oomMu.Unlock()
}
// OOMKilledStacks returns stack name → OOM-killed container names from the last scan.
func (m *Manager) OOMKilledStacks() map[string][]string {
m.oomMu.Lock()
defer m.oomMu.Unlock()
out := make(map[string][]string, len(m.oomCache))
for k, v := range m.oomCache {
out[k] = append([]string(nil), v...)
}
return out
}