controller v0.265.0: R-634 cause fixed, held apps say so, OOM storm alarm, R-647 leftovers
gates / gates (push) Successful in 27s

R-634: a whole-box backup no longer stops/restarts a DEPLOYING app (the
measured cause of containers running under 'not deployed'); StopStack
and StartStack refuse a deploying stack for every caller.
R-625: held badge 'Stopped - restore needed', no Update button.
R-636: kernel oom_kill counter; 20+ in 30 min -> one app_oom_storm.
R-647: held error per reader, copy_holds key, two log wordings.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-23 17:16:57 +02:00
parent 0a3026180a
commit 0054d4bd69
28 changed files with 832 additions and 48 deletions
+50 -1
View File
@@ -1,7 +1,9 @@
package stacks
import (
"fmt"
"sort"
"strconv"
"strings"
)
@@ -18,6 +20,13 @@ type OOMContainer struct {
Stack string
Container string
StartedAt string // identifies the container run — one event per run
// Kills is the kernel's own count of OOM kills in this container's cgroup (`memory.events`
// oom_kill) — R-636, v0.265.0. OOMKilled is a STICKY flag: it stays true for the container's whole
// life after ONE kill, so "the key re-fired" means nothing; this counter is what separates one
// hiccup from RomM's 4,530 kills in six hours. -1 when it could not be read.
Kills int64
// MemLimit / Peak are the cgroup's memory.max and memory.peak, as "<n>M" (or "max"); "" unread.
MemLimit, Peak string
}
// ScanOOMKilled inspects the containers of every deployed, running-ish stack in ONE docker call and
@@ -56,13 +65,53 @@ func (m *Manager) ScanOOMKilled() ([]OOMContainer, error) {
}
cname := strings.TrimPrefix(f[0], "/")
if st, ok := owner[cname]; ok {
found = append(found, OOMContainer{Stack: st, Container: cname, StartedAt: f[2]})
found = append(found, OOMContainer{Stack: st, Container: cname, StartedAt: f[2], Kills: -1})
}
}
// R-636: only for the containers already flagged — one `docker exec` each, and the flagged set is
// empty on a healthy box. Read from INSIDE the container (its cgroup namespace makes
// /sys/fs/cgroup its own cgroup); the controller's own namespace cannot see the others.
for i := range found {
out, _ := m.execCommand("docker", "exec", found[i].Container, "cat",
"/sys/fs/cgroup/memory.events", "/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory.peak")
found[i].Kills, found[i].MemLimit, found[i].Peak = parseCgroupMemory(out)
}
m.setOOMCache(found)
return found, nil
}
// parseCgroupMemory reads `cat memory.events memory.max memory.peak`: the key/value lines give
// oom_kill; the first bare line is memory.max, the second memory.peak (absent on older kernels — cat
// then fails on that file and still prints the others). Unreadable → Kills -1.
func parseCgroupMemory(out string) (kills int64, limit, peak string) {
kills = -1
var bare []string
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
f := strings.Fields(line)
switch {
case len(f) == 2 && f[0] == "oom_kill":
if n, err := strconv.ParseInt(f[1], 10, 64); err == nil {
kills = n
}
case len(f) == 1:
bare = append(bare, f[0])
}
}
mb := func(v string) string {
if n, err := strconv.ParseInt(v, 10, 64); err == nil {
return fmt.Sprintf("%dM", n/(1024*1024))
}
return v // "max"
}
if len(bare) > 0 {
limit = mb(bare[0])
}
if len(bare) > 1 {
peak = mb(bare[1])
}
return kills, limit, peak
}
func (m *Manager) setOOMCache(found []OOMContainer) {
cache := map[string][]string{}
for _, o := range found {