controller v0.265.0: R-634 cause fixed, held apps say so, OOM storm alarm, R-647 leftovers
gates / gates (push) Successful in 27s
gates / gates (push) Successful in 27s
R-634: a whole-box backup no longer stops/restarts a DEPLOYING app (the measured cause of containers running under 'not deployed'); StopStack and StartStack refuse a deploying stack for every caller. R-625: held badge 'Stopped - restore needed', no Update button. R-636: kernel oom_kill counter; 20+ in 30 min -> one app_oom_storm. R-647: held error per reader, copy_holds key, two log wordings. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -1,7 +1,9 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
@@ -18,6 +20,13 @@ type OOMContainer struct {
|
||||
Stack string
|
||||
Container string
|
||||
StartedAt string // identifies the container run — one event per run
|
||||
// Kills is the kernel's own count of OOM kills in this container's cgroup (`memory.events`
|
||||
// oom_kill) — R-636, v0.265.0. OOMKilled is a STICKY flag: it stays true for the container's whole
|
||||
// life after ONE kill, so "the key re-fired" means nothing; this counter is what separates one
|
||||
// hiccup from RomM's 4,530 kills in six hours. -1 when it could not be read.
|
||||
Kills int64
|
||||
// MemLimit / Peak are the cgroup's memory.max and memory.peak, as "<n>M" (or "max"); "" unread.
|
||||
MemLimit, Peak string
|
||||
}
|
||||
|
||||
// ScanOOMKilled inspects the containers of every deployed, running-ish stack in ONE docker call and
|
||||
@@ -56,13 +65,53 @@ func (m *Manager) ScanOOMKilled() ([]OOMContainer, error) {
|
||||
}
|
||||
cname := strings.TrimPrefix(f[0], "/")
|
||||
if st, ok := owner[cname]; ok {
|
||||
found = append(found, OOMContainer{Stack: st, Container: cname, StartedAt: f[2]})
|
||||
found = append(found, OOMContainer{Stack: st, Container: cname, StartedAt: f[2], Kills: -1})
|
||||
}
|
||||
}
|
||||
// R-636: only for the containers already flagged — one `docker exec` each, and the flagged set is
|
||||
// empty on a healthy box. Read from INSIDE the container (its cgroup namespace makes
|
||||
// /sys/fs/cgroup its own cgroup); the controller's own namespace cannot see the others.
|
||||
for i := range found {
|
||||
out, _ := m.execCommand("docker", "exec", found[i].Container, "cat",
|
||||
"/sys/fs/cgroup/memory.events", "/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory.peak")
|
||||
found[i].Kills, found[i].MemLimit, found[i].Peak = parseCgroupMemory(out)
|
||||
}
|
||||
m.setOOMCache(found)
|
||||
return found, nil
|
||||
}
|
||||
|
||||
// parseCgroupMemory reads `cat memory.events memory.max memory.peak`: the key/value lines give
|
||||
// oom_kill; the first bare line is memory.max, the second memory.peak (absent on older kernels — cat
|
||||
// then fails on that file and still prints the others). Unreadable → Kills -1.
|
||||
func parseCgroupMemory(out string) (kills int64, limit, peak string) {
|
||||
kills = -1
|
||||
var bare []string
|
||||
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
|
||||
f := strings.Fields(line)
|
||||
switch {
|
||||
case len(f) == 2 && f[0] == "oom_kill":
|
||||
if n, err := strconv.ParseInt(f[1], 10, 64); err == nil {
|
||||
kills = n
|
||||
}
|
||||
case len(f) == 1:
|
||||
bare = append(bare, f[0])
|
||||
}
|
||||
}
|
||||
mb := func(v string) string {
|
||||
if n, err := strconv.ParseInt(v, 10, 64); err == nil {
|
||||
return fmt.Sprintf("%dM", n/(1024*1024))
|
||||
}
|
||||
return v // "max"
|
||||
}
|
||||
if len(bare) > 0 {
|
||||
limit = mb(bare[0])
|
||||
}
|
||||
if len(bare) > 1 {
|
||||
peak = mb(bare[1])
|
||||
}
|
||||
return kills, limit, peak
|
||||
}
|
||||
|
||||
func (m *Manager) setOOMCache(found []OOMContainer) {
|
||||
cache := map[string][]string{}
|
||||
for _, o := range found {
|
||||
|
||||
Reference in New Issue
Block a user