ac8ea72025
gates / gates (push) Successful in 25s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
384 lines
14 KiB
Go
384 lines
14 KiB
Go
package stacks
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
"sync"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
|
)
|
|
|
|
// ── Image retention (R-736, `09` §3 decision 53) ──────────────────────────────────────────────────────
|
|
//
|
|
// A remove ran `compose down --rmi local` (which never removes a registry-pulled image) and an update left the old
|
|
// version's image behind; nothing else deleted any. On scratch guest 9202 that filled the Docker disk until the box
|
|
// refused an install (2026-09-30: 84 images, 53 GB used by no container). The ruling: a box keeps, per app service,
|
|
// the image it runs now and the image before it (the undo's); it deletes older images of that app by itself; it
|
|
// NEVER deletes an image that any container (running or stopped) or any installed app's compose still names.
|
|
// Removing an app deletes that app's images under the same rule. Kept data (decision 40) is data, not images.
|
|
//
|
|
// THE KEEP SET is box-wide and rebuilt at every pass, at delete time: every container's image ID, every image an
|
|
// installed app's live compose names (by tag and by digest), and every installed app's installed_images and
|
|
// previous_images. A CANDIDATE is an image whose repository is one of THIS app's service repositories and whose ID
|
|
// is not kept. It is deleted by exact ID, never forced (Docker itself refuses an image a container uses) and never
|
|
// by prune; an ID that carries several repositories' tags is left alone. Every deletion is logged with its size.
|
|
// Pinned by internal/stacks/image_retention_test.go.
|
|
|
|
// imageDocker runs one docker command (a seam: tests never reach Docker).
|
|
var imageDocker = func(args ...string) (string, error) {
|
|
out, err := dockerexec.Command("docker", args...).CombinedOutput()
|
|
return string(out), err
|
|
}
|
|
|
|
var imageRetentionMu sync.Mutex
|
|
|
|
type localImage struct {
|
|
ID, Repo, Tag, Digest, Size string
|
|
}
|
|
|
|
func splitRepoTag(ref string) (repo, tag, digest string) {
|
|
if i := strings.Index(ref, "@"); i >= 0 {
|
|
ref, digest = ref[:i], ref[i+1:]
|
|
}
|
|
if c := strings.LastIndex(ref, ":"); c > strings.LastIndex(ref, "/") {
|
|
return ref[:c], ref[c+1:], digest
|
|
}
|
|
return ref, "latest", digest
|
|
}
|
|
|
|
// normRepo makes Docker Hub's short forms comparable: "library/postgres" and "docker.io/postgres" are "postgres".
|
|
func normRepo(r string) string {
|
|
r = strings.TrimPrefix(r, "docker.io/")
|
|
return strings.TrimPrefix(r, "library/")
|
|
}
|
|
|
|
func listLocalImages() ([]localImage, error) {
|
|
// -a (v0.284.2): the product pins `tag@digest`, and an image pulled that way is stored UNTAGGED (`repo:<none>`) —
|
|
// measured on 9202 2026-09-30: `docker image ls` without -a did not list any of them, so v0.284.1 could not see most
|
|
// app images. A fully anonymous `<none>:<none>` entry names no repository and is never a candidate.
|
|
out, err := imageDocker("image", "ls", "-a", "--digests", "--no-trunc", "--format", "{{.ID}}\t{{.Repository}}\t{{.Tag}}\t{{.Digest}}\t{{.Size}}")
|
|
if err != nil {
|
|
return nil, fmt.Errorf("docker image ls: %v: %s", err, truncateStr(out, 200))
|
|
}
|
|
var imgs []localImage
|
|
for _, l := range strings.Split(strings.TrimSpace(out), "\n") {
|
|
f := strings.Split(l, "\t")
|
|
if len(f) < 5 || f[0] == "" {
|
|
continue
|
|
}
|
|
imgs = append(imgs, localImage{ID: f[0], Repo: normRepo(f[1]), Tag: f[2], Digest: f[3], Size: f[4]})
|
|
}
|
|
return imgs, nil
|
|
}
|
|
|
|
func imagesUsedByContainers() (map[string]bool, error) {
|
|
out, err := imageDocker("ps", "-a", "-q", "--no-trunc")
|
|
if err != nil {
|
|
return nil, fmt.Errorf("docker ps: %v", err)
|
|
}
|
|
ids := strings.Fields(out)
|
|
used := map[string]bool{}
|
|
if len(ids) == 0 {
|
|
return used, nil
|
|
}
|
|
out, err = imageDocker(append([]string{"inspect", "--format", "{{.Image}}"}, ids...)...)
|
|
if err != nil {
|
|
// a container removed between the two calls fails the inspect: FAIL CLOSED — nothing is deleted
|
|
return nil, fmt.Errorf("docker inspect containers: %v", err)
|
|
}
|
|
for _, id := range strings.Fields(out) {
|
|
used[id] = true
|
|
}
|
|
return used, nil
|
|
}
|
|
|
|
// matchImages: the local image IDs a reference names — by repo+tag, or by repo+digest.
|
|
func matchImages(imgs []localImage, ref, digest string) []string {
|
|
repo, tag, d := splitRepoTag(ref)
|
|
repo = normRepo(repo)
|
|
if digest == "" {
|
|
digest = d
|
|
}
|
|
var ids []string
|
|
for _, im := range imgs {
|
|
if im.Repo != repo {
|
|
continue
|
|
}
|
|
if (tag != "" && im.Tag == tag) || (digest != "" && im.Digest == digest) {
|
|
ids = append(ids, im.ID)
|
|
}
|
|
}
|
|
return ids
|
|
}
|
|
|
|
// imageKeepSet is the box-wide keep set (see the header). except = an app being removed (its records do not keep).
|
|
func (m *Manager) imageKeepSet(imgs []localImage, except string) (map[string]bool, error) {
|
|
keep, err := imagesUsedByContainers()
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
m.mu.RLock()
|
|
type app struct {
|
|
dir string
|
|
installed map[string]InstalledImage
|
|
previous map[string]InstalledImage
|
|
}
|
|
var apps []app
|
|
for n, st := range m.stacks {
|
|
if n == except || !st.Deployed {
|
|
continue
|
|
}
|
|
a := app{dir: filepath.Dir(st.ComposePath)}
|
|
if st.AppConfig != nil {
|
|
a.installed, a.previous = st.AppConfig.InstalledImages, st.AppConfig.PreviousImages
|
|
}
|
|
apps = append(apps, a)
|
|
}
|
|
m.mu.RUnlock()
|
|
for _, a := range apps {
|
|
if refs, err := ParseComposeImages(ComposePathIn(a.dir)); err == nil {
|
|
for _, ref := range refs {
|
|
for _, id := range matchImages(imgs, ref, "") {
|
|
keep[id] = true
|
|
}
|
|
}
|
|
}
|
|
for _, set := range []map[string]InstalledImage{a.installed, a.previous} {
|
|
for _, ii := range set {
|
|
for _, id := range matchImages(imgs, ii.Ref, ii.Digest) {
|
|
keep[id] = true
|
|
}
|
|
}
|
|
}
|
|
}
|
|
return keep, nil
|
|
}
|
|
|
|
// appImageRepos: the repositories an app's services use (its live compose, its records).
|
|
func appImageRepos(dir string, cfg *AppConfig) map[string]bool {
|
|
repos := map[string]bool{}
|
|
if refs, err := ParseComposeImages(ComposePathIn(dir)); err == nil {
|
|
for _, r := range refs {
|
|
rp, _, _ := splitRepoTag(r)
|
|
repos[normRepo(rp)] = true
|
|
}
|
|
}
|
|
if cfg != nil {
|
|
for _, set := range []map[string]InstalledImage{cfg.InstalledImages, cfg.PreviousImages} {
|
|
for _, ii := range set {
|
|
rp, _, _ := splitRepoTag(ii.Ref)
|
|
repos[normRepo(rp)] = true
|
|
}
|
|
}
|
|
}
|
|
return repos
|
|
}
|
|
|
|
// deleteUnkeptImages deletes every image of repos whose ID is not kept. Returns what it deleted.
|
|
func (m *Manager) deleteUnkeptImages(why string, repos map[string]bool, except string) ([]string, error) {
|
|
imageRetentionMu.Lock()
|
|
defer imageRetentionMu.Unlock()
|
|
// An update IN FLIGHT has already replaced its containers and its compose; the image its undo needs is then named
|
|
// by nothing the keep set reads. So no pass runs while any update runs (the next pass catches up).
|
|
m.mu.RLock()
|
|
busy := ""
|
|
for n, st := range m.stacks {
|
|
if st.Updating {
|
|
busy = n
|
|
break
|
|
}
|
|
}
|
|
m.mu.RUnlock()
|
|
if busy != "" {
|
|
m.logger.Printf("[INFO] [stacks] image retention (%s): skipped — %s is updating (its undo may need an image nothing else names)", why, busy)
|
|
return nil, nil
|
|
}
|
|
imgs, err := listLocalImages()
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
keep, err := m.imageKeepSet(imgs, except)
|
|
if err != nil {
|
|
m.logger.Printf("[WARN] [stacks] image retention (%s): the keep set could not be read (%v) — NOTHING is deleted", why, err)
|
|
return nil, err
|
|
}
|
|
byID := map[string][]localImage{}
|
|
for _, im := range imgs {
|
|
byID[im.ID] = append(byID[im.ID], im)
|
|
}
|
|
ids := make([]string, 0, len(byID))
|
|
for id := range byID {
|
|
ids = append(ids, id)
|
|
}
|
|
sort.Strings(ids)
|
|
var deleted []string
|
|
candidates := 0
|
|
defer func() {
|
|
// ONE line per pass, whatever it did (v0.284.1): an absent deletion line must never be read as "it ran and
|
|
// found nothing" — this line is the positive observable that the pass ran (R-96 rule 3).
|
|
m.logger.Printf("[INFO] [stacks] image retention (%s): pass over %d image(s) of %v — %d candidate(s), %d deleted, the rest kept",
|
|
why, len(ids), sortedKeys(repos), candidates, len(deleted))
|
|
}()
|
|
for _, id := range ids {
|
|
group := byID[id]
|
|
if keep[id] {
|
|
continue
|
|
}
|
|
inRepos, names := true, []string{}
|
|
for _, im := range group {
|
|
if !repos[im.Repo] || strings.Contains(im.Repo, "felhom-controller") {
|
|
inRepos = false
|
|
}
|
|
names = append(names, im.Repo+":"+im.Tag)
|
|
}
|
|
if !inRepos {
|
|
continue
|
|
}
|
|
candidates++
|
|
if len(uniqueRepos(group)) > 1 {
|
|
m.logger.Printf("[INFO] [stacks] image retention (%s): %s carries several repositories' names %v — left alone", why, shortID(id), names)
|
|
continue
|
|
}
|
|
if out, err := imageDocker("rmi", id); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] image retention (%s): docker refused to delete %v (%s): %s", why, names, shortID(id), truncateStr(strings.TrimSpace(out), 160))
|
|
continue
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] image retention (%s): deleted %v (%s, %s) — no container, installed app or undo names it (decision 53)", why, names, shortID(id), group[0].Size)
|
|
deleted = append(deleted, strings.Join(names, ","))
|
|
}
|
|
return deleted, nil
|
|
}
|
|
|
|
func uniqueRepos(g []localImage) map[string]bool {
|
|
r := map[string]bool{}
|
|
for _, im := range g {
|
|
r[im.Repo] = true
|
|
}
|
|
return r
|
|
}
|
|
|
|
func shortID(id string) string {
|
|
id = strings.TrimPrefix(id, "sha256:")
|
|
if len(id) > 12 {
|
|
return id[:12]
|
|
}
|
|
return id
|
|
}
|
|
|
|
// RetainImagesAfterUpdate is called when a guarded Update ends. previous = what the app ran BEFORE the update when it
|
|
// ended done (the image before the new one); when it was undone, the images of the attempt (the app runs the old
|
|
// ones again and the attempt is the most recent other image). It records previous_images, then deletes the app's
|
|
// older images.
|
|
func (m *Manager) RetainImagesAfterUpdate(name string, previous map[string]InstalledImage) {
|
|
st, ok := m.GetStack(name)
|
|
if !ok || !st.Deployed {
|
|
return
|
|
}
|
|
dir := filepath.Dir(st.ComposePath)
|
|
if len(previous) > 0 {
|
|
m.mutateAppConfig(name, dir, "previous_images", func(cfg *AppConfig) bool {
|
|
cfg.PreviousImages = previous
|
|
return true
|
|
})
|
|
if err := m.ScanStacks(); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] image retention %s: rescan failed: %v", name, err)
|
|
}
|
|
}
|
|
// R-751: the app can be gone by now (removed right after the update, or the rescan no longer finds it) — a nil
|
|
// stack here panicked in this goroutine and took the controller down. Its images are then the remove's to judge.
|
|
if st, ok = m.GetStack(name); !ok || st == nil {
|
|
m.logger.Printf("[INFO] [stacks] image retention after the update of %s: the app is gone — nothing to do here", name)
|
|
return
|
|
}
|
|
if _, err := m.deleteUnkeptImages("update of "+name, appImageRepos(dir, st.AppConfig), ""); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] image retention after the update of %s: %v", name, err)
|
|
}
|
|
}
|
|
|
|
// retainAfterUpdateFn runs the retention after an update ends (a seam: a no-op in this package's tests — TestMain, R-751).
|
|
var retainAfterUpdateFn = func(m *Manager, name string, previous map[string]InstalledImage) {
|
|
go m.RetainImagesAfterUpdate(name, previous)
|
|
}
|
|
|
|
func (m *Manager) retainAfterUpdate(name string, previous map[string]InstalledImage) {
|
|
retainAfterUpdateFn(m, name, previous)
|
|
}
|
|
|
|
// retainAfterRemoveFn runs the retention after a remove (a seam, so a test can see the remove path calls it).
|
|
var retainAfterRemoveFn = func(m *Manager, name string, repos map[string]bool) {
|
|
go m.RetainImagesAfterRemove(name, repos)
|
|
}
|
|
|
|
// RetainImagesAfterRemove deletes a removed app's images (its repos, read BEFORE the remove) that nothing else keeps.
|
|
func (m *Manager) RetainImagesAfterRemove(name string, repos map[string]bool) {
|
|
if len(repos) == 0 {
|
|
return
|
|
}
|
|
if _, err := m.deleteUnkeptImages("remove of "+name, repos, name); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] image retention after the remove of %s: %v", name, err)
|
|
}
|
|
}
|
|
|
|
// catalogImageRepos: every repository any catalog template or step names (the one-time sweep's reach: app images
|
|
// only — never the controller's, traefik's or another infrastructure image).
|
|
func (m *Manager) catalogImageRepos() map[string]bool {
|
|
repos := map[string]bool{}
|
|
root := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates")
|
|
_ = filepath.Walk(root, func(p string, info os.FileInfo, err error) error {
|
|
if err != nil || info.IsDir() || !(strings.HasSuffix(p, "docker-compose.yml") || (strings.Contains(p, string(filepath.Separator)+"steps"+string(filepath.Separator)) && strings.HasSuffix(p, ".yml") && !strings.HasSuffix(p, ".felhom.yml"))) {
|
|
return nil
|
|
}
|
|
if refs, err := ParseComposeImages(p); err == nil {
|
|
for _, r := range refs {
|
|
rp, _, _ := splitRepoTag(r)
|
|
repos[normRepo(rp)] = true
|
|
}
|
|
}
|
|
return nil
|
|
})
|
|
return repos
|
|
}
|
|
|
|
// imageRetentionMarker: the one-time sweep runs once per box (decision 53's clean-up for boxes older than it).
|
|
func (m *Manager) imageRetentionMarker() string {
|
|
return filepath.Join(m.cfg.Paths.DataDir, "image-retention-v2.done") // v2 (v0.284.2): the v1 sweep could not see untagged images
|
|
}
|
|
|
|
// RunImageRetentionOnce is the one-time clean-up at the first start of this release: the same rule, applied to every
|
|
// app image the catalog names (so the images of apps removed before this release go too). Logged; a marker file
|
|
// keeps it to once. Returns what it deleted.
|
|
func (m *Manager) RunImageRetentionOnce() []string {
|
|
if _, err := os.Stat(m.imageRetentionMarker()); err == nil {
|
|
return nil
|
|
}
|
|
repos := m.catalogImageRepos()
|
|
if len(repos) == 0 {
|
|
m.logger.Printf("[WARN] [stacks] image retention (one-time): no catalog read — skipped, tried again at the next start")
|
|
return nil
|
|
}
|
|
before, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
|
|
deleted, err := m.deleteUnkeptImages("one-time clean-up", repos, "")
|
|
if err != nil {
|
|
m.logger.Printf("[WARN] [stacks] image retention (one-time): %v — tried again at the next start", err)
|
|
return nil
|
|
}
|
|
after, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
|
|
m.logger.Printf("[INFO] [stacks] image retention (one-time): deleted %d image(s). docker disk before: %s | after: %s",
|
|
len(deleted), strings.Join(strings.Fields(firstLine(before)), " "), strings.Join(strings.Fields(firstLine(after)), " "))
|
|
_ = os.MkdirAll(filepath.Dir(m.imageRetentionMarker()), 0o755)
|
|
_ = os.WriteFile(m.imageRetentionMarker(), []byte(fmt.Sprintf("deleted %d\n%s\n", len(deleted), strings.Join(deleted, "\n"))), 0o644)
|
|
return deleted
|
|
}
|
|
|
|
func sortedKeys(m map[string]bool) []string {
|
|
out := make([]string, 0, len(m))
|
|
for k := range m {
|
|
out = append(out, k)
|
|
}
|
|
sort.Strings(out)
|
|
return out
|
|
}
|