Files
felhom-controller/controller/internal/stacks/image_retention.go
T

421 lines
16 KiB
Go

package stacks
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"sort"
"strings"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
)
// ── Image retention (R-736, `09` §3 decision 53) ──────────────────────────────────────────────────────
//
// A remove ran `compose down --rmi local` (which never removes a registry-pulled image) and an update left the old
// version's image behind; nothing else deleted any. On scratch guest 9202 that filled the Docker disk until the box
// refused an install (2026-09-30: 84 images, 53 GB used by no container). The ruling: a box keeps, per app service,
// the image it runs now and the image before it (the undo's); it deletes older images of that app by itself; it
// NEVER deletes an image that any container (running or stopped) or any installed app's compose still names.
// Removing an app deletes that app's images under the same rule. Kept data (decision 40) is data, not images.
//
// THE KEEP SET is box-wide and rebuilt at every pass, at delete time: every container's image ID, every image an
// installed app's live compose names (by tag and by digest), and every installed app's installed_images and
// previous_images. A CANDIDATE is an image whose repository is one of THIS app's service repositories and whose ID
// is not kept. It is deleted by exact ID, never forced (Docker itself refuses an image a container uses) and never
// by prune; an ID that carries several repositories' tags is left alone. Every deletion is logged with its size.
// Pinned by internal/stacks/image_retention_test.go.
// imageDocker runs one docker command (a seam: tests never reach Docker).
var imageDocker = func(args ...string) (string, error) {
out, err := dockerexec.Command("docker", args...).CombinedOutput()
return string(out), err
}
var imageRetentionMu sync.Mutex
// errImageRetentionBusy: a pass did not run because an image may be in use by work in flight (an update, or
// any compose command that pulls — R-863). The caller tries again later; nothing was judged.
var errImageRetentionBusy = errors.New("image work in flight")
type localImage struct {
ID, Repo, Tag, Digest, Size string
}
func splitRepoTag(ref string) (repo, tag, digest string) {
if i := strings.Index(ref, "@"); i >= 0 {
ref, digest = ref[:i], ref[i+1:]
}
if c := strings.LastIndex(ref, ":"); c > strings.LastIndex(ref, "/") {
return ref[:c], ref[c+1:], digest
}
return ref, "latest", digest
}
// normRepo makes Docker Hub's short forms comparable: "library/postgres" and "docker.io/postgres" are "postgres".
func normRepo(r string) string {
r = strings.TrimPrefix(r, "docker.io/")
return strings.TrimPrefix(r, "library/")
}
func listLocalImages() ([]localImage, error) {
// -a (v0.284.2): the product pins `tag@digest`, and an image pulled that way is stored UNTAGGED (`repo:<none>`) —
// measured on 9202 2026-09-30: `docker image ls` without -a did not list any of them, so v0.284.1 could not see most
// app images. A fully anonymous `<none>:<none>` entry names no repository and is never a candidate.
out, err := imageDocker("image", "ls", "-a", "--digests", "--no-trunc", "--format", "{{.ID}}\t{{.Repository}}\t{{.Tag}}\t{{.Digest}}\t{{.Size}}")
if err != nil {
return nil, fmt.Errorf("docker image ls: %v: %s", err, truncateStr(out, 200))
}
var imgs []localImage
for _, l := range strings.Split(strings.TrimSpace(out), "\n") {
f := strings.Split(l, "\t")
if len(f) < 5 || f[0] == "" {
continue
}
imgs = append(imgs, localImage{ID: f[0], Repo: normRepo(f[1]), Tag: f[2], Digest: f[3], Size: f[4]})
}
return imgs, nil
}
func imagesUsedByContainers() (map[string]bool, error) {
out, err := imageDocker("ps", "-a", "-q", "--no-trunc")
if err != nil {
return nil, fmt.Errorf("docker ps: %v", err)
}
ids := strings.Fields(out)
used := map[string]bool{}
if len(ids) == 0 {
return used, nil
}
out, err = imageDocker(append([]string{"inspect", "--format", "{{.Image}}"}, ids...)...)
if err != nil {
// a container removed between the two calls fails the inspect: FAIL CLOSED — nothing is deleted
return nil, fmt.Errorf("docker inspect containers: %v", err)
}
for _, id := range strings.Fields(out) {
used[id] = true
}
return used, nil
}
// matchImages: the local image IDs a reference names — by repo+tag, or by repo+digest.
func matchImages(imgs []localImage, ref, digest string) []string {
repo, tag, d := splitRepoTag(ref)
repo = normRepo(repo)
if digest == "" {
digest = d
}
var ids []string
for _, im := range imgs {
if im.Repo != repo {
continue
}
if (tag != "" && im.Tag == tag) || (digest != "" && im.Digest == digest) {
ids = append(ids, im.ID)
}
}
return ids
}
// imageKeepSet is the box-wide keep set (see the header). except = an app being removed (its records do not keep).
func (m *Manager) imageKeepSet(imgs []localImage, except string) (map[string]bool, error) {
keep, err := imagesUsedByContainers()
if err != nil {
return nil, err
}
m.mu.RLock()
type app struct {
dir string
installed map[string]InstalledImage
previous map[string]InstalledImage
}
var apps []app
for n, st := range m.stacks {
if n == except || !st.Deployed {
continue
}
a := app{dir: filepath.Dir(st.ComposePath)}
if st.AppConfig != nil {
a.installed, a.previous = st.AppConfig.InstalledImages, st.AppConfig.PreviousImages
}
apps = append(apps, a)
}
m.mu.RUnlock()
for _, a := range apps {
if refs, err := ParseComposeImages(ComposePathIn(a.dir)); err == nil {
for _, ref := range refs {
for _, id := range matchImages(imgs, ref, "") {
keep[id] = true
}
}
}
for _, set := range []map[string]InstalledImage{a.installed, a.previous} {
for _, ii := range set {
for _, id := range matchImages(imgs, ii.Ref, ii.Digest) {
keep[id] = true
}
}
}
}
return keep, nil
}
// appImageRepos: the repositories an app's services use (its live compose, its records).
func appImageRepos(dir string, cfg *AppConfig) map[string]bool {
repos := map[string]bool{}
if refs, err := ParseComposeImages(ComposePathIn(dir)); err == nil {
for _, r := range refs {
rp, _, _ := splitRepoTag(r)
repos[normRepo(rp)] = true
}
}
if cfg != nil {
for _, set := range []map[string]InstalledImage{cfg.InstalledImages, cfg.PreviousImages} {
for _, ii := range set {
rp, _, _ := splitRepoTag(ii.Ref)
repos[normRepo(rp)] = true
}
}
}
return repos
}
// deleteUnkeptImages deletes every image of repos whose ID is not kept. Returns what it deleted.
func (m *Manager) deleteUnkeptImages(why string, repos map[string]bool, except string) ([]string, error) {
imageRetentionMu.Lock()
defer imageRetentionMu.Unlock()
// An update IN FLIGHT has already replaced its containers and its compose; the image its undo needs is then named
// by nothing the keep set reads. So no pass runs while any update runs (the next pass catches up).
m.mu.RLock()
busy := ""
for n, st := range m.stacks {
if st.Updating {
busy = n
break
}
}
m.mu.RUnlock()
if busy != "" {
m.logger.Printf("[INFO] [stacks] image retention (%s): skipped — %s is updating (its undo may need an image nothing else names)", why, busy)
return nil, errImageRetentionBusy
}
// R-863: an install, a restore or an undo inside `compose up` may have pulled an image (by digest, so
// untagged) that no container names YET. No pass while any image-pulling compose command runs; one that
// starts now waits for this pass (seconds).
endCleanup, ok := dockerexec.TryImageCleanup()
if !ok {
m.logger.Printf("[INFO] [stacks] image retention (%s): skipped — an app install, update, restore or undo is pulling images now (R-863); tried again later", why)
return nil, errImageRetentionBusy
}
defer endCleanup()
imgs, err := listLocalImages()
if err != nil {
return nil, err
}
keep, err := m.imageKeepSet(imgs, except)
if err != nil {
m.logger.Printf("[WARN] [stacks] image retention (%s): the keep set could not be read (%v) — NOTHING is deleted", why, err)
return nil, err
}
byID := map[string][]localImage{}
for _, im := range imgs {
byID[im.ID] = append(byID[im.ID], im)
}
ids := make([]string, 0, len(byID))
for id := range byID {
ids = append(ids, id)
}
sort.Strings(ids)
var deleted []string
candidates := 0
defer func() {
// ONE line per pass, whatever it did (v0.284.1): an absent deletion line must never be read as "it ran and
// found nothing" — this line is the positive observable that the pass ran (R-96 rule 3).
m.logger.Printf("[INFO] [stacks] image retention (%s): pass over %d image(s) of %v — %d candidate(s), %d deleted, the rest kept",
why, len(ids), sortedKeys(repos), candidates, len(deleted))
}()
for _, id := range ids {
group := byID[id]
if keep[id] {
continue
}
inRepos, names := true, []string{}
for _, im := range group {
if !repos[im.Repo] || strings.Contains(im.Repo, "felhom-controller") {
inRepos = false
}
names = append(names, im.Repo+":"+im.Tag)
}
if !inRepos {
continue
}
candidates++
if len(uniqueRepos(group)) > 1 {
m.logger.Printf("[INFO] [stacks] image retention (%s): %s carries several repositories' names %v — left alone", why, shortID(id), names)
continue
}
if out, err := imageDocker("rmi", id); err != nil {
m.logger.Printf("[WARN] [stacks] image retention (%s): docker refused to delete %v (%s): %s", why, names, shortID(id), truncateStr(strings.TrimSpace(out), 160))
continue
}
m.logger.Printf("[INFO] [stacks] image retention (%s): deleted %v (%s, %s) — no container, installed app or undo names it (decision 53)", why, names, shortID(id), group[0].Size)
deleted = append(deleted, strings.Join(names, ","))
}
return deleted, nil
}
func uniqueRepos(g []localImage) map[string]bool {
r := map[string]bool{}
for _, im := range g {
r[im.Repo] = true
}
return r
}
func shortID(id string) string {
id = strings.TrimPrefix(id, "sha256:")
if len(id) > 12 {
return id[:12]
}
return id
}
// RetainImagesAfterUpdate is called when a guarded Update ends. previous = what the app ran BEFORE the update when it
// ended done (the image before the new one); when it was undone, the images of the attempt (the app runs the old
// ones again and the attempt is the most recent other image). It records previous_images, then deletes the app's
// older images.
func (m *Manager) RetainImagesAfterUpdate(name string, previous map[string]InstalledImage) {
st, ok := m.GetStack(name)
if !ok || !st.Deployed {
return
}
dir := filepath.Dir(st.ComposePath)
if len(previous) > 0 {
m.mutateAppConfig(name, dir, "previous_images", func(cfg *AppConfig) bool {
cfg.PreviousImages = previous
return true
})
if err := m.ScanStacks(); err != nil {
m.logger.Printf("[WARN] [stacks] image retention %s: rescan failed: %v", name, err)
}
}
// R-751: the app can be gone by now (removed right after the update, or the rescan no longer finds it) — a nil
// stack here panicked in this goroutine and took the controller down. Its images are then the remove's to judge.
if st, ok = m.GetStack(name); !ok || st == nil {
m.logger.Printf("[INFO] [stacks] image retention after the update of %s: the app is gone — nothing to do here", name)
return
}
if _, err := m.deleteUnkeptImages("update of "+name, appImageRepos(dir, st.AppConfig), ""); err != nil && !errors.Is(err, errImageRetentionBusy) {
m.logger.Printf("[WARN] [stacks] image retention after the update of %s: %v", name, err)
}
}
// retainAfterUpdateFn runs the retention after an update ends (a seam: a no-op in this package's tests — TestMain, R-751).
var retainAfterUpdateFn = func(m *Manager, name string, previous map[string]InstalledImage) {
go m.RetainImagesAfterUpdate(name, previous)
}
func (m *Manager) retainAfterUpdate(name string, previous map[string]InstalledImage) {
retainAfterUpdateFn(m, name, previous)
}
// retainAfterRemoveFn runs the retention after a remove (a seam, so a test can see the remove path calls it).
var retainAfterRemoveFn = func(m *Manager, name string, repos map[string]bool) {
go m.RetainImagesAfterRemove(name, repos)
}
// RetainImagesAfterRemove deletes a removed app's images (its repos, read BEFORE the remove) that nothing else keeps.
func (m *Manager) RetainImagesAfterRemove(name string, repos map[string]bool) {
if len(repos) == 0 {
return
}
if _, err := m.deleteUnkeptImages("remove of "+name, repos, name); err != nil && !errors.Is(err, errImageRetentionBusy) {
m.logger.Printf("[WARN] [stacks] image retention after the remove of %s: %v", name, err)
}
}
// catalogImageRepos: every repository any catalog template or step names (the one-time sweep's reach: app images
// only — never the controller's, traefik's or another infrastructure image).
func (m *Manager) catalogImageRepos() map[string]bool {
repos := map[string]bool{}
root := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates")
_ = filepath.Walk(root, func(p string, info os.FileInfo, err error) error {
if err != nil || info.IsDir() || !(strings.HasSuffix(p, "docker-compose.yml") || (strings.Contains(p, string(filepath.Separator)+"steps"+string(filepath.Separator)) && strings.HasSuffix(p, ".yml") && !strings.HasSuffix(p, ".felhom.yml"))) {
return nil
}
if refs, err := ParseComposeImages(p); err == nil {
for _, r := range refs {
rp, _, _ := splitRepoTag(r)
repos[normRepo(rp)] = true
}
}
return nil
})
return repos
}
// imageRetentionMarker: the one-time sweep runs once per box (decision 53's clean-up for boxes older than it).
func (m *Manager) imageRetentionMarker() string {
return filepath.Join(m.cfg.Paths.DataDir, "image-retention-v2.done") // v2 (v0.284.2): the v1 sweep could not see untagged images
}
// RunImageRetentionOnce is the one-time clean-up at the first start of this release: the same rule, applied to every
// app image the catalog names (so the images of apps removed before this release go too). Logged; a marker file
// keeps it to once. Returns what it deleted, and done=false when it must be tried again (no catalog yet, or image
// work in flight — R-863: the marker is written ONLY after a pass that ran).
func (m *Manager) RunImageRetentionOnce() (deleted []string, done bool) {
if _, err := os.Stat(m.imageRetentionMarker()); err == nil {
return nil, true
}
repos := m.catalogImageRepos()
if len(repos) == 0 {
m.logger.Printf("[WARN] [stacks] image retention (one-time): no catalog read — skipped, tried again later")
return nil, false
}
before, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
deleted, err := m.deleteUnkeptImages("one-time clean-up", repos, "")
if errors.Is(err, errImageRetentionBusy) {
return nil, false // logged by the pass; tried again later
}
if err != nil {
m.logger.Printf("[WARN] [stacks] image retention (one-time): %v — tried again later", err)
return nil, false
}
after, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
m.logger.Printf("[INFO] [stacks] image retention (one-time): deleted %d image(s). docker disk before: %s | after: %s",
len(deleted), strings.Join(strings.Fields(firstLine(before)), " "), strings.Join(strings.Fields(firstLine(after)), " "))
_ = os.MkdirAll(filepath.Dir(m.imageRetentionMarker()), 0o755)
_ = os.WriteFile(m.imageRetentionMarker(), []byte(fmt.Sprintf("deleted %d\n%s\n", len(deleted), strings.Join(deleted, "\n"))), 0o644)
return deleted, true
}
// RunImageRetentionOnceUntilDone runs the one-time clean-up, and again every `every` until it has run (R-863:
// a pass that met image work in flight, or found no catalog yet, is retried — no longer only at the next start),
// at most `tries` times.
func (m *Manager) RunImageRetentionOnceUntilDone(ctx context.Context, every time.Duration, tries int) {
for i := 0; i < tries; i++ {
if _, done := m.RunImageRetentionOnce(); done {
return
}
select {
case <-ctx.Done():
return
case <-time.After(every):
}
}
m.logger.Printf("[WARN] [stacks] image retention (one-time): not run after %d tries — tried again at the next start", tries)
}
func sortedKeys(m map[string]bool) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
sort.Strings(out)
return out
}