v0.284.0 — a box deletes old app images (decision 53, R-736); an after_install app is held until its known login is replaced (R-741)
gates / gates (push) Successful in 27s
gates / gates (push) Successful in 27s
Image retention: after a done/undone guarded Update and at remove, an app's images older than its running and previous one are deleted — never an image any container, installed compose or installed/previous record names (box-wide keep set read at delete time); exact id, never forced or pruned; paused while any update runs; a one-time sweep of catalog app images at the first start. Install hold: an after_install app is installed behind the setup gate's door and opens when after_install succeeds or the household says it changed the login. Tests TestImageRetention_* and TestInstallHold_* with red-proofs; parity fixture for the held card. MinAgent: 0.131.0 (unchanged). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,353 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
||||
)
|
||||
|
||||
// ── Image retention (R-736, `09` §3 decision 53) ──────────────────────────────────────────────────────
|
||||
//
|
||||
// A remove ran `compose down --rmi local` (which never removes a registry-pulled image) and an update left the old
|
||||
// version's image behind; nothing else deleted any. On scratch guest 9202 that filled the Docker disk until the box
|
||||
// refused an install (2026-09-30: 84 images, 53 GB used by no container). The ruling: a box keeps, per app service,
|
||||
// the image it runs now and the image before it (the undo's); it deletes older images of that app by itself; it
|
||||
// NEVER deletes an image that any container (running or stopped) or any installed app's compose still names.
|
||||
// Removing an app deletes that app's images under the same rule. Kept data (decision 40) is data, not images.
|
||||
//
|
||||
// THE KEEP SET is box-wide and rebuilt at every pass, at delete time: every container's image ID, every image an
|
||||
// installed app's live compose names (by tag and by digest), and every installed app's installed_images and
|
||||
// previous_images. A CANDIDATE is an image whose repository is one of THIS app's service repositories and whose ID
|
||||
// is not kept. It is deleted by exact ID, never forced (Docker itself refuses an image a container uses) and never
|
||||
// by prune; an ID that carries several repositories' tags is left alone. Every deletion is logged with its size.
|
||||
// Pinned by internal/stacks/image_retention_test.go.
|
||||
|
||||
// imageDocker runs one docker command (a seam: tests never reach Docker).
|
||||
var imageDocker = func(args ...string) (string, error) {
|
||||
out, err := dockerexec.Command("docker", args...).CombinedOutput()
|
||||
return string(out), err
|
||||
}
|
||||
|
||||
var imageRetentionMu sync.Mutex
|
||||
|
||||
type localImage struct {
|
||||
ID, Repo, Tag, Digest, Size string
|
||||
}
|
||||
|
||||
func splitRepoTag(ref string) (repo, tag, digest string) {
|
||||
if i := strings.Index(ref, "@"); i >= 0 {
|
||||
ref, digest = ref[:i], ref[i+1:]
|
||||
}
|
||||
if c := strings.LastIndex(ref, ":"); c > strings.LastIndex(ref, "/") {
|
||||
return ref[:c], ref[c+1:], digest
|
||||
}
|
||||
return ref, "latest", digest
|
||||
}
|
||||
|
||||
// normRepo makes Docker Hub's short forms comparable: "library/postgres" and "docker.io/postgres" are "postgres".
|
||||
func normRepo(r string) string {
|
||||
r = strings.TrimPrefix(r, "docker.io/")
|
||||
return strings.TrimPrefix(r, "library/")
|
||||
}
|
||||
|
||||
func listLocalImages() ([]localImage, error) {
|
||||
out, err := imageDocker("image", "ls", "--digests", "--no-trunc", "--format", "{{.ID}}\t{{.Repository}}\t{{.Tag}}\t{{.Digest}}\t{{.Size}}")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("docker image ls: %v: %s", err, truncateStr(out, 200))
|
||||
}
|
||||
var imgs []localImage
|
||||
for _, l := range strings.Split(strings.TrimSpace(out), "\n") {
|
||||
f := strings.Split(l, "\t")
|
||||
if len(f) < 5 || f[0] == "" {
|
||||
continue
|
||||
}
|
||||
imgs = append(imgs, localImage{ID: f[0], Repo: normRepo(f[1]), Tag: f[2], Digest: f[3], Size: f[4]})
|
||||
}
|
||||
return imgs, nil
|
||||
}
|
||||
|
||||
func imagesUsedByContainers() (map[string]bool, error) {
|
||||
out, err := imageDocker("ps", "-a", "-q", "--no-trunc")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("docker ps: %v", err)
|
||||
}
|
||||
ids := strings.Fields(out)
|
||||
used := map[string]bool{}
|
||||
if len(ids) == 0 {
|
||||
return used, nil
|
||||
}
|
||||
out, err = imageDocker(append([]string{"inspect", "--format", "{{.Image}}"}, ids...)...)
|
||||
if err != nil {
|
||||
// a container removed between the two calls fails the inspect: FAIL CLOSED — nothing is deleted
|
||||
return nil, fmt.Errorf("docker inspect containers: %v", err)
|
||||
}
|
||||
for _, id := range strings.Fields(out) {
|
||||
used[id] = true
|
||||
}
|
||||
return used, nil
|
||||
}
|
||||
|
||||
// matchImages: the local image IDs a reference names — by repo+tag, or by repo+digest.
|
||||
func matchImages(imgs []localImage, ref, digest string) []string {
|
||||
repo, tag, d := splitRepoTag(ref)
|
||||
repo = normRepo(repo)
|
||||
if digest == "" {
|
||||
digest = d
|
||||
}
|
||||
var ids []string
|
||||
for _, im := range imgs {
|
||||
if im.Repo != repo {
|
||||
continue
|
||||
}
|
||||
if (tag != "" && im.Tag == tag) || (digest != "" && im.Digest == digest) {
|
||||
ids = append(ids, im.ID)
|
||||
}
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
// imageKeepSet is the box-wide keep set (see the header). except = an app being removed (its records do not keep).
|
||||
func (m *Manager) imageKeepSet(imgs []localImage, except string) (map[string]bool, error) {
|
||||
keep, err := imagesUsedByContainers()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
m.mu.RLock()
|
||||
type app struct {
|
||||
dir string
|
||||
installed map[string]InstalledImage
|
||||
previous map[string]InstalledImage
|
||||
}
|
||||
var apps []app
|
||||
for n, st := range m.stacks {
|
||||
if n == except || !st.Deployed {
|
||||
continue
|
||||
}
|
||||
a := app{dir: filepath.Dir(st.ComposePath)}
|
||||
if st.AppConfig != nil {
|
||||
a.installed, a.previous = st.AppConfig.InstalledImages, st.AppConfig.PreviousImages
|
||||
}
|
||||
apps = append(apps, a)
|
||||
}
|
||||
m.mu.RUnlock()
|
||||
for _, a := range apps {
|
||||
if refs, err := ParseComposeImages(ComposePathIn(a.dir)); err == nil {
|
||||
for _, ref := range refs {
|
||||
for _, id := range matchImages(imgs, ref, "") {
|
||||
keep[id] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, set := range []map[string]InstalledImage{a.installed, a.previous} {
|
||||
for _, ii := range set {
|
||||
for _, id := range matchImages(imgs, ii.Ref, ii.Digest) {
|
||||
keep[id] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return keep, nil
|
||||
}
|
||||
|
||||
// appImageRepos: the repositories an app's services use (its live compose, its records).
|
||||
func appImageRepos(dir string, cfg *AppConfig) map[string]bool {
|
||||
repos := map[string]bool{}
|
||||
if refs, err := ParseComposeImages(ComposePathIn(dir)); err == nil {
|
||||
for _, r := range refs {
|
||||
rp, _, _ := splitRepoTag(r)
|
||||
repos[normRepo(rp)] = true
|
||||
}
|
||||
}
|
||||
if cfg != nil {
|
||||
for _, set := range []map[string]InstalledImage{cfg.InstalledImages, cfg.PreviousImages} {
|
||||
for _, ii := range set {
|
||||
rp, _, _ := splitRepoTag(ii.Ref)
|
||||
repos[normRepo(rp)] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
return repos
|
||||
}
|
||||
|
||||
// deleteUnkeptImages deletes every image of repos whose ID is not kept. Returns what it deleted.
|
||||
func (m *Manager) deleteUnkeptImages(why string, repos map[string]bool, except string) ([]string, error) {
|
||||
imageRetentionMu.Lock()
|
||||
defer imageRetentionMu.Unlock()
|
||||
// An update IN FLIGHT has already replaced its containers and its compose; the image its undo needs is then named
|
||||
// by nothing the keep set reads. So no pass runs while any update runs (the next pass catches up).
|
||||
m.mu.RLock()
|
||||
busy := ""
|
||||
for n, st := range m.stacks {
|
||||
if st.Updating {
|
||||
busy = n
|
||||
break
|
||||
}
|
||||
}
|
||||
m.mu.RUnlock()
|
||||
if busy != "" {
|
||||
m.logger.Printf("[INFO] [stacks] image retention (%s): skipped — %s is updating (its undo may need an image nothing else names)", why, busy)
|
||||
return nil, nil
|
||||
}
|
||||
imgs, err := listLocalImages()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
keep, err := m.imageKeepSet(imgs, except)
|
||||
if err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] image retention (%s): the keep set could not be read (%v) — NOTHING is deleted", why, err)
|
||||
return nil, err
|
||||
}
|
||||
byID := map[string][]localImage{}
|
||||
for _, im := range imgs {
|
||||
byID[im.ID] = append(byID[im.ID], im)
|
||||
}
|
||||
ids := make([]string, 0, len(byID))
|
||||
for id := range byID {
|
||||
ids = append(ids, id)
|
||||
}
|
||||
sort.Strings(ids)
|
||||
var deleted []string
|
||||
for _, id := range ids {
|
||||
group := byID[id]
|
||||
if keep[id] {
|
||||
continue
|
||||
}
|
||||
inRepos, names := true, []string{}
|
||||
for _, im := range group {
|
||||
if !repos[im.Repo] || strings.Contains(im.Repo, "felhom-controller") {
|
||||
inRepos = false
|
||||
}
|
||||
names = append(names, im.Repo+":"+im.Tag)
|
||||
}
|
||||
if !inRepos {
|
||||
continue
|
||||
}
|
||||
if len(uniqueRepos(group)) > 1 {
|
||||
m.logger.Printf("[INFO] [stacks] image retention (%s): %s carries several repositories' names %v — left alone", why, shortID(id), names)
|
||||
continue
|
||||
}
|
||||
if out, err := imageDocker("rmi", id); err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] image retention (%s): docker refused to delete %v (%s): %s", why, names, shortID(id), truncateStr(strings.TrimSpace(out), 160))
|
||||
continue
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] image retention (%s): deleted %v (%s, %s) — no container, installed app or undo names it (decision 53)", why, names, shortID(id), group[0].Size)
|
||||
deleted = append(deleted, strings.Join(names, ","))
|
||||
}
|
||||
return deleted, nil
|
||||
}
|
||||
|
||||
func uniqueRepos(g []localImage) map[string]bool {
|
||||
r := map[string]bool{}
|
||||
for _, im := range g {
|
||||
r[im.Repo] = true
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
func shortID(id string) string {
|
||||
id = strings.TrimPrefix(id, "sha256:")
|
||||
if len(id) > 12 {
|
||||
return id[:12]
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
// RetainImagesAfterUpdate is called when a guarded Update ends. previous = what the app ran BEFORE the update when it
|
||||
// ended done (the image before the new one); when it was undone, the images of the attempt (the app runs the old
|
||||
// ones again and the attempt is the most recent other image). It records previous_images, then deletes the app's
|
||||
// older images.
|
||||
func (m *Manager) RetainImagesAfterUpdate(name string, previous map[string]InstalledImage) {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok || !st.Deployed {
|
||||
return
|
||||
}
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
if len(previous) > 0 {
|
||||
m.mutateAppConfig(name, dir, "previous_images", func(cfg *AppConfig) bool {
|
||||
cfg.PreviousImages = previous
|
||||
return true
|
||||
})
|
||||
if err := m.ScanStacks(); err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] image retention %s: rescan failed: %v", name, err)
|
||||
}
|
||||
}
|
||||
st, _ = m.GetStack(name)
|
||||
if _, err := m.deleteUnkeptImages("update of "+name, appImageRepos(dir, st.AppConfig), ""); err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] image retention after the update of %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
|
||||
// retainAfterUpdateFn runs the retention after an update ends (a seam: the update tests do not exercise it).
|
||||
var retainAfterUpdateFn = func(m *Manager, name string, previous map[string]InstalledImage) {
|
||||
go m.RetainImagesAfterUpdate(name, previous)
|
||||
}
|
||||
|
||||
func (m *Manager) retainAfterUpdate(name string, previous map[string]InstalledImage) {
|
||||
retainAfterUpdateFn(m, name, previous)
|
||||
}
|
||||
|
||||
// RetainImagesAfterRemove deletes a removed app's images (its repos, read BEFORE the remove) that nothing else keeps.
|
||||
func (m *Manager) RetainImagesAfterRemove(name string, repos map[string]bool) {
|
||||
if len(repos) == 0 {
|
||||
return
|
||||
}
|
||||
if _, err := m.deleteUnkeptImages("remove of "+name, repos, name); err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] image retention after the remove of %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
|
||||
// catalogImageRepos: every repository any catalog template or step names (the one-time sweep's reach: app images
|
||||
// only — never the controller's, traefik's or another infrastructure image).
|
||||
func (m *Manager) catalogImageRepos() map[string]bool {
|
||||
repos := map[string]bool{}
|
||||
root := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates")
|
||||
_ = filepath.Walk(root, func(p string, info os.FileInfo, err error) error {
|
||||
if err != nil || info.IsDir() || !(strings.HasSuffix(p, "docker-compose.yml") || (strings.Contains(p, string(filepath.Separator)+"steps"+string(filepath.Separator)) && strings.HasSuffix(p, ".yml") && !strings.HasSuffix(p, ".felhom.yml"))) {
|
||||
return nil
|
||||
}
|
||||
if refs, err := ParseComposeImages(p); err == nil {
|
||||
for _, r := range refs {
|
||||
rp, _, _ := splitRepoTag(r)
|
||||
repos[normRepo(rp)] = true
|
||||
}
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return repos
|
||||
}
|
||||
|
||||
// imageRetentionMarker: the one-time sweep runs once per box (decision 53's clean-up for boxes older than it).
|
||||
func (m *Manager) imageRetentionMarker() string {
|
||||
return filepath.Join(m.cfg.Paths.DataDir, "image-retention-v1.done")
|
||||
}
|
||||
|
||||
// RunImageRetentionOnce is the one-time clean-up at the first start of this release: the same rule, applied to every
|
||||
// app image the catalog names (so the images of apps removed before this release go too). Logged; a marker file
|
||||
// keeps it to once. Returns what it deleted.
|
||||
func (m *Manager) RunImageRetentionOnce() []string {
|
||||
if _, err := os.Stat(m.imageRetentionMarker()); err == nil {
|
||||
return nil
|
||||
}
|
||||
repos := m.catalogImageRepos()
|
||||
if len(repos) == 0 {
|
||||
m.logger.Printf("[WARN] [stacks] image retention (one-time): no catalog read — skipped, tried again at the next start")
|
||||
return nil
|
||||
}
|
||||
before, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
|
||||
deleted, err := m.deleteUnkeptImages("one-time clean-up", repos, "")
|
||||
if err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] image retention (one-time): %v — tried again at the next start", err)
|
||||
return nil
|
||||
}
|
||||
after, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
|
||||
m.logger.Printf("[INFO] [stacks] image retention (one-time): deleted %d image(s). docker disk before: %s | after: %s",
|
||||
len(deleted), strings.Join(strings.Fields(firstLine(before)), " "), strings.Join(strings.Fields(firstLine(after)), " "))
|
||||
_ = os.MkdirAll(filepath.Dir(m.imageRetentionMarker()), 0o755)
|
||||
_ = os.WriteFile(m.imageRetentionMarker(), []byte(fmt.Sprintf("deleted %d\n%s\n", len(deleted), strings.Join(deleted, "\n"))), 0o644)
|
||||
return deleted
|
||||
}
|
||||
Reference in New Issue
Block a user