v0.294.0: off-site clean-up guard follows the policy's own constants (R-867); no image clean-up while compose pulls (R-863); stderr tail (R-864); move-aside destination logged (R-869)
gates / gates (push) Successful in 28s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 07:13:02 +02:00
parent 69e914534f
commit 7861bf9dde
16 changed files with 607 additions and 36 deletions
+48 -11
View File
@@ -1,12 +1,15 @@
package stacks
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"sort"
"strings"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
)
@@ -35,6 +38,10 @@ var imageDocker = func(args ...string) (string, error) {
var imageRetentionMu sync.Mutex
// errImageRetentionBusy: a pass did not run because an image may be in use by work in flight (an update, or
// any compose command that pulls — R-863). The caller tries again later; nothing was judged.
var errImageRetentionBusy = errors.New("image work in flight")
type localImage struct {
ID, Repo, Tag, Digest, Size string
}
@@ -194,8 +201,17 @@ func (m *Manager) deleteUnkeptImages(why string, repos map[string]bool, except s
m.mu.RUnlock()
if busy != "" {
m.logger.Printf("[INFO] [stacks] image retention (%s): skipped — %s is updating (its undo may need an image nothing else names)", why, busy)
return nil, nil
return nil, errImageRetentionBusy
}
// R-863: an install, a restore or an undo inside `compose up` may have pulled an image (by digest, so
// untagged) that no container names YET. No pass while any image-pulling compose command runs; one that
// starts now waits for this pass (seconds).
endCleanup, ok := dockerexec.TryImageCleanup()
if !ok {
m.logger.Printf("[INFO] [stacks] image retention (%s): skipped — an app install, update, restore or undo is pulling images now (R-863); tried again later", why)
return nil, errImageRetentionBusy
}
defer endCleanup()
imgs, err := listLocalImages()
if err != nil {
return nil, err
@@ -293,7 +309,7 @@ func (m *Manager) RetainImagesAfterUpdate(name string, previous map[string]Insta
m.logger.Printf("[INFO] [stacks] image retention after the update of %s: the app is gone — nothing to do here", name)
return
}
if _, err := m.deleteUnkeptImages("update of "+name, appImageRepos(dir, st.AppConfig), ""); err != nil {
if _, err := m.deleteUnkeptImages("update of "+name, appImageRepos(dir, st.AppConfig), ""); err != nil && !errors.Is(err, errImageRetentionBusy) {
m.logger.Printf("[WARN] [stacks] image retention after the update of %s: %v", name, err)
}
}
@@ -317,7 +333,7 @@ func (m *Manager) RetainImagesAfterRemove(name string, repos map[string]bool) {
if len(repos) == 0 {
return
}
if _, err := m.deleteUnkeptImages("remove of "+name, repos, name); err != nil {
if _, err := m.deleteUnkeptImages("remove of "+name, repos, name); err != nil && !errors.Is(err, errImageRetentionBusy) {
m.logger.Printf("[WARN] [stacks] image retention after the remove of %s: %v", name, err)
}
}
@@ -349,28 +365,49 @@ func (m *Manager) imageRetentionMarker() string {
// RunImageRetentionOnce is the one-time clean-up at the first start of this release: the same rule, applied to every
// app image the catalog names (so the images of apps removed before this release go too). Logged; a marker file
// keeps it to once. Returns what it deleted.
func (m *Manager) RunImageRetentionOnce() []string {
// keeps it to once. Returns what it deleted, and done=false when it must be tried again (no catalog yet, or image
// work in flight — R-863: the marker is written ONLY after a pass that ran).
func (m *Manager) RunImageRetentionOnce() (deleted []string, done bool) {
if _, err := os.Stat(m.imageRetentionMarker()); err == nil {
return nil
return nil, true
}
repos := m.catalogImageRepos()
if len(repos) == 0 {
m.logger.Printf("[WARN] [stacks] image retention (one-time): no catalog read — skipped, tried again at the next start")
return nil
m.logger.Printf("[WARN] [stacks] image retention (one-time): no catalog read — skipped, tried again later")
return nil, false
}
before, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
deleted, err := m.deleteUnkeptImages("one-time clean-up", repos, "")
if errors.Is(err, errImageRetentionBusy) {
return nil, false // logged by the pass; tried again later
}
if err != nil {
m.logger.Printf("[WARN] [stacks] image retention (one-time): %v — tried again at the next start", err)
return nil
m.logger.Printf("[WARN] [stacks] image retention (one-time): %v — tried again later", err)
return nil, false
}
after, _ := imageDocker("system", "df", "--format", "{{.Type}} {{.Size}} {{.Reclaimable}}")
m.logger.Printf("[INFO] [stacks] image retention (one-time): deleted %d image(s). docker disk before: %s | after: %s",
len(deleted), strings.Join(strings.Fields(firstLine(before)), " "), strings.Join(strings.Fields(firstLine(after)), " "))
_ = os.MkdirAll(filepath.Dir(m.imageRetentionMarker()), 0o755)
_ = os.WriteFile(m.imageRetentionMarker(), []byte(fmt.Sprintf("deleted %d\n%s\n", len(deleted), strings.Join(deleted, "\n"))), 0o644)
return deleted
return deleted, true
}
// RunImageRetentionOnceUntilDone runs the one-time clean-up, and again every `every` until it has run (R-863:
// a pass that met image work in flight, or found no catalog yet, is retried — no longer only at the next start),
// at most `tries` times.
func (m *Manager) RunImageRetentionOnceUntilDone(ctx context.Context, every time.Duration, tries int) {
for i := 0; i < tries; i++ {
if _, done := m.RunImageRetentionOnce(); done {
return
}
select {
case <-ctx.Done():
return
case <-time.After(every):
}
}
m.logger.Printf("[WARN] [stacks] image retention (one-time): not run after %d tries — tried again at the next start", tries)
}
func sortedKeys(m map[string]bool) []string {