8025304acc
gates / gates (push) Successful in 12s
Update arc slices 1 and 2. NEITHER CHANGES ANY BEHAVIOUR — no new endpoint, no auto-update, the three lifecycle buttons byte-identical. Slice 1 — app.yaml gains installed_images, keyed by compose SERVICE name, each entry carrying ref + repo digest + first-seen timestamp. Written by Manager.recordInstalledImages after a successful compose up from StartStack, RestartStack, UpdateStack and runComposeDeploy. Read from the CONTAINER, never from docker-compose.yml: the syncer overwrites a deployed app's compose on a 15-minute cycle and the two disagreed for 25 minutes in the spike's own measurement. A failed write NEVER refuses the action - the deliberate opposite of SetDesiredState, because this is an observation and that is an intent. Not called from StartStackServices (the R-47 DB-only window). Its own docker seam with a context and a 30s timeout, which neither existing exec helper has. Slice 2 — .felhom.yml gains optional catalog_since; web.updateBadge compares the recorded ref per service against what the current template pins and returns a *MetaBadge through the EXISTING meta_badge partial. No new markup, no new CSS. NO RECORD RENDERS NOTHING: absent means unknown and never means current. No version number reaches the customer and no registry is queried. Known limitation, filed not hidden: 23 catalog pins float, so those apps can read Naprakesz when the image behind the tag has moved. +17 tests (1707 -> 1724), 28 packages green. Wiring proven through a real RestartStack plus an AST walk of the four call sites. Three companion red-proofs run and reverted.
439 lines
16 KiB
Go
439 lines
16 KiB
Go
package stacks
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// installedRecordTimeout bounds every docker call this file makes.
|
|
//
|
|
// REUSE.md's trap table says it in terms: composeExecCustomEnv and execCommand have NO context and
|
|
// NO timeout, so a hung docker CLI blocks forever. That is tolerable for the compose `up` a customer
|
|
// is waiting on; it is NOT tolerable here, because this runs AFTER every successful start, restart,
|
|
// update and deploy purely to write a note down. A bookkeeping read must never be able to wedge a
|
|
// lifecycle action. Three short reads share this budget generously.
|
|
const installedRecordTimeout = 30 * time.Second
|
|
|
|
// execRunner is this file's process boundary — the one seam the installed-images recorder uses.
|
|
//
|
|
// It is deliberately its OWN seam rather than Manager.execFn / composeExecCustomEnv: those two are
|
|
// already load-bearing for refreshStatusLocked and for the compose lifecycle, and both lack the
|
|
// context this code needs. dir "" means "do not chdir"; env nil means inherit os.Environ().
|
|
// nil in production (defaultExecRunner); tests script argv → output and never touch docker.
|
|
type execRunner func(ctx context.Context, dir string, env []string, name string, args ...string) (string, error)
|
|
|
|
func defaultExecRunner(ctx context.Context, dir string, env []string, name string, args ...string) (string, error) {
|
|
cmd := exec.CommandContext(ctx, name, args...)
|
|
if dir != "" {
|
|
cmd.Dir = dir
|
|
}
|
|
if env != nil {
|
|
cmd.Env = env
|
|
}
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
stderr := ""
|
|
if ee, ok := err.(*exec.ExitError); ok {
|
|
stderr = truncateStr(string(ee.Stderr), 500)
|
|
}
|
|
return string(out), fmt.Errorf("exec %s %s: %w\nstderr: %s", name, strings.Join(args, " "), err, stderr)
|
|
}
|
|
return string(out), nil
|
|
}
|
|
|
|
func (m *Manager) runInstalled(ctx context.Context, dir string, env []string, name string, args ...string) (string, error) {
|
|
if m.installedExecFn != nil {
|
|
return m.installedExecFn(ctx, dir, env, name, args...)
|
|
}
|
|
return defaultExecRunner(ctx, dir, env, name, args...)
|
|
}
|
|
|
|
// composeArgv splits the configured compose command into an argv prefix, mirroring
|
|
// composeExecCustomEnv's own docker-compose-v1-vs-v2 branch. One rule, two callers.
|
|
func (m *Manager) composeArgv(args ...string) (string, []string) {
|
|
if m.composeCmd == "docker compose" {
|
|
return "docker", append([]string{"compose"}, args...)
|
|
}
|
|
return "docker-compose", args
|
|
}
|
|
|
|
// --- The template side: what the compose FILE currently pins, per service ---
|
|
|
|
// composeImagesDoc is the minimal view of a compose file needed here.
|
|
//
|
|
// A REAL YAML parse and never a line scan, for the reason dbservices.go's composeServicesDoc already
|
|
// records: immich's top-level `immich_ml_cache:` volume key has exactly the shape a naive scan
|
|
// misreads as a service. Manager.checkLocalImages IS such a line scan and is deliberately not reused
|
|
// — it also cannot say which service an image belongs to, which is the whole comparison.
|
|
type composeImagesDoc struct {
|
|
Services map[string]struct {
|
|
Image string `yaml:"image"`
|
|
} `yaml:"services"`
|
|
}
|
|
|
|
// ParseComposeImages returns compose SERVICE name -> the image reference the file pins for it.
|
|
//
|
|
// Services with no `image:` (a `build:`-only service — none in the catalog today) are omitted rather
|
|
// than recorded as an empty pin, because "" would compare equal to nothing useful. An unreadable or
|
|
// unparseable file returns an error: CANNOT-TELL must never read as "no images", or an app whose
|
|
// compose file is briefly mid-write would render as up to date.
|
|
func ParseComposeImages(composePath string) (map[string]string, error) {
|
|
data, err := os.ReadFile(composePath)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("reading compose file: %w", err)
|
|
}
|
|
var doc composeImagesDoc
|
|
if err := yaml.Unmarshal(data, &doc); err != nil {
|
|
return nil, fmt.Errorf("parsing compose file %s: %w", composePath, err)
|
|
}
|
|
out := make(map[string]string, len(doc.Services))
|
|
for svc, def := range doc.Services {
|
|
if strings.TrimSpace(def.Image) == "" {
|
|
continue
|
|
}
|
|
out[svc] = strings.TrimSpace(def.Image)
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// --- The container side: what is ACTUALLY running ---
|
|
|
|
// composePSEntry is the subset of `docker compose ps --format json` this needs.
|
|
type composePSEntry struct {
|
|
ID string `json:"ID"`
|
|
Name string `json:"Name"`
|
|
Service string `json:"Service"`
|
|
}
|
|
|
|
// parseComposePS tolerates BOTH shapes compose v2 has emitted for `ps --format json`: a single JSON
|
|
// array (compose < 2.21) and newline-delimited objects (2.21+). Neither shape is guessed at from a
|
|
// version string — the output is tried as an array first and falls back to per-line objects, so an
|
|
// upgrade of the docker CLI underneath a customer's box cannot silently stop the recording.
|
|
func parseComposePS(out string) ([]composePSEntry, error) {
|
|
trimmed := strings.TrimSpace(out)
|
|
if trimmed == "" {
|
|
return nil, nil
|
|
}
|
|
if strings.HasPrefix(trimmed, "[") {
|
|
var arr []composePSEntry
|
|
if err := json.Unmarshal([]byte(trimmed), &arr); err != nil {
|
|
return nil, fmt.Errorf("parsing compose ps JSON array: %w", err)
|
|
}
|
|
return arr, nil
|
|
}
|
|
var entries []composePSEntry
|
|
for _, line := range strings.Split(trimmed, "\n") {
|
|
line = strings.TrimSpace(line)
|
|
if line == "" {
|
|
continue
|
|
}
|
|
var e composePSEntry
|
|
if err := json.Unmarshal([]byte(line), &e); err != nil {
|
|
return nil, fmt.Errorf("parsing compose ps JSON line: %w", err)
|
|
}
|
|
entries = append(entries, e)
|
|
}
|
|
return entries, nil
|
|
}
|
|
|
|
// containerFacts is one container's two identifiers as docker reports them.
|
|
type containerFacts struct {
|
|
ref string // .Config.Image — the reference the container was CREATED FROM
|
|
imageID string // .Image — the local image id it actually resolved to
|
|
}
|
|
|
|
const inspectSep = "\x1f" // ASCII unit separator: cannot occur in an image ref or an id
|
|
|
|
// observeInstalledImages reads what every compose service of this stack is running.
|
|
//
|
|
// Three short docker reads, in order: which containers belong to which service; what reference and
|
|
// image id each container carries; and what repo digest each of those images has. Nothing is read
|
|
// from docker-compose.yml — that file is the value the syncer has already moved.
|
|
func (m *Manager) observeInstalledImages(stackDir string, env []string) (map[string]InstalledImage, error) {
|
|
ctx, cancel := context.WithTimeout(context.Background(), installedRecordTimeout)
|
|
defer cancel()
|
|
|
|
bin, argv := m.composeArgv("ps", "-a", "--format", "json")
|
|
psOut, err := m.runInstalled(ctx, stackDir, env, bin, argv...)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("listing containers: %w", err)
|
|
}
|
|
entries, err := parseComposePS(psOut)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
// service -> container id, deterministic when a service has replicas (none in the catalog, but
|
|
// a compose `deploy.replicas` would produce several; take the first by name so two runs agree).
|
|
sort.Slice(entries, func(i, j int) bool { return entries[i].Name < entries[j].Name })
|
|
svcContainer := make(map[string]string)
|
|
var ids []string
|
|
for _, e := range entries {
|
|
if e.Service == "" || e.ID == "" {
|
|
continue
|
|
}
|
|
if _, seen := svcContainer[e.Service]; seen {
|
|
continue
|
|
}
|
|
svcContainer[e.Service] = e.ID
|
|
ids = append(ids, e.ID)
|
|
}
|
|
if len(ids) == 0 {
|
|
return map[string]InstalledImage{}, nil
|
|
}
|
|
|
|
facts, err := m.inspectContainers(ctx, ids)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
digests, err := m.inspectImageDigests(ctx, facts)
|
|
if err != nil {
|
|
// A missing digest is a recorded empty string, never a failed recording — see the edge-case
|
|
// table. Log and carry on with refs only.
|
|
m.logger.Printf("[WARN] [stacks] installed-images: reading repo digests failed, recording refs only: %v", err)
|
|
digests = map[string]string{}
|
|
}
|
|
|
|
now := time.Now().UTC().Format(time.RFC3339)
|
|
out := make(map[string]InstalledImage, len(svcContainer))
|
|
for svc, id := range svcContainer {
|
|
f, ok := facts[id]
|
|
if !ok {
|
|
continue
|
|
}
|
|
out[svc] = InstalledImage{
|
|
Ref: f.ref,
|
|
Digest: pickDigest(f.ref, digests[f.imageID]),
|
|
At: now,
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func (m *Manager) inspectContainers(ctx context.Context, ids []string) (map[string]containerFacts, error) {
|
|
args := append([]string{"inspect", "--type", "container",
|
|
"--format", "{{.Id}}" + inspectSep + "{{.Config.Image}}" + inspectSep + "{{.Image}}"}, ids...)
|
|
out, err := m.runInstalled(ctx, "", nil, "docker", args...)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("inspecting containers: %w", err)
|
|
}
|
|
facts := make(map[string]containerFacts, len(ids))
|
|
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
|
|
parts := strings.Split(strings.TrimSpace(line), inspectSep)
|
|
if len(parts) != 3 {
|
|
continue
|
|
}
|
|
facts[parts[0]] = containerFacts{ref: parts[1], imageID: parts[2]}
|
|
// docker accepts short ids on the way in and returns full ones on the way out; key both so
|
|
// the caller's compose-ps id (12 hex) finds its row.
|
|
if len(parts[0]) > 12 {
|
|
facts[parts[0][:12]] = containerFacts{ref: parts[1], imageID: parts[2]}
|
|
}
|
|
}
|
|
return facts, nil
|
|
}
|
|
|
|
// inspectImageDigests maps image id -> its RepoDigests, joined by a space. Empty when an image has
|
|
// none (built or imported locally, never pulled) — recorded as an empty digest, not as a failure.
|
|
func (m *Manager) inspectImageDigests(ctx context.Context, facts map[string]containerFacts) (map[string]string, error) {
|
|
seen := map[string]bool{}
|
|
var imgs []string
|
|
for _, f := range facts {
|
|
if f.imageID != "" && !seen[f.imageID] {
|
|
seen[f.imageID] = true
|
|
imgs = append(imgs, f.imageID)
|
|
}
|
|
}
|
|
if len(imgs) == 0 {
|
|
return map[string]string{}, nil
|
|
}
|
|
sort.Strings(imgs)
|
|
args := append([]string{"image", "inspect",
|
|
"--format", "{{.Id}}" + inspectSep + "{{range $i, $d := .RepoDigests}}{{if $i}} {{end}}{{$d}}{{end}}"}, imgs...)
|
|
out, err := m.runInstalled(ctx, "", nil, "docker", args...)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("inspecting images: %w", err)
|
|
}
|
|
digests := make(map[string]string, len(imgs))
|
|
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
|
|
parts := strings.SplitN(strings.TrimSpace(line), inspectSep, 2)
|
|
if len(parts) != 2 {
|
|
continue
|
|
}
|
|
digests[parts[0]] = strings.TrimSpace(parts[1])
|
|
}
|
|
return digests, nil
|
|
}
|
|
|
|
// pickDigest turns a RepoDigests list into the ONE sha256 that belongs to the ref we asked for.
|
|
//
|
|
// A local image can carry several repo digests (the same bytes tagged from two registries), and
|
|
// picking the wrong one would record a digest for a repository this app never used. Match on the
|
|
// repository part of the ref first; fall back to the single entry when there is exactly one; give up
|
|
// (empty) rather than guess between several unrelated ones.
|
|
func pickDigest(ref, repoDigests string) string {
|
|
fields := strings.Fields(repoDigests)
|
|
if len(fields) == 0 {
|
|
return ""
|
|
}
|
|
repo := refRepository(ref)
|
|
for _, rd := range fields {
|
|
at := strings.LastIndex(rd, "@")
|
|
if at < 0 {
|
|
continue
|
|
}
|
|
if repo != "" && rd[:at] == repo {
|
|
return rd[at+1:]
|
|
}
|
|
}
|
|
if len(fields) == 1 {
|
|
if at := strings.LastIndex(fields[0], "@"); at >= 0 {
|
|
return fields[0][at+1:]
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// refRepository strips the tag and/or digest from an image reference, leaving the repository.
|
|
// Careful with a registry port ("registry:5000/app:1.2"): only a colon AFTER the last slash is a tag.
|
|
func refRepository(ref string) string {
|
|
if at := strings.LastIndex(ref, "@"); at >= 0 {
|
|
ref = ref[:at]
|
|
}
|
|
slash := strings.LastIndex(ref, "/")
|
|
if colon := strings.LastIndex(ref, ":"); colon > slash {
|
|
ref = ref[:colon]
|
|
}
|
|
return ref
|
|
}
|
|
|
|
// --- The write ---
|
|
|
|
// sameInstalled compares two records on Ref and Digest ONLY, deliberately ignoring At.
|
|
// Including At would make every restart a change, and app.yaml would be rewritten — with its
|
|
// encrypted secrets — on every lifecycle action for no new information.
|
|
func sameInstalled(a, b map[string]InstalledImage) bool {
|
|
if len(a) != len(b) {
|
|
return false
|
|
}
|
|
for k, va := range a {
|
|
vb, ok := b[k]
|
|
if !ok || va.Ref != vb.Ref || va.Digest != vb.Digest {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
// recordInstalledImages writes what this stack is ACTUALLY running into its app.yaml.
|
|
//
|
|
// ── WHY A FAILURE HERE NEVER REFUSES THE ACTION ──────────────────────────────────────────────
|
|
//
|
|
// This is deliberately the OPPOSITE of SetDesiredState, and the difference is what the field means.
|
|
// `desired_state` is the customer's INTENT: performing an act whose intent could not be recorded
|
|
// recreates exactly the ambiguity R-166 closed, so a failed write there correctly refuses the act.
|
|
// `installed_images` is an OBSERVATION. Refusing to start a customer's app because we could not write
|
|
// down which version it is would trade a real outage for a bookkeeping gap. So: log at ERROR, loudly,
|
|
// naming the app — and return. The app stays up.
|
|
//
|
|
// Called after a SUCCESSFUL compose up from StartStack, RestartStack, UpdateStack and
|
|
// runComposeDeploy. NOT from StartStackServices: that path starts only the database service for the
|
|
// R-47 restore window, and recording a partial stack there would overwrite a complete record with an
|
|
// incomplete one.
|
|
func (m *Manager) recordInstalledImages(name, stackDir string, env []string) {
|
|
cfg := LoadAppConfig(stackDir)
|
|
if cfg == nil {
|
|
// No app.yaml: an infra/protected stack, or nothing deployed here. Nothing to record on.
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] installed-images %s: no app.yaml — nothing to record", name)
|
|
}
|
|
return
|
|
}
|
|
|
|
observed, err := m.observeInstalledImages(stackDir, env)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] installed-images %s: could not observe running images (the app is unaffected): %v", name, err)
|
|
return
|
|
}
|
|
if len(observed) == 0 {
|
|
m.logger.Printf("[WARN] [stacks] installed-images %s: no containers observed — nothing recorded, previous record left intact", name)
|
|
return
|
|
}
|
|
|
|
// A partial read is recorded AND said out loud, never written silently: a record that quietly
|
|
// lost a service would read as a complete answer to "what is this app running".
|
|
if tpl, terr := ParseComposeImages(filepath.Join(stackDir, "docker-compose.yml")); terr == nil && len(tpl) > len(observed) {
|
|
var missing []string
|
|
for svc := range tpl {
|
|
if _, ok := observed[svc]; !ok {
|
|
missing = append(missing, svc)
|
|
}
|
|
}
|
|
sort.Strings(missing)
|
|
m.logger.Printf("[WARN] [stacks] installed-images %s: recorded %d of %d compose service(s) — not observed: %s",
|
|
name, len(observed), len(tpl), strings.Join(missing, ", "))
|
|
}
|
|
|
|
// Carry forward the first-seen timestamp of every entry whose ref+digest is unchanged, so `at`
|
|
// answers "running since" rather than "last looked at".
|
|
for svc, prev := range cfg.InstalledImages {
|
|
if cur, ok := observed[svc]; ok && cur.Ref == prev.Ref && cur.Digest == prev.Digest && prev.At != "" {
|
|
cur.At = prev.At
|
|
observed[svc] = cur
|
|
}
|
|
}
|
|
|
|
if sameInstalled(cfg.InstalledImages, observed) {
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] installed-images %s: unchanged (%d service(s)) — app.yaml not rewritten", name, len(observed))
|
|
}
|
|
return
|
|
}
|
|
|
|
cfg.InstalledImages = observed
|
|
meta := LoadMetadata(stackDir)
|
|
if err := SaveAppConfig(stackDir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] installed-images %s: recording failed, the app is running and unaffected: %v", name, err)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] installed-images %s: recorded %d service(s) (%s)", name, len(observed), summariseInstalled(observed))
|
|
|
|
// Keep the in-memory view in step so the badge does not lag a full ScanStacks behind the file.
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok && s.AppConfig != nil {
|
|
s.AppConfig.InstalledImages = observed
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
// summariseInstalled renders the record for ONE log line. Image refs and digests only — this file
|
|
// never logs anything out of app.yaml's env map, which holds encrypted secrets.
|
|
func summariseInstalled(m map[string]InstalledImage) string {
|
|
svcs := make([]string, 0, len(m))
|
|
for svc := range m {
|
|
svcs = append(svcs, svc)
|
|
}
|
|
sort.Strings(svcs)
|
|
parts := make([]string, 0, len(svcs))
|
|
for _, svc := range svcs {
|
|
d := m[svc].Digest
|
|
if len(d) > 19 {
|
|
d = d[:19] + "…"
|
|
}
|
|
if d == "" {
|
|
d = "no digest"
|
|
}
|
|
parts = append(parts, fmt.Sprintf("%s=%s (%s)", svc, m[svc].Ref, d))
|
|
}
|
|
return strings.Join(parts, ", ")
|
|
}
|