v0.233.0: record what each compose service actually installed, and badge whether it is current
gates / gates (push) Successful in 12s

Update arc slices 1 and 2. NEITHER CHANGES ANY BEHAVIOUR — no new endpoint, no
auto-update, the three lifecycle buttons byte-identical.

Slice 1 — app.yaml gains installed_images, keyed by compose SERVICE name, each
entry carrying ref + repo digest + first-seen timestamp. Written by
Manager.recordInstalledImages after a successful compose up from StartStack,
RestartStack, UpdateStack and runComposeDeploy. Read from the CONTAINER, never
from docker-compose.yml: the syncer overwrites a deployed app's compose on a
15-minute cycle and the two disagreed for 25 minutes in the spike's own
measurement. A failed write NEVER refuses the action - the deliberate opposite
of SetDesiredState, because this is an observation and that is an intent. Not
called from StartStackServices (the R-47 DB-only window). Its own docker seam
with a context and a 30s timeout, which neither existing exec helper has.

Slice 2 — .felhom.yml gains optional catalog_since; web.updateBadge compares the
recorded ref per service against what the current template pins and returns a
*MetaBadge through the EXISTING meta_badge partial. No new markup, no new CSS.
NO RECORD RENDERS NOTHING: absent means unknown and never means current. No
version number reaches the customer and no registry is queried.

Known limitation, filed not hidden: 23 catalog pins float, so those apps can read
Naprakesz when the image behind the tag has moved.

+17 tests (1707 -> 1724), 28 packages green. Wiring proven through a real
RestartStack plus an AST walk of the four call sites. Three companion red-proofs
run and reverted.
This commit is contained in:
2026-09-02 20:18:01 +02:00
parent 960d29b061
commit 8025304acc
14 changed files with 1637 additions and 9 deletions
+37 -7
View File
@@ -150,6 +150,13 @@ type Stack struct {
// forgetting costs at most one threshold window, whereas persisting could carry a stale
// "this app is crash-looping" verdict across the restart that fixed it.
RestartingSince time.Time `json:"restarting_since,omitempty"`
// TemplateImages is what the stack's CURRENT docker-compose.yml pins, per compose service —
// i.e. what the catalog says this app should be running right now. Refreshed by ScanStacks for
// deployed, non-protected apps only; nil for everything else and nil when the file cannot be
// parsed. Nil means CANNOT-TELL and never means "matches": web.updateBadge renders nothing.
// Not persisted — it is a read of a file the syncer owns, and re-reading is cheaper than a
// second copy that can go stale.
TemplateImages map[string]string `json:"template_images,omitempty"`
}
// Manager handles all docker compose stack operations.
@@ -200,6 +207,11 @@ type Manager struct {
// Debug dump network section); nil in production. One seam for all guest-net reads — tests
// script canned `ip`/resolv.conf outputs per argv and never touch docker.
guestNetExecFn func(args ...string) (string, error)
// installedExecFn is the installed-images recorder's OWN process boundary (installed.go); nil in
// production (defaultExecRunner). Separate from execFn deliberately: this one carries a context
// and a timeout, which execFn/composeExecCustomEnv do not, and a bookkeeping read must never be
// able to wedge a lifecycle action. Tests script argv -> output and never touch docker.
installedExecFn execRunner
}
// SetSambaRunProbe injects the samba liveness probe. Exported for the same reason
@@ -502,6 +514,19 @@ func (m *Manager) ScanStacks() error {
m.logger.Printf("[DEBUG] [stacks] ScanStacks: found stack %q deployed=%v composePath=%s", name, deployed, composePath)
}
// What the CURRENT template pins, for the update badge. Deployed non-protected apps only:
// an undeployed template has nothing to compare against, and infra stacks are not the
// customer's to update. A parse failure leaves this nil, which reads as CANNOT-TELL.
var tplImages map[string]string
if deployed && !m.cfg.IsProtectedStack(name) {
imgs, ierr := ParseComposeImages(composePath)
if ierr != nil {
m.logger.Printf("[WARN] [stacks] ScanStacks: cannot read image pins from %s: %v", composePath, ierr)
} else {
tplImages = imgs
}
}
if existing, ok := m.stacks[name]; ok {
existing.ComposePath = composePath
existing.Meta = meta
@@ -511,16 +536,18 @@ func (m *Manager) ScanStacks() error {
if !existing.Deploying {
existing.Deployed = deployed
existing.AppConfig = appCfg
existing.TemplateImages = tplImages
}
} else {
m.stacks[name] = &Stack{
Name: name,
Meta: meta,
ComposePath: composePath,
State: StateNotDeployed,
Deployed: deployed,
Protected: m.cfg.IsProtectedStack(name),
AppConfig: appCfg,
Name: name,
Meta: meta,
ComposePath: composePath,
State: StateNotDeployed,
Deployed: deployed,
Protected: m.cfg.IsProtectedStack(name),
AppConfig: appCfg,
TemplateImages: tplImages,
}
}
}
@@ -1052,6 +1079,7 @@ func (m *Manager) StartStack(name string) error {
}
m.logger.Printf("[INFO] [stacks] Stack %s started successfully (took %.1fs)", name, time.Since(start).Seconds())
m.recordInstalledImages(name, dir, env)
m.logPostStartStatus(name, dir, env)
// Clear stale health probe so refreshStatus won't re-apply an old unhealthy override.
@@ -1155,6 +1183,7 @@ func (m *Manager) RestartStack(name string) error {
}
m.logger.Printf("[INFO] [stacks] Stack %s restarted successfully (took %.1fs)", name, time.Since(start).Seconds())
m.recordInstalledImages(name, dir, env)
m.logPostStartStatus(name, dir, env)
// Clear stale health probe so refreshStatus won't re-apply an old unhealthy override.
@@ -1193,6 +1222,7 @@ func (m *Manager) UpdateStack(name string) error {
}
m.logger.Printf("[INFO] [stacks] Stack %s updated successfully (took %.1fs)", name, time.Since(start).Seconds())
m.recordInstalledImages(name, dir, env)
m.logPostStartStatus(name, dir, env)
return m.RefreshStatus()
}