Files
felhom-controller/controller/internal/stacks/manager.go
T
admin b53721db34 R-772: a health probe that could not run records not_checked + NOT healthy, is looked at again on the 10 s cycle, and the app page says so
The state stays the containers' (probeSaysUnhealthy skips a not-checked record), so R-630 holds.
Red-proof RP-D1 — felhom.eu audits/visitors-2026-10-01/D.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-10-01 21:01:42 +02:00

1740 lines
70 KiB
Go

package stacks
import (
"bytes"
"context"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
"log"
"os"
"os/exec"
"path/filepath"
"sort"
"strconv"
"strings"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
"gitea.dooplex.hu/admin/felhom-controller/internal/crypto"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
)
// ContainerState represents the current state of a container.
type ContainerState string
const (
StateRunning ContainerState = "running"
StateStarting ContainerState = "starting" // running but health: starting
StateUnhealthy ContainerState = "unhealthy" // running but health: unhealthy
StateStopped ContainerState = "stopped"
StateDegraded ContainerState = "degraded" // multi-container stack: a SUPERVISED member is dead (R-51)
StateRestarting ContainerState = "restarting"
StateExited ContainerState = "exited"
StatePaused ContainerState = "paused"
StateUnknown ContainerState = "unknown"
StateNotDeployed ContainerState = "not_deployed"
StateDeploying ContainerState = "deploying" // compose up in progress (image pull, etc.)
StateOrphaned ContainerState = "orphaned"
)
// IsDownState reports whether a container state means a DEPLOYED app is not running and won't recover
// on its own (fix-3, CAMPAIGN-3). Only `stopped`, `exited` and `degraded` qualify — a Docker
// "created"/"dead" container (a failed-at-boot app, the F11 case) resolves to `stopped`. Deliberately
// NOT `starting` / `unhealthy` (running, with their own health handling), `restarting`
// (self-recovering), `deploying` (mid-deploy), `paused` (a deliberate user action), or `unknown`
// (ambiguous — fail-open, never manufacture a dead-app alert from an inconclusive read).
//
// R-51 (v0.156.0) added `degraded`: a multi-container stack whose SUPERVISED member is dead is as
// unreachable as a single-container app that exited (immich-server sat Exited for 18 h with the app
// 100 % dead and no alert, while single-container Calibre-Web alerted in 90 s). This is deliberately
// NOT the same as folding `unhealthy` into down — that exclusion stays byte-identical, because
// `unhealthy` is a *running* container whose healthcheck is failing and folding it in reintroduces
// the flapping fix-3 was added to stop.
func IsDownState(s ContainerState) bool {
return s == StateStopped || s == StateExited || s == StateDegraded
}
// C9-F2 — a SUSTAINED `restarting` is a crash loop, and a crash loop is a dead app.
//
// THE BUG THIS EXISTS TO KILL. `IsDownState` above excludes `restarting` as "self-recovering", and
// for a brief restart that is exactly right. But Docker sets `restarting` while a container is being
// restarted BY POLICY, and for the catalog's standard `restart: unless-stopped` that is precisely the
// crash-loop signal — the retry count is unlimited, so "self-recovering" is a promise Docker never
// made. Campaign 9 watched docmost loop for nine minutes (restartcount 18, policy `unless-stopped`)
// while the F-OBS heartbeat printed "180 scans since boot, 4 deployed app(s) evaluated, 0 currently
// down". No banner, no app_start_failed, no email, no hub event — indefinitely.
//
// This is CONTEXT.md's own lesson one state over: "Docker's .State says 'running' even for unhealthy
// containers — must parse .Status". Same trap, different state, and this state means something worse.
//
// ── WHY A THRESHOLD AND NOT A DOWN-STATE ─────────────────────────────────────────────────────
//
// Adding StateRestarting to IsDownState would alarm on every deploy and every update, fleet-wide,
// because the normal `docker compose up -d` path passes through `restarting`. An alarm that fires on
// routine operations is one the operator learns to ignore — which is what F-A1 nearly cost us right
// after R-97a built it. So `restarting` becomes down only once it has PERSISTED.
//
// ── WHERE 5 MINUTES COMES FROM ───────────────────────────────────────────────────────────────
//
// Measured against the three real numbers already in this codebase, not picked round:
// - the deploy flow allows **120 s** for a stack to come up healthy — the project's own existing
// answer to "how long is too long"; an app still restarting past it has failed deployment;
// - the slowest catalog healthcheck start_period is Mealie's **60 s**, after which a couple of
// check intervals must still elapse before any verdict is meaningful;
// - R-97b's quiesce grace is **180 s**, and this must sit ABOVE it so the two windows compose into
// one bounded delay rather than a gap where an app is un-suppressed but not yet sustained.
//
// 300 s clears all three with margin. It is also unambiguous against Docker's own backoff, which
// grows 100 ms → 200 ms → … and caps at 60 s: a genuine crash loop registers at least four restart
// attempts inside this window, so a stack that is still `restarting` at 5 minutes is not mid-deploy.
//
// The cost is a bounded DELAY in reporting a real crash loop, never its loss — the same trade R-97b
// made deliberately, and the opposite of the indefinite silence this replaces.
const crashLoopAfter = 5 * time.Minute
// CrashLooping reports whether the stack has been `restarting` for longer than crashLoopAfter.
// `now` is injected so the rule is a unit-testable contract rather than a property of the clock.
// A zero RestartingSince means "not restarting, or not yet observed restarting" — never a crash loop.
func (s *Stack) CrashLooping(now time.Time) bool {
if s == nil || s.State != StateRestarting || s.RestartingSince.IsZero() {
return false
}
return now.Sub(s.RestartingSince) >= crashLoopAfter
}
// ContainerInfo holds status info about a single container within a stack.
type ContainerInfo struct {
Name string `json:"name"`
Image string `json:"image"`
State ContainerState `json:"state"`
Status string `json:"status"` // e.g. "Up 3 hours (healthy)"
}
// HealthProbeResult holds the latest controller-side health probe result.
// probeSaysUnhealthy: a probe that RAN and failed overrides a running state to unhealthy; a not-checked record (R-772)
// does not — the containers decide. Pinned by TestProbeSaysUnhealthy_NotCheckedLeavesTheStateAlone.
func probeSaysUnhealthy(hp *HealthProbeResult) bool {
return hp != nil && !hp.Healthy && !hp.NotChecked
}
type HealthProbeResult struct {
Healthy bool `json:"healthy"`
LastCheck time.Time `json:"last_check"`
Details []HealthCheckDetail `json:"details"`
// NotChecked: the declared check could not run — no container to probe (R-772). Healthy is then FALSE: a record of a
// check that did not happen never says healthy (presence is not success). The stack's STATE is still read from its
// containers (manager.go's override skips a not-checked record), so a running app is never painted unhealthy
// because the box could not look. Pinned by TestRunHealthProbes_NoContainerIsNotHealthy.
NotChecked bool `json:"not_checked,omitempty"`
}
// HealthCheckDetail holds the result of a single health check item.
type HealthCheckDetail struct {
Type string `json:"type"` // "http", "api", "tcp"
Target string `json:"target"` // e.g. ":3456/api/v1/info"
Healthy bool `json:"healthy"`
Status int `json:"status,omitempty"` // HTTP status code (for http/api)
Latency string `json:"latency"` // e.g. "45ms"
Error string `json:"error,omitempty"` // error message if unhealthy
// MessageKey is the i18n bundle key for Error when the text is a PRODUCT sentence rather than a
// transport error. It follows util.MsgError's contract: Error carries the Hungarian bytes (the
// default language), MessageKey lets a localising consumer render the other one. Added for
// R-630, whose "no health check ran" detail is the first such sentence here.
MessageKey string `json:"message_key,omitempty"`
}
// Stack represents a docker compose stack on disk.
type Stack struct {
Name string `json:"name"`
Meta Metadata `json:"meta"`
ComposePath string `json:"compose_path"`
State ContainerState `json:"state"`
Deployed bool `json:"deployed"` // Has app.yaml with deployed=true
Protected bool `json:"protected"`
Orphaned bool `json:"orphaned"` // Deployed but no catalog template
Containers []ContainerInfo `json:"containers"`
AppConfig *AppConfig `json:"app_config,omitempty"`
Deploying bool `json:"deploying"` // compose up in progress
DeployError string `json:"deploy_error,omitempty"` // last async deploy error
// InstallInterrupted (R-681, v0.270.0): the last install was cut off by a controller restart; the
// page says so until the next install (read from the stack dir's marker at every scan).
InstallInterrupted bool `json:"install_interrupted,omitempty"`
// Updating / UpdatePhase / UpdatePhaseLabel / UpdateError (update arc slice 4, v0.237.0) are the
// guarded update's in-memory progress, the same shape as Deploying/DeployError: the API answers
// 202 at once and the page polls GET /api/stacks/{name}. See update.go.
Updating bool `json:"updating"`
UpdatePhase string `json:"update_phase,omitempty"`
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
UpdateError string `json:"update_error,omitempty"`
// UpdateErrorKey / UpdateErrorArgs (v0.264.0, R-606) are UpdateError as a bundle key + arguments,
// so a page renders it in the READER's language (UpdateErrorIn). UpdateError stays the Hungarian —
// what the API, the logs and older readers have always had. Empty key = an old or composed sentence,
// rendered as stored, never as an empty line.
UpdateErrorKey string `json:"-"`
UpdateErrorArgs []interface{} `json:"-"`
// updateHeld (R-480, v0.240.0) — the last update ended HELD, so UpdateError is the hold's own
// sentence („… leállítva marad"). It is shown only while that hold is in force; fillHoldReason.
updateHeld bool
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
HoldReason string `json:"hold_reason,omitempty"`
// HoldNoWholeCopy (v0.268.0, R-659): the hold names NO copy — none on this box brings the app back
// whole — so the page offers no restore button beside the sentence (the restore would refuse).
HoldNoWholeCopy bool `json:"hold_no_whole_copy,omitempty"`
// HoldKind (v0.269.0): which hold — "update_failed", "unhealthy_stop" (decision 28: the page offers
// Start, not a restore), "restore"; "" when none.
HoldKind string `json:"hold_kind,omitempty"`
HealthProbe *HealthProbeResult `json:"health_probe,omitempty"` // controller-side probe result
LastUpdated time.Time `json:"last_updated"`
// RestartingSince (C9-F2) is when this stack was FIRST observed in StateRestarting during the
// current restarting run; zero whenever the stack is in any other state. It is what turns a brief
// restart (normal: deploy, update, quiesce restart) into a distinguishable crash loop — see
// CrashLooping. Not persisted: a controller restart re-observes the state within one refresh, and
// forgetting costs at most one threshold window, whereas persisting could carry a stale
// "this app is crash-looping" verdict across the restart that fixed it.
RestartingSince time.Time `json:"restarting_since,omitempty"`
// TemplateImages is what the stack's LIVE docker-compose.yml pins, per compose service.
//
// ⚠ SINCE v0.235.0 THIS IS NOT "WHAT THE CATALOG OFFERS". The live file is RENDERED: for a
// pinned app whose version the catalog has moved past, it is the app's own frozen definition.
// So TemplateImages answers "what will the next `compose up -d` bring this app to" — which is
// exactly what a debugger wants and exactly the WRONG input for the update badge, because a
// frozen app's live file names the OLD version and the comparison would read „Naprakész".
// THE BADGE USES CatalogImages. See web.compareInstalledToTemplate.
//
// Refreshed by ScanStacks for deployed, non-protected apps only; nil otherwise and nil when the
// file cannot be parsed. Not persisted.
TemplateImages map[string]string `json:"template_images,omitempty"`
// CatalogImages is what the CATALOG currently offers for this app, read from the syncer's git
// clone (`<DataDir>/catalog-cache/templates/<app>/docker-compose.yml`) rather than from the
// stack dir. Added in v0.235.0 because the render made the live file unusable for the question
// "is this app behind?".
//
// Nil means CANNOT-TELL — the cache is missing, unreadable, or the app is not in the catalog —
// and the badge then renders NOTHING. Absent is unknown; it is never „Naprakész".
CatalogImages map[string]string `json:"catalog_images,omitempty"`
// LadderStepsLeft (v0.268.0, `09` §3 decision 14) — how many tested steps lie between the app's
// pin and the catalog's head; one press climbs one. 0 = unknown or none. The page shows it while
// the badge says „Frissítés elérhető".
LadderStepsLeft int `json:"ladder_steps_left,omitempty"`
// CatalogDigests / CatalogTestedAt (v0.269.0, `09` §6.4 part 6): the TESTED digest per service of the
// catalog's current refs and when that test ran — the badge's input for a floating tag (never a
// registry query). Empty when the ladder has no entry for them.
CatalogDigests map[string]string `json:"catalog_digests,omitempty"`
CatalogTestedAt time.Time `json:"catalog_tested_at,omitempty"`
}
// Manager handles all docker compose stack operations.
type Manager struct {
// lastVerifyMeta (v0.269.0) is the .felhom.yml the last update verify judged by — read by a test.
lastVerifyMeta string
// updateHealthMetaFn (test seam, v0.269.0): the verify with the chosen .felhom.yml; nil → the real wait.
updateHealthMetaFn func(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string)
// unhealthy (v0.269.0, decision 28): RestartCount / OOM-kill samples per app for the crash-loop stop.
unhealthy unhealthyWatch
// leg (v0.271.0, `09` §6.4 part 7): the automatic update leg's state (unattended.go).
leg updateLegState
// selfUpdating (v0.261.0) reports whether the CONTROLLER is swapping itself. Set by
// SetSelfUpdatingCheck; nil means no gate. See update.go.
selfUpdating func() bool
cfg *config.Config
logger *log.Logger
composeCmd string
stacks map[string]*Stack
mu sync.RWMutex
encKey []byte // AES-256 key for encrypting sensitive values in app.yaml
infraMu sync.Mutex // single-flight guard for EnsureBaseStack (base-infra bring-up/self-heal)
// Migration engine (B1): single-flight + live job + deps wired via SetMigrationDeps.
migrateMu sync.Mutex
migrating bool
migJob *MigrationJob
settings *settings.Settings
sysDataPath string
backupRunning func() bool // mutual exclusion with the backup orchestrator (Change 3)
migDoneHook func(*MigrationJob) // fired on successful completion (decommission policy lives in caller)
// deployDoneHook (R-536) fires when an ASYNC deploy reaches its end — ok=true with the observed
// state, or ok=false with the reason. The hub event that says „Alkalmazás telepítve" hangs off
// this and nothing else: it used to be sent beside the 202 that merely accepted the request, so
// an install interrupted five seconds later stayed on the timeline as a completed one.
deployDoneHook func(name string, ok bool, detail string)
testSeams *migSeams // nil in production; tests inject fakes
// R-51: docker restart policies for DOWN members of mixed stacks. Keyed by
// containerName+"|"+state so a transitioned or recreated container re-reads rather than
// answering from a stale entry; pruned every refresh to the live container set. Guarded by mu
// (every read/write happens under refreshStatusLocked's write lock).
restartPolicyCache map[string]string
// R-514: last OOM scan (oom.go), for the dashboard.
oomMu sync.Mutex
oomCache map[string][]string
// R-513: FileBrowser admin-password seams (filebrowser_password.go); nil/"" in production.
fbHTTP fbHTTPDo
fbBaseURL string
// execFn replaces execCommand's process boundary in tests; nil in production.
execFn func(name string, args ...string) (string, error)
// composeExecFn replaces the initial deploy's compose call (composeExecWithEnv) in tests; nil in production.
composeExecFn func(dir string, env map[string]string, args ...string) (string, error)
// afterSetupUpFn replaces after_setup's `compose up -d` in tests; nil in production.
afterSetupUpFn func(name string) error
// nativeLockBusy: one after_setup run per app at a time (the loop, the window and a press can meet).
nativeLockBusy sync.Map
afterSetupSync bool // tests: run after_setup in the caller
// --- guarded update (slice 4, update.go) ---
updateGuards UpdateGuards // init-only, SetUpdateGuards; nil ⇒ every update is REFUSED
updateComposeFn func(dir string, env []string, args ...string) (string, error)
// composeDownFn is RecoverInterruptedInstalls' seam (R-681): nil → the real `compose down`. Tests set a
// fake so they never reach real Docker (R-650).
composeDownFn func(dir string) error
updateHealthFn func(ctx context.Context, name string, timeout time.Duration) (bool, string)
// v0.263.0 undo seams (undo.go): the volume copier (nil ⇒ docker) and the undo's health wait,
// which receives the probe it must use (nil ⇒ waitUpdateHealthyMeta).
undoCopier volumeCopier
// v0.273.0 conversion seams (pgconvert.go): the PostgreSQL process boundary (nil ⇒ docker) and the free
// space beside the stack dir (nil ⇒ statfs).
pgConv pgConverter
convertFreeFn func(path string) (int64, bool)
updateUndoHealthFn func(ctx context.Context, name string, timeout time.Duration, meta *Metadata) (bool, string)
// afterLoadFn is RunAfterLoad's exec seam (kept.go); nil = compose exec with the app's env.
afterLoadFn func(dir string, args ...string) (string, error)
// keptUnitFn names the removed app's recovery unit on a drive (backup.RemovedAppUnitFor, wired in
// main.go) so a start-fresh moves it in with the files it belongs to. nil = files only.
keptUnitFn func(app, drive string) string
probeRunFn func(t probeTarget) *HealthProbeResult // the health wait's network probe; nil ⇒ runChecks
updateEventSink func(UpdateEvent) // v0.264.0: the notifier; nil ⇒ no events
updateMemoryFn func(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal error, warning string)
updateDiskFreeFn func() (freeGiB float64, ok bool)
updateNowFn func() time.Time
updateJournalMu sync.Mutex
updateResume []string // apps whose update was interrupted after `up`; resumed once guards exist
// inspectRestartPolicyFn is the docker-inspect seam for the above; nil in production
// (dockerRestartPolicy). Tests inject a scripted lookup and never touch docker.
inspectRestartPolicyFn func(containerName string) (string, error)
// isMountPoint reports whether a path is a live mountpoint; defaults to system.IsMountPoint.
// Injectable so the userdata-belt drive-absent gate is testable (a t.TempDir is never a real mount).
isMountPoint func(string) bool
// Samba (R-7) seams — nil in production. sambaUpFn replaces the `compose up -d` call (tests
// assert the idempotent no-op performs ZERO calls); sambaPasswdFn replaces the smbpasswd
// docker-exec so no unit test touches docker or handles a real secret.
sambaUpFn func(dir string) error
sambaPasswdFn func(password string) error
sambaRunFn func() bool // replaces the docker-inspect liveness probe
sambaImgFn func() bool // replaces the `docker image inspect` local-presence probe (4b card)
// sambaAddrFn replaces the docker-exec that reads the guest's LAN IPv4 out of the samba
// container's network namespace (v0.151.0, S-2/S-5 connect-address card).
sambaAddrFn func() (string, error)
// guestNetExecFn replaces guestnet.go's docker-exec into the samba netns (R-66 gateway row +
// Debug dump network section); nil in production. One seam for all guest-net reads — tests
// script canned `ip`/resolv.conf outputs per argv and never touch docker.
guestNetExecFn func(args ...string) (string, error)
// installedExecFn is the installed-images recorder's OWN process boundary (installed.go); nil in
// production (defaultExecRunner). Separate from execFn deliberately: this one carries a context
// and a timeout, which execFn/composeExecCustomEnv do not, and a bookkeeping read must never be
// able to wedge a lifecycle action. Tests script argv -> output and never touch docker.
installedExecFn execRunner
}
// SetSambaRunProbe injects the samba liveness probe. Exported for the same reason
// SetMigrationDoneHook and SetOffboxStreamRunner are: the seam has to be reachable from ANOTHER
// package's tests — here internal/web, which needs a live-container world to prove that a live
// container no longer manufactures a job phase (S-1). Production never calls it; nil keeps the real
// docker-inspect probe.
func (m *Manager) SetSambaRunProbe(fn func() bool) { m.sambaRunFn = fn }
// NewManager creates a new stack manager.
func NewManager(cfg *config.Config, logger *log.Logger) (*Manager, error) {
composeCmd := cfg.Stacks.ComposeCommand
if composeCmd == "" {
composeCmd = detectComposeCommand()
}
if composeCmd == "" {
return nil, fmt.Errorf("docker compose not found (tried 'docker compose' and 'docker-compose')")
}
logger.Printf("[INFO] [stacks] Using compose command: %s", composeCmd)
if err := os.MkdirAll(cfg.Paths.StacksDir, 0755); err != nil {
return nil, fmt.Errorf("creating stacks directory %s: %w", cfg.Paths.StacksDir, err)
}
return &Manager{
cfg: cfg,
logger: logger,
composeCmd: composeCmd,
stacks: make(map[string]*Stack),
isMountPoint: system.IsMountPoint,
}, nil
}
// GetImportRoot returns the CANONICAL drop-zone root (R-75): <system namespace root>/userdata/import.
//
// It is resolved from the SYSTEM drive, never from the app's HDD_PATH, so every app's drop-zone lands
// in one place regardless of which drive the app was deployed to. The system drive holds a felhom-data
// SUBDIR (it is not itself the namespace root — that is the inGuestDrive=false case), which is why
// NamespaceRoot is applied rather than using the configured path directly.
//
// Returns "" when the system data path is unconfigured. Callers must NOT substitute a per-drive
// fallback: that would put a folder that looks like a drop-zone on every drive while only one works.
// withPathVars leaves IMPORT_PATH unset instead, so compose fails loudly on ${IMPORT_PATH}.
//
// NOTE: the system drive is deliberately NOT a registered StoragePath (verified on both demo boxes,
// 2026-07-26), so this root is invisible to the storage UI, to buildFileBrowserPaths' per-path loop
// and to sharingResolvePath's owning-root check. Everything that must reach it does so explicitly —
// see EnsureImportRoot, the FileBrowser import bind, and the System SMB share.
func (m *Manager) GetImportRoot() string {
sys := m.cfg.Paths.SystemDataPath
if sys == "" {
m.logger.Printf("[ERROR] [stacks] IMPORT_PATH unresolvable: paths.system_data_path is empty — a drop-zone bind will fail to resolve rather than silently land on a data drive")
return ""
}
return appbackup.ImportDir(appbackup.NamespaceRoot(sys, false))
}
// ensureUserdataMounts is the deploy belt: pre-create every ${USERDATA_PATH}/... and ${IMPORT_PATH}/...
// bind source the stack declares with the userdata convention, so Docker never auto-creates one as
// guest-root.
//
// The two roots are gated DIFFERENTLY and that is load-bearing. ${USERDATA_PATH} is on the app's own
// data drive and is subject to the drive-absent gate; ${IMPORT_PATH} (R-75) is on the SYSTEM drive,
// which is always present, so gating it on a detached data drive would refuse to create a directory
// that has nothing to do with that drive.
func (m *Manager) ensureUserdataMounts(stackDir string, env []string) {
composePath := filepath.Join(stackDir, "docker-compose.yml")
binds := ParseComposeClassifiableBinds(composePath)
// --- import binds: system drive, never drive-gated ---
if importPath := envLookup(env, "IMPORT_PATH"); importPath != "" {
for _, b := range binds {
if b.Root != appbackup.RootImport {
continue
}
src := filepath.Join(importPath, filepath.FromSlash(b.RelPath))
if err := appbackup.EnsureUserdataDir(src); err != nil {
m.logger.Printf("[WARN] [stacks] import belt: ensure %s: %v", src, err)
}
}
} else {
for _, b := range binds {
if b.Root == appbackup.RootImport {
m.logger.Printf("[ERROR] [stacks] import belt: stack declares a ${IMPORT_PATH} bind but IMPORT_PATH is unset — compose will fail rather than bind a wrong-drive path")
break
}
}
}
// --- userdata binds: the app's own drive, drive-absent gated ---
userdataPath := envLookup(env, "USERDATA_PATH")
if userdataPath == "" {
return
}
// Drive-absent gate: an external drive root (not the system/local path) that isn't currently a live
// mountpoint means the drive is detached. Creating ${USERDATA_PATH}/... now would write app data onto
// the guest ROOTFS, shadowed when the drive returns (data-integrity + rootfs-fill hazard). Skip — the
// app is held by the drive gate (planDriveGates). The system/local path is legitimately not a
// mountpoint, so it is never gated.
if hdd := envLookup(env, "HDD_PATH"); hdd != "" && hdd != m.sysDataPath && !m.isMountPoint(hdd) {
m.logger.Printf("[INFO] [stacks] userdata belt: drive %s not mounted — skipping ensure (held by drive gate)", hdd)
return
}
for _, b := range binds {
if b.Root != appbackup.RootUserdata {
continue
}
src := filepath.Join(userdataPath, filepath.FromSlash(b.RelPath))
if err := appbackup.EnsureUserdataDir(src); err != nil {
m.logger.Printf("[WARN] [stacks] userdata belt: ensure %s: %v", src, err)
}
}
}
// envLookup returns the value of key in a "K=V" env slice (last occurrence wins, "" if absent).
func envLookup(env []string, key string) string {
prefix := key + "="
val := ""
for _, e := range env {
if strings.HasPrefix(e, prefix) {
val = e[len(prefix):]
}
}
return val
}
// SetEncryptionKey sets the AES-256 key used to encrypt/decrypt sensitive values in app.yaml.
func (m *Manager) SetEncryptionKey(key []byte) {
m.mu.Lock()
defer m.mu.Unlock()
m.encKey = key
}
// GetStacksBaseDir returns the base directory where stacks live.
func (m *Manager) GetStacksBaseDir() string {
return m.cfg.Paths.StacksDir
}
// MigrateEncryption re-saves app.yaml for deployed stacks that still have
// plaintext values in sensitive fields. Called once on startup.
func (m *Manager) MigrateEncryption() {
m.mu.Lock()
defer m.mu.Unlock()
if m.encKey == nil {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] MigrateEncryption: no encryption key set, skipping")
}
return
}
if m.isDebug() {
deployedCount := 0
for _, s := range m.stacks {
if s.Deployed {
deployedCount++
}
}
m.logger.Printf("[DEBUG] [stacks] MigrateEncryption: checking %d deployed stacks for plaintext sensitive values", deployedCount)
}
migrated := 0
for _, s := range m.stacks {
if !s.Deployed {
continue
}
stackDir := filepath.Dir(s.ComposePath)
appCfg := LoadAppConfig(stackDir)
if appCfg == nil {
continue
}
meta := LoadMetadata(stackDir)
sensitive := SensitiveEnvVars(&meta)
if len(sensitive) == 0 {
continue
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] MigrateEncryption: checking stack %q (%d sensitive fields)", s.Name, len(sensitive))
}
needsMigration := false
for _, envVar := range sensitive {
if v, ok := appCfg.Env[envVar]; ok && v != "" && !crypto.IsEncrypted(v) {
needsMigration = true
break
}
}
if needsMigration {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] MigrateEncryption: stack %q needs migration — re-saving with encryption", s.Name)
}
if err := SaveAppConfig(stackDir, appCfg, m.encKey, sensitive); err != nil {
m.logger.Printf("[WARN] [stacks] Encryption migration failed for %s: %v", s.Name, err)
} else {
migrated++
}
}
}
if migrated > 0 {
m.logger.Printf("[INFO] [stacks] Encrypted sensitive values in %d app.yaml file(s)", migrated)
} else {
m.logger.Printf("[INFO] [stacks] Encryption migration: no stacks needed migration")
}
}
// toTitleCase capitalizes the first letter of each word.
func toTitleCase(s string) string {
words := strings.Fields(s)
for i, w := range words {
if len(w) > 0 {
words[i] = strings.ToUpper(w[:1]) + w[1:]
}
}
return strings.Join(words, " ")
}
func detectComposeCommand() string {
if err := dockerexec.Command("docker", "compose", "version").Run(); err == nil {
return "docker compose"
}
if _, err := exec.LookPath("docker-compose"); err == nil {
return "docker-compose"
}
return ""
}
// DeployedStackNames returns the names of all deployed stacks.
func (m *Manager) DeployedStackNames() []string {
m.mu.RLock()
defer m.mu.RUnlock()
var names []string
for name, stack := range m.stacks {
if stack.Deployed {
names = append(names, name)
}
}
return names
}
// RunningAppStacks returns the names of deployed, NON-protected stacks that currently have
// containers up (running/starting/unhealthy/restarting/degraded) — the set the quiesce loop (slice 8B)
// stops before an app-consistent backup and restarts after. Protected infra (traefik, cloudflared,
// felhom-controller) is excluded so the controller never stops its own tunnel/proxy or itself.
// Sorted for deterministic stop/start order.
func (m *Manager) RunningAppStacks() []string {
m.mu.RLock()
defer m.mu.RUnlock()
var names []string
for name, stack := range m.stacks {
if !stack.Deployed || stack.Protected || m.cfg.IsProtectedStack(name) {
continue
}
switch stack.State {
// StateDegraded (R-51) belongs here: a degraded stack still has LIVE members, and the
// quiesce loop must stop them before an app-consistent backup and start them after.
case StateRunning, StateStarting, StateUnhealthy, StateRestarting, StateDegraded:
names = append(names, name)
}
}
sort.Strings(names)
return names
}
// ScanStacks discovers all compose stacks in the stacks directory.
func (m *Manager) ScanStacks() error {
m.mu.Lock()
defer m.mu.Unlock()
entries, err := os.ReadDir(m.cfg.Paths.StacksDir)
if err != nil {
return fmt.Errorf("reading stacks directory: %w", err)
}
found := make(map[string]bool)
for _, entry := range entries {
if !entry.IsDir() {
continue
}
name := entry.Name()
stackDir := filepath.Join(m.cfg.Paths.StacksDir, name)
composePath := filepath.Join(stackDir, "docker-compose.yml")
if _, err := os.Stat(composePath); os.IsNotExist(err) {
composePath = filepath.Join(stackDir, "docker-compose.yaml")
if _, err := os.Stat(composePath); os.IsNotExist(err) {
continue
}
}
found[name] = true
meta := LoadMetadata(stackDir)
appCfg := LoadAppConfig(stackDir)
deployed := appCfg != nil && appCfg.Deployed
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] ScanStacks: found stack %q deployed=%v composePath=%s", name, deployed, composePath)
}
// What the CURRENT template pins, for the update badge. Deployed non-protected apps only:
// an undeployed template has nothing to compare against, and infra stacks are not the
// customer's to update. A parse failure leaves this nil, which reads as CANNOT-TELL.
var tplImages map[string]string
if deployed && !m.cfg.IsProtectedStack(name) {
imgs, ierr := ParseComposeImages(composePath)
if ierr != nil {
m.logger.Printf("[WARN] [stacks] ScanStacks: cannot read image pins from %s: %v", composePath, ierr)
} else {
tplImages = imgs
}
}
// What the CATALOG offers — the badge's input, and deliberately a different file from the
// one above (v0.235.0). A missing catalog entry is silent at INFO: an orphaned app has no
// catalog template by definition, and warning once per app per scan would be noise.
var catImages map[string]string
stepsLeft := 0
var catDigests map[string]string
var catTestedAt time.Time
if deployed && !m.cfg.IsProtectedStack(name) {
catPath := m.CatalogTemplatePath(name, "docker-compose.yml")
if appCfg != nil {
stepsLeft = ladderStepsLeft(filepath.Dir(catPath), appCfg.PinnedImages)
}
if imgs, cerr := ParseComposeImages(catPath); cerr == nil {
catImages = imgs
catDigests, catTestedAt = catalogTestedDigests(filepath.Dir(catPath), imgs)
} else if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] ScanStacks: no readable catalog template for %s (%v) — the update badge will render nothing", name, cerr)
}
}
if existing, ok := m.stacks[name]; ok {
existing.ComposePath = composePath
existing.Meta = meta
existing.Protected = m.cfg.IsProtectedStack(name)
// Don't overwrite Deployed/AppConfig while an async deploy is in
// progress — the goroutine manages these fields (H3 fix).
if !existing.Deploying {
existing.Deployed = deployed
existing.AppConfig = appCfg
existing.TemplateImages = tplImages
existing.CatalogImages = catImages
existing.LadderStepsLeft = stepsLeft
existing.CatalogDigests, existing.CatalogTestedAt = catDigests, catTestedAt
existing.InstallInterrupted = !deployed && installInterrupted(stackDir) // R-681
}
} else {
m.stacks[name] = &Stack{
Name: name,
Meta: meta,
ComposePath: composePath,
State: StateNotDeployed,
Deployed: deployed,
Protected: m.cfg.IsProtectedStack(name),
AppConfig: appCfg,
TemplateImages: tplImages,
CatalogImages: catImages,
LadderStepsLeft: stepsLeft,
InstallInterrupted: !deployed && installInterrupted(stackDir), // R-681
CatalogDigests: catDigests,
CatalogTestedAt: catTestedAt,
}
}
}
// Remove stacks no longer on disk
for name := range m.stacks {
if !found[name] {
delete(m.stacks, name)
}
}
// Detect orphaned stacks (deployed but no longer in catalog)
catalogTemplates := m.getCatalogTemplateSlugs()
if m.isDebug() {
if catalogTemplates != nil {
m.logger.Printf("[DEBUG] [stacks] ScanStacks: catalog has %d template slugs for orphan detection", len(catalogTemplates))
} else {
m.logger.Printf("[DEBUG] [stacks] ScanStacks: catalog templates unavailable, skipping orphan detection")
}
}
if catalogTemplates != nil {
orphanCount := 0
for _, stack := range m.stacks {
if stack.Protected || !stack.Deployed {
stack.Orphaned = false
continue
}
stack.Orphaned = !catalogTemplates[stack.Name]
if stack.Orphaned {
orphanCount++
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] ScanStacks: stack %q is orphaned (deployed but not in catalog)", stack.Name)
}
}
}
if orphanCount > 0 {
m.logger.Printf("[INFO] [stacks] Detected %d orphaned stack(s)", orphanCount)
}
}
deployedCount := 0
for _, s := range m.stacks {
if s.Deployed {
deployedCount++
}
}
m.logger.Printf("[INFO] [stacks] ScanStacks complete: %d stacks found (%d deployed, %d available)",
len(m.stacks), deployedCount, len(m.stacks)-deployedCount)
return m.refreshStatusLocked()
}
// RefreshStatus updates container status for all known stacks.
func (m *Manager) RefreshStatus() error {
m.mu.Lock()
defer m.mu.Unlock()
return m.refreshStatusLocked()
}
func (m *Manager) refreshStatusLocked() error {
output, err := m.execCommand("docker", "ps", "-a",
"--format", "{{.Names}}\t{{.Image}}\t{{.State}}\t{{.Status}}\t{{.Label \"com.docker.compose.project\"}}",
"--no-trunc")
if err != nil {
return fmt.Errorf("docker ps: %w", err)
}
projectContainers := make(map[string][]ContainerInfo)
liveContainers := make(map[string]bool)
totalContainers := 0
for _, line := range strings.Split(strings.TrimSpace(output), "\n") {
if line == "" {
continue
}
parts := strings.SplitN(line, "\t", 5)
if len(parts) < 5 || parts[4] == "" {
continue
}
ci := ContainerInfo{
Name: parts[0],
Image: parts[1],
State: resolveContainerState(parts[2], parts[3]),
Status: parts[3],
}
projectContainers[parts[4]] = append(projectContainers[parts[4]], ci)
liveContainers[ci.Name] = true
totalContainers++
}
m.pruneRestartPolicyCacheLocked(liveContainers)
policyOf := m.restartPolicyLookupLocked()
// fix-6: refreshStatusLocked runs every 10s (the status-refresh job) — its per-cycle enumeration
// lines are TRACE (dropped from the debug ring) so they don't eat the post-incident window. A real
// state change is logged elsewhere at INFO; a docker error returns up the stack.
m.logger.Printf("[TRACE] [stacks] refreshStatusLocked: docker ps returned %d containers across %d projects", totalContainers, len(projectContainers))
m.logger.Printf("[INFO] [stacks] Status refresh: %d containers across %d stacks", totalContainers, len(m.stacks))
for name, stack := range m.stacks {
containers, exists := projectContainers[name]
if !exists {
stack.Containers = nil
if stack.Deploying {
stack.State = StateDeploying
} else if stack.Deployed {
stack.State = StateStopped
} else {
stack.State = StateNotDeployed
}
} else {
stack.Containers = containers
stack.State = aggregateState(containers, policyOf)
}
// Re-apply controller-side health probe results: if the last probe
// failed and Docker thinks the container is running, override to unhealthy.
if stack.State == StateRunning && probeSaysUnhealthy(stack.HealthProbe) {
stack.State = StateUnhealthy
}
// C9-F2: stamp the start of a restarting RUN, and clear it the moment the stack is anything
// else. Set AFTER the health-probe override above so the stamp always agrees with the state
// that is actually stored. Clearing on any other state is what keeps a normal deploy — which
// passes through restarting briefly — from ever accumulating toward the threshold.
if stack.State == StateRestarting {
if stack.RestartingSince.IsZero() {
stack.RestartingSince = time.Now()
}
} else {
stack.RestartingSince = time.Time{}
}
if m.isDebug() {
m.logger.Printf("[TRACE] [stacks] refreshStatusLocked: stack %q → state=%s containers=%d", name, stack.State, len(stack.Containers))
}
stack.LastUpdated = time.Now()
}
return nil
}
// dockerRestartPolicy reads one container's configured restart policy via docker inspect.
// Returns ("", err) when the container is gone or the inspect fails — the caller treats that as
// UNKNOWN (see supervisedPolicy).
func (m *Manager) dockerRestartPolicy(containerName string) (string, error) {
if m.inspectRestartPolicyFn != nil {
return m.inspectRestartPolicyFn(containerName)
}
out, err := m.execCommand("docker", "inspect", "-f", "{{.HostConfig.RestartPolicy.Name}}", containerName)
if err != nil {
return "", err
}
return strings.TrimSpace(out), nil
}
// restartPolicyLookupLocked returns the cached-and-memoizing lookup handed to aggregateState.
// MUST be called with m.mu held for writing (it populates the cache).
func (m *Manager) restartPolicyLookupLocked() restartPolicyLookup {
return func(name string) string {
key := name + "|policy"
if m.restartPolicyCache != nil {
if p, ok := m.restartPolicyCache[key]; ok {
return p
}
}
p, err := m.dockerRestartPolicy(name)
if err != nil {
// UNKNOWN — deliberately not cached, so a transient docker hiccup does not pin a
// container to "unknown" for the rest of the process lifetime.
m.logger.Printf("[WARN] [stacks] restart-policy inspect failed for container %q: %v (treating as supervised)", name, err)
return ""
}
if m.restartPolicyCache == nil {
m.restartPolicyCache = map[string]string{}
}
m.restartPolicyCache[key] = p
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] restart-policy of down member %q = %q", name, p)
}
return p
}
}
// pruneRestartPolicyCacheLocked drops cache entries for containers that no longer exist, so a
// long-lived controller cannot accumulate entries for deleted apps. MUST hold mu for writing.
func (m *Manager) pruneRestartPolicyCacheLocked(live map[string]bool) {
for key := range m.restartPolicyCache {
name := strings.TrimSuffix(key, "|policy")
if !live[name] {
delete(m.restartPolicyCache, key)
}
}
}
// resolveContainerState determines the effective state by combining Docker's
// State field (running/exited/etc.) with the Status field that contains health info.
//
// Docker State: "running", "exited", "restarting", "paused", "created", "dead", "removing"
// Docker Status: "Up 3 hours (healthy)", "Up 9 seconds (health: starting)", "Up 2 min (unhealthy)"
func resolveContainerState(dockerState, dockerStatus string) ContainerState {
state := strings.ToLower(strings.TrimSpace(dockerState))
status := strings.ToLower(dockerStatus)
switch state {
case "running":
// Check health sub-status for containers with healthchecks
if strings.Contains(status, "(health: starting)") {
return StateStarting
}
if strings.Contains(status, "(unhealthy)") {
return StateUnhealthy
}
// "(healthy)" or no healthcheck = running
return StateRunning
case "exited":
return StateExited
case "restarting":
return StateRestarting
case "paused":
return StatePaused
case "created", "dead", "removing":
return StateStopped
default:
return StateUnknown
}
}
// restartPolicyLookup returns a container's docker restart policy name ("always", "unless-stopped",
// "on-failure", "no"). An empty string means UNKNOWN — the inspect failed or no lookup was supplied.
type restartPolicyLookup func(containerName string) string
// supervisedPolicy reports whether a restart policy means "docker is supposed to keep this container
// running" — i.e. its being Exited is a fault, not a design.
//
// UNKNOWN ("") counts as supervised, deliberately fail-CLOSED, which is the opposite of the
// IsDownState fail-open rule and for a different reason: there the input is an *ambiguous state*,
// here we already KNOW a member is dead and only the excuse is missing. The P2 census (2026-07-21,
// 53 catalog templates / 78 services) found **every** catalog service on `unless-stopped` and zero
// one-shot init/migrate containers, so "unknown" in the field is an inspect failure on a container
// that is almost certainly supervised. Missing a real dead-primary alarm is the failure that cost
// 18 h; a false alarm is a banner.
func supervisedPolicy(policy string) bool {
switch policy {
case "no", "on-failure":
return false
default: // "always", "unless-stopped", "" (unknown)
return true
}
}
// aggregateState determines the overall stack state from its containers.
//
// Priority: **degraded > unhealthy/starting > restarting > all-running > stopped**
//
// R-384 (v0.222.0) MOVED `degraded` to the front, and the move is the whole fix. It used to sit last,
// below the `unhealthy > 0` return, which meant a different question was answering it.
//
// ── WHY DEGRADED IS ASKED FIRST ──────────────────────────────────────────────────────────────
//
// "Is a SUPERVISED member of this app dead?" and "is a RUNNING member failing its healthcheck?" are
// two different questions, and until v0.222.0 the second was allowed to answer the first. A two-
// container app whose database exits will normally have its front end go `unhealthy` moments later —
// it cannot reach its database. The old order returned `StateUnhealthy` immediately at that point and
// never reached the R-51 mixed-case block below, so the dead database was never asked about.
// `unhealthy` is deliberately NOT a down state (see IsDownState), so `classifyRunStates` never marked
// the app down and `NotifyAppStartFailures` never fired.
//
// **Measured live on `demo-hp` 2026-08-22:** `bookstack-db` stopped at 21:27:01 and the F-OBS
// heartbeat printed "0 currently down" throughout. That is R-51's own 18-hour immich failure back
// through a different door — the dead member now hiding behind an unhealthy survivor instead of
// behind three live helpers.
//
// ── WHAT "SOME MEMBERS ARE UP" MEANS, AND WHY IT WIDENED ─────────────────────────────────────
//
// The R-51 block was additionally guarded by `running > 0`, where `running` counts only StateRunning.
// That guard made the block UNREACHABLE in exactly the case it was written for: an unhealthy survivor
// beside a dead database counted as nothing up. "Up" now means **any member not in the down bucket** —
// running, unhealthy, starting or restarting — because each of those is a container Docker still has,
// and a dead supervised peer beside any of them is a fault either way.
//
// ── WHAT DID NOT CHANGE, DELIBERATELY ────────────────────────────────────────────────────────
//
// `IsDownState` is byte-identical. `unhealthy` is still excluded from it, for the reason recorded
// there: an unhealthy container is RUNNING, and folding it in reintroduces the flapping that
// exclusion exists to stop. This change does not fold it in — it asks a prior question first. No new
// state was minted either: `StateDegraded` already means precisely this and every consumer already
// handles it.
//
// The benign case is untouched: a one-shot init/migrate container that has legitimately finished has
// policy `no`/`on-failure`, `supervisedPolicy` returns false, and the stack reads exactly as before.
// Without that filter every app with a migration step would alarm on every start.
//
// policyOf is consulted ONLY when some members are up and some are down, and ONLY for the down
// members. nil is allowed (every exited member then reads as supervised).
func aggregateState(containers []ContainerInfo, policyOf restartPolicyLookup) ContainerState {
if len(containers) == 0 {
return StateNotDeployed
}
running := 0
starting := 0
unhealthy := 0
restarting := 0
stopped := 0
var down []ContainerInfo
for _, c := range containers {
switch c.State {
case StateRunning:
running++
case StateStarting:
starting++
case StateUnhealthy:
unhealthy++
case StateRestarting:
restarting++
case StateStopped, StateExited:
stopped++
down = append(down, c)
}
}
total := len(containers)
// R-384 — ASKED FIRST. "Is a supervised member dead?" must be answered before "is a running
// member unhealthy?", because a dying database drags its front end unhealthy and the unhealthy
// return below then swallowed the whole question. `up` counts every member NOT in the down
// bucket; `running > 0` alone made this unreachable in the exact case R-51 was written for.
if up := running + unhealthy + starting + restarting; up > 0 && len(down) > 0 {
for _, c := range down {
policy := ""
if policyOf != nil {
policy = policyOf(c.Name)
}
if supervisedPolicy(policy) {
return StateDegraded
}
}
}
// Any unhealthy → whole stack is unhealthy
if unhealthy > 0 {
return StateUnhealthy
}
// Any still starting → stack is starting
if starting > 0 {
return StateStarting
}
// Any restarting → stack is restarting
if restarting > 0 {
return StateRestarting
}
// All running (and healthy) → stack is running
if running == total {
return StateRunning
}
// All stopped → stack is stopped
if stopped == total {
return StateStopped
}
// Mix (some members up, some down) — R-51. Until v0.156.0 this reported StateRunning
// unconditionally ("partial"), which is why a dead immich-server behind three live helpers was
// invisible to fix-3 for 18 hours. A down member whose restart policy says docker should be
// keeping it up is a FAULT → the whole stack is degraded (and degraded is a down state).
//
// R-384 (v0.222.0) HOISTED that supervised test to the top of this function, so by the time
// control reaches here every down member has already been proven BENIGN — policy `no`/`on-failure`,
// a one-shot init/migrate container that has legitimately finished. The stack stays running. This
// branch therefore no longer decides anything about supervision; it only records the benign
// verdict, and the test that pins it is the one that says a finished migration must not alarm.
if running > 0 {
return StateRunning
}
return StateStopped
}
// --- Stack accessors ---
func (m *Manager) GetStacks() []Stack {
m.mu.RLock()
result := make([]Stack, 0, len(m.stacks))
for _, s := range m.stacks {
result = append(result, deepCopyStack(s))
}
g := m.updateGuards
m.mu.RUnlock()
for i := range result {
fillHoldReason(g, &result[i])
}
// Sort alphabetically by display name for consistent UI ordering
sort.Slice(result, func(i, j int) bool {
return result[i].Meta.DisplayName < result[j].Meta.DisplayName
})
return result
}
func (m *Manager) GetStack(name string) (*Stack, bool) {
m.mu.RLock()
defer m.mu.RUnlock()
s, ok := m.stacks[name]
if !ok {
return nil, false
}
cp := deepCopyStack(s)
fillHoldReason(m.updateGuards, &cp)
return &cp, true
}
// deepCopyStack creates a deep copy of a Stack, including pointer fields.
func deepCopyStack(s *Stack) Stack {
cp := *s
// Deep-copy Containers slice
if s.Containers != nil {
cp.Containers = make([]ContainerInfo, len(s.Containers))
copy(cp.Containers, s.Containers)
}
// Deep-copy AppConfig pointer
if s.AppConfig != nil {
acCopy := *s.AppConfig
if s.AppConfig.Env != nil {
acCopy.Env = make(map[string]string, len(s.AppConfig.Env))
for k, v := range s.AppConfig.Env {
acCopy.Env[k] = v
}
}
if s.AppConfig.LockedFields != nil {
acCopy.LockedFields = make([]string, len(s.AppConfig.LockedFields))
copy(acCopy.LockedFields, s.AppConfig.LockedFields)
}
cp.AppConfig = &acCopy
}
// Deep-copy HealthProbe pointer
if s.HealthProbe != nil {
hpCopy := *s.HealthProbe
if s.HealthProbe.Details != nil {
hpCopy.Details = make([]HealthCheckDetail, len(s.HealthProbe.Details))
copy(hpCopy.Details, s.HealthProbe.Details)
}
cp.HealthProbe = &hpCopy
}
// Deep-copy Meta.DeployFields slice (including nested Options)
if s.Meta.DeployFields != nil {
cp.Meta.DeployFields = make([]DeployField, len(s.Meta.DeployFields))
copy(cp.Meta.DeployFields, s.Meta.DeployFields)
for i, f := range s.Meta.DeployFields {
if f.Options != nil {
cp.Meta.DeployFields[i].Options = make([]SelectOption, len(f.Options))
copy(cp.Meta.DeployFields[i].Options, f.Options)
}
}
}
// Deep-copy Meta.OptionalConfig (slice of groups with nested Fields slices)
if s.Meta.OptionalConfig != nil {
cp.Meta.OptionalConfig = make([]OptionalConfigGroup, len(s.Meta.OptionalConfig))
copy(cp.Meta.OptionalConfig, s.Meta.OptionalConfig)
for i, g := range s.Meta.OptionalConfig {
if g.Fields != nil {
cp.Meta.OptionalConfig[i].Fields = make([]OptionalConfigField, len(g.Fields))
copy(cp.Meta.OptionalConfig[i].Fields, g.Fields)
}
}
}
// Deep-copy Meta.Integrations
if s.Meta.Integrations != nil {
cp.Meta.Integrations = make([]IntegrationDef, len(s.Meta.Integrations))
copy(cp.Meta.Integrations, s.Meta.Integrations)
}
// Deep-copy Meta.HealthCheck pointer
if s.Meta.HealthCheck != nil {
hcCopy := *s.Meta.HealthCheck
if s.Meta.HealthCheck.Checks != nil {
hcCopy.Checks = make([]HealthCheckItem, len(s.Meta.HealthCheck.Checks))
copy(hcCopy.Checks, s.Meta.HealthCheck.Checks)
for i, c := range s.Meta.HealthCheck.Checks {
if c.Expect != nil {
eCopy := *c.Expect
hcCopy.Checks[i].Expect = &eCopy
}
}
}
cp.Meta.HealthCheck = &hcCopy
}
// Deep-copy Meta.InitialCreds pointer
if s.Meta.InitialCreds != nil {
icCopy := *s.Meta.InitialCreds
cp.Meta.InitialCreds = &icCopy
}
return cp
}
// --- Stack operations ---
// StartStack, StopStack, etc. now load app.yaml env for deployed stacks.
func (m *Manager) StartStack(name string) error {
stack, ok := m.GetStack(name)
if !ok {
return fmt.Errorf("stack %q not found", name)
}
if stack.Deploying {
// R-634: a second `compose up -d` beside the deploy's own is the race that left apps running
// under „not deployed". The deploy brings the app up itself; nothing else may.
m.logger.Printf("[WARN] [stacks] StartStack %s refused: the app's deploy is still running (R-634)", name)
return fmt.Errorf("starting stack %s: %w", name, ErrStackDeploying)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] StartStack %s: current state=%s deployed=%v", name, stack.State, stack.Deployed)
}
m.logger.Printf("[INFO] [stacks] Starting stack: %s", name)
start := time.Now()
dir := filepath.Dir(stack.ComposePath)
env := m.stackEnv(dir)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] StartStack %s: prepared %d env vars for compose", name, len(env))
}
if _, err := m.composeExecCustomEnv(dir, env, "up", "-d"); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s start failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
return fmt.Errorf("starting stack %s: %w", name, err)
}
m.logger.Printf("[INFO] [stacks] Stack %s started successfully (took %.1fs)", name, time.Since(start).Seconds())
m.recordInstalledImages(name, dir, env)
m.logPostStartStatus(name, dir, env)
// Clear stale health probe so refreshStatus won't re-apply an old unhealthy override.
// The next health-probes tick (≤10s) will run a fresh probe.
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.HealthProbe = nil
}
m.mu.Unlock()
return m.RefreshStatus()
}
// StartStackServices brings up ONLY the named compose services (`docker compose up -d <svc>...`),
// leaving the rest of the stack down. It exists for R-47: a database dump must be replayed into a
// running DB container while the application that owns the schema is still stopped, otherwise the
// app's own schema management races the replay (proven live — H4,
// DIAG-immich-restore-round2-2026-07-19). Every catalog template's dependency direction is app→db,
// so naming the DB service starts the DB and nothing else.
//
// An EMPTY service list is refused rather than passed through: `up -d` with no arguments is a FULL
// start, which is precisely the behaviour this function exists to avoid — a silent fall-through
// would reintroduce the race at the one call site that most needs it not to.
//
// Deliberately no logPostStartStatus: the app containers are absent ON PURPOSE here, and it would
// WARN about every one of them. The full StartStack that always follows logs the real post-start
// state.
func (m *Manager) StartStackServices(name string, services []string) error {
if len(services) == 0 {
return fmt.Errorf("starting services of stack %s: empty service list", name)
}
stack, ok := m.GetStack(name)
if !ok {
return fmt.Errorf("stack %q not found", name)
}
m.logger.Printf("[INFO] [stacks] Starting stack %s services only: %v", name, services)
start := time.Now()
dir := filepath.Dir(stack.ComposePath)
env := m.stackEnv(dir)
if _, err := m.composeExecCustomEnv(dir, env, append([]string{"up", "-d"}, services...)...); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s service start failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
return fmt.Errorf("starting services %v of stack %s: %w", services, name, err)
}
m.logger.Printf("[INFO] [stacks] Stack %s services %v started (took %.1fs)", name, services, time.Since(start).Seconds())
return m.RefreshStatus()
}
func (m *Manager) StopStack(name string) error {
if m.cfg.IsProtectedStack(name) {
return fmt.Errorf("stack %q is protected and cannot be stopped", name)
}
stack, ok := m.GetStack(name)
if !ok {
return fmt.Errorf("stack %q not found", name)
}
if stack.Deploying {
// R-634, the backstop for EVERY caller (backup, quiesce, restore, export, storage, the Stop
// button): `compose down` in the middle of a deploy was measured to make the deploy fail and
// record „not deployed" while the caller's own restart brought the containers back.
m.logger.Printf("[WARN] [stacks] StopStack %s refused: the app's deploy is still running (R-634)", name)
return fmt.Errorf("stopping stack %s: %w", name, ErrStackDeploying)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] StopStack %s: current state=%s deployed=%v containers=%d", name, stack.State, stack.Deployed, len(stack.Containers))
}
m.logger.Printf("[INFO] [stacks] Stopping stack: %s", name)
start := time.Now()
dir := filepath.Dir(stack.ComposePath)
if _, err := m.composeExec(dir, "down"); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s stop failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
return fmt.Errorf("stopping stack %s: %w", name, err)
}
m.logger.Printf("[INFO] [stacks] Stack %s stopped successfully (took %.1fs)", name, time.Since(start).Seconds())
return m.RefreshStatus()
}
func (m *Manager) RestartStack(name string) error {
stack, ok := m.GetStack(name)
if !ok {
return fmt.Errorf("stack %q not found", name)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RestartStack %s: current state=%s deployed=%v containers=%d", name, stack.State, stack.Deployed, len(stack.Containers))
}
m.logger.Printf("[INFO] [stacks] Restarting stack: %s", name)
start := time.Now()
dir := filepath.Dir(stack.ComposePath)
env := m.stackEnv(dir)
// Use "up -d" instead of bare "restart" so that env vars from app.yaml
// are injected and any template changes (new images, healthchecks) are
// picked up. Plain "docker compose restart" only sends SIGTERM+start
// to existing containers without re-reading the compose file or env.
if _, err := m.composeExecCustomEnv(dir, env, "up", "-d"); err != nil {
m.logger.Printf("[ERROR] [stacks] Stack %s restart failed after %.1fs: %v", name, time.Since(start).Seconds(), err)
return fmt.Errorf("restarting stack %s: %w", name, err)
}
m.logger.Printf("[INFO] [stacks] Stack %s restarted successfully (took %.1fs)", name, time.Since(start).Seconds())
m.recordInstalledImages(name, dir, env)
m.logPostStartStatus(name, dir, env)
// Clear stale health probe so refreshStatus won't re-apply an old unhealthy override.
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.HealthProbe = nil
}
m.mu.Unlock()
return m.RefreshStatus()
}
// UpdateStack was REMOVED in v0.237.0 (update arc slice 4). It advanced the pin, pulled, ran `up -d`
// and reported success on the compose exit code — no copy first, no refusals, and HTTP 200 over a
// crash loop (R-443). Its only caller was the API, which now runs StartGuardedUpdate (update.go).
// Deleting it rather than leaving it is deliberate: an unguarded update path that still compiles is
// one caller away from being the next R-439.
func (m *Manager) GetLogs(name string, lines int) (string, error) {
stack, ok := m.GetStack(name)
if !ok {
return "", fmt.Errorf("stack %q not found", name)
}
if lines <= 0 {
lines = 100
}
if lines > 1000 {
lines = 1000
}
m.logger.Printf("[INFO] [stacks] Fetching logs for stack %s (tail=%d)", name, lines)
dir := filepath.Dir(stack.ComposePath)
output, err := m.composeExec(dir, "logs", "--tail", fmt.Sprintf("%d", lines), "--no-color")
if err != nil {
m.logger.Printf("[WARN] [stacks] Failed to fetch logs for %s: %v", name, err)
return "", fmt.Errorf("getting logs for %s: %w", name, err)
}
if len(output) == 0 {
m.logger.Printf("[DEBUG] Logs result for %s: 0 bytes returned (empty)", name)
} else {
m.logger.Printf("[DEBUG] Logs result for %s: %d bytes returned", name, len(output))
}
return output, nil
}
// --- Env and compose helpers ---
// stackEnv builds the full OS env slice for a stack, merging app.yaml values.
func (m *Manager) stackEnv(stackDir string) []string {
env := os.Environ()
// Always inject DOMAIN
env = append(env, fmt.Sprintf("DOMAIN=%s", m.cfg.Customer.Domain))
// Load app.yaml if it exists — merge its env vars (decrypted for docker-compose)
appCfg := LoadAppConfigDecrypted(stackDir, m.encKey)
if appCfg != nil {
for k, v := range appCfg.Env {
env = append(env, fmt.Sprintf("%s=%s", k, v))
}
// Inject USERDATA_PATH = <namespace root>/userdata alongside HDD_PATH (v0.66.0). HDD_PATH IS
// the namespace root (the chosen StoragePath: a Model-A user drive's mount, or the SSD's
// felhom-data dir), so the catalog's ${USERDATA_PATH}/... mounts resolve under userdata/.
// IMPORT_PATH (R-75) rides along but is derived from the SYSTEM drive, never from HDD_PATH.
env = withPathVars(env, appCfg.Env["HDD_PATH"], m.sysDataPath, m.GetImportRoot())
}
// App-email relay env (appended LAST so it wins over any app.yaml default). Returns nil unless
// global + per-app toggles are on and the app declares an smtp_mapping — so apps with email off
// get nothing injected. Derived, never persisted to app.yaml.
emailOn := appCfg != nil && appCfg.EmailEnabled
meta := LoadMetadata(stackDir)
if smtp := m.smtpEnv(&meta, emailOn); len(smtp) > 0 {
env = append(env, smtp...)
}
return env
}
func (m *Manager) composeExec(dir string, args ...string) (string, error) {
return m.composeExecCustomEnv(dir, nil, args...)
}
func (m *Manager) composeExecCustomEnv(dir string, env []string, args ...string) (string, error) {
var cmd *exec.Cmd
if m.composeCmd == "docker compose" {
fullArgs := append([]string{"compose"}, args...)
cmd = dockerexec.Command("docker", fullArgs...)
} else {
cmd = dockerexec.Command("docker-compose", args...)
}
cmd.Dir = dir
if env != nil {
cmd.Env = env
} else {
env = m.stackEnv(dir)
cmd.Env = env
}
// Deploy belt (v0.66.0): before `up`, pre-create every ${USERDATA_PATH}/... bind source with the
// userdata convention (2775 setgid, gid 1000) so the Docker daemon never auto-creates one as
// guest-root — covers apps not in the skeleton too. Best-effort: a perms hiccup never blocks deploy.
if len(args) > 0 && args[0] == "up" {
m.ensureUserdataMounts(dir, env)
}
// Log env var keys at debug level
if m.isDebug() {
var appKeys []string
sysCount := 0
for _, e := range env {
parts := strings.SplitN(e, "=", 2)
if len(parts) == 2 {
key := parts[0]
// Only log non-system env vars (skip PATH, HOME, etc.)
if strings.ToUpper(key) == key && !strings.HasPrefix(key, "_") {
appKeys = append(appKeys, key)
} else {
sysCount++
}
}
}
if len(appKeys) > 0 {
m.logger.Printf("[DEBUG] Env vars for compose: [%s] (%d app + %d system)",
strings.Join(appKeys, ", "), len(appKeys), sysCount)
}
}
var stdout, stderr bytes.Buffer
cmd.Stdout = &stdout
cmd.Stderr = &stderr
m.logger.Printf("[DEBUG] Running: %s %s (in %s)", m.composeCmd, strings.Join(args, " "), dir)
start := time.Now()
if err := cmd.Run(); err != nil {
elapsed := time.Since(start)
exitCode := -1
if exitErr, ok := err.(*exec.ExitError); ok {
exitCode = exitErr.ExitCode()
}
m.logger.Printf("[ERROR] [stacks] Command failed: %s %s (in %s) — exit code %d (took %.1fs)",
m.composeCmd, strings.Join(args, " "), dir, exitCode, elapsed.Seconds())
if stdoutStr := truncateStr(stdout.String(), 500); stdoutStr != "" {
m.logger.Printf("[ERROR] [stacks] stdout: %s", stdoutStr)
}
if stderrStr := truncateStr(stderr.String(), 500); stderrStr != "" {
m.logger.Printf("[ERROR] [stacks] stderr: %s", stderrStr)
}
return stdout.String(), fmt.Errorf("exit code %d\nstderr: %s", exitCode, truncateStr(stderr.String(), 500))
}
m.logger.Printf("[DEBUG] Command completed: %s %s (took %.1fs)", m.composeCmd, strings.Join(args, " "), time.Since(start).Seconds())
return stdout.String(), nil
}
func (m *Manager) execCommand(name string, args ...string) (string, error) {
// execFn is the process-boundary seam (nil in production). It exists so R-51 can be proven
// through the REAL refreshStatusLocked path — docker ps → aggregateState → docker inspect —
// rather than only through an injected aggregation helper, which would prove the helper and
// not the caller (the v0.154.0 / v0.91.0 inert-seam class).
if m.execFn != nil {
return m.execFn(name, args...)
}
cmd := dockerexec.Command(name, args...)
var stdout, stderr bytes.Buffer
cmd.Stdout = &stdout
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
m.logger.Printf("[ERROR] [stacks] execCommand failed: %v", err)
return "", fmt.Errorf("exec %s %s: %w\nstderr: %s", name, strings.Join(args, " "), err, stderr.String())
}
return stdout.String(), nil
}
// isDebug returns true if logging level is "debug".
func (m *Manager) isDebug() bool {
return m.cfg.Logging.Level == "debug"
}
// truncateStr truncates a string to maxLen characters, appending "..." if truncated.
func truncateStr(s string, maxLen int) string {
s = strings.TrimSpace(s)
if len(s) <= maxLen {
return s
}
return s[:maxLen] + "..."
}
// logPostStartStatus queries container states after a start/deploy operation
// and logs them. This runs asynchronously to avoid blocking the HTTP response.
func (m *Manager) logPostStartStatus(name, stackDir string, env []string) {
envCopy := make([]string, len(env))
copy(envCopy, env)
go func() {
time.Sleep(3 * time.Second)
output, err := m.composeExecCustomEnv(stackDir, envCopy, "ps", "-a", "--format", "table {{.Name}}\t{{.Image}}\t{{.State}}\t{{.Status}}")
if err != nil {
m.logger.Printf("[WARN] [stacks] Post-start status check failed for %s: %v", name, err)
return
}
lines := strings.Split(strings.TrimSpace(output), "\n")
if len(lines) <= 1 {
m.logger.Printf("[WARN] [stacks] Post-start status for %s: no containers found", name)
return
}
m.logger.Printf("[INFO] [stacks] Stack %s post-start status:", name)
// Skip header line
for _, line := range lines[1:] {
m.logger.Printf("[INFO] [stacks] %s", line)
}
}()
}
// checkLocalImages parses docker-compose.yml for image: lines and checks which
// are available locally. Informational only — logs results but never fails.
func (m *Manager) checkLocalImages(name, stackDir string) {
composePath := filepath.Join(stackDir, "docker-compose.yml")
data, err := os.ReadFile(composePath)
if err != nil {
composePath = filepath.Join(stackDir, "docker-compose.yaml")
data, err = os.ReadFile(composePath)
if err != nil {
m.logger.Printf("[DEBUG] Could not read compose file for image check: %v", err)
return
}
}
var images []string
for _, line := range strings.Split(string(data), "\n") {
trimmed := strings.TrimSpace(line)
if strings.HasPrefix(trimmed, "image:") {
img := strings.TrimSpace(strings.TrimPrefix(trimmed, "image:"))
img = strings.Trim(img, "\"'")
if img != "" && !strings.Contains(img, "${") {
images = append(images, img)
}
}
}
if len(images) == 0 {
m.logger.Printf("[DEBUG] No static image references found in %s compose file", name)
return
}
m.logger.Printf("[INFO] [stacks] Deploying stack %s — checking %d images...", name, len(images))
for _, img := range images {
cmd := dockerexec.Command("docker", "image", "inspect", img)
if err := cmd.Run(); err != nil {
m.logger.Printf("[DEBUG] %s — not found locally, will pull", img)
} else {
m.logger.Printf("[DEBUG] %s — found locally", img)
}
}
}
// --- Memory helpers ---
// ParseMemoryMB parses a memory string like "500M", "1G", "1.5G", "1024M", "768"
// into megabytes. Returns 0 for empty or unparseable values. Case-insensitive.
func ParseMemoryMB(s string) int {
s = strings.TrimSpace(s)
if s == "" {
return 0
}
upper := strings.ToUpper(s)
if strings.HasSuffix(upper, "GB") {
val, err := strconv.ParseFloat(strings.TrimSuffix(upper, "GB"), 64)
if err != nil {
return 0
}
return int(val * 1024)
}
if strings.HasSuffix(upper, "G") {
val, err := strconv.ParseFloat(strings.TrimSuffix(upper, "G"), 64)
if err != nil {
return 0
}
return int(val * 1024)
}
if strings.HasSuffix(upper, "MB") {
val, err := strconv.ParseFloat(strings.TrimSuffix(upper, "MB"), 64)
if err != nil {
return 0
}
return int(val)
}
if strings.HasSuffix(upper, "M") {
val, err := strconv.ParseFloat(strings.TrimSuffix(upper, "M"), 64)
if err != nil {
return 0
}
return int(val)
}
// Plain number — assume MB
val, err := strconv.ParseFloat(s, 64)
if err != nil {
return 0
}
return int(val)
}
// CommittedMemory returns the sum of mem_request and mem_limit across all
// deployed stacks that are currently running (or starting/unhealthy/restarting).
// Stopped and exited apps are excluded since they do not consume memory.
func (m *Manager) CommittedMemory() (requestMB int, limitMB int) {
m.mu.RLock()
defer m.mu.RUnlock()
for _, s := range m.stacks {
if !s.Deployed {
continue
}
if s.State == StateStopped || s.State == StateExited {
continue
}
requestMB += ParseMemoryMB(s.Meta.Resources.MemRequest)
limitMB += ParseMemoryMB(s.Meta.Resources.MemLimit)
}
return
}
// StackMemoryMB returns the mem_request for a specific stack.
func (m *Manager) StackMemoryMB(name string) int {
m.mu.RLock()
defer m.mu.RUnlock()
if s, ok := m.stacks[name]; ok {
return ParseMemoryMB(s.Meta.Resources.MemRequest)
}
return 0
}
// getCatalogTemplateSlugs reads the synced catalog cache and returns a set of
// template slugs (directory names) that have a docker-compose.yml.
func (m *Manager) getCatalogTemplateSlugs() map[string]bool {
cacheDir := filepath.Join(m.cfg.Paths.DataDir, "catalog-cache", "templates")
entries, err := os.ReadDir(cacheDir)
if err != nil {
m.logger.Printf("[WARN] [stacks] Cannot read catalog cache for orphan detection: %v", err)
return nil
}
slugs := make(map[string]bool, len(entries))
for _, e := range entries {
if e.IsDir() {
composePath := filepath.Join(cacheDir, e.Name(), "docker-compose.yml")
if _, err := os.Stat(composePath); err == nil {
slugs[e.Name()] = true
}
}
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] getCatalogTemplateSlugs: found %d template slugs in %s", len(slugs), cacheDir)
}
return slugs
}
// AggregateStateForTest exposes aggregateState with every member supervised, to tests outside this
// package — R-456's pin lives in bootrecon.
func AggregateStateForTest(containers []ContainerInfo) ContainerState {
return aggregateState(containers, func(string) string { return "unless-stopped" })
}
// SetKeptUnitFinder wires the backup side's removed-app unit lookup (INIT-ONLY; main.go).
func (m *Manager) SetKeptUnitFinder(fn func(app, drive string) string) { m.keptUnitFn = fn }