Files
felhom-controller/controller/internal/stacks/delete.go
T
admin 8fc2b4a1a9
gates / gates (push) Successful in 26s
v0.263.0: a failed update puts the app back by itself (09 decision 15, R-637)
The guarded update gains a folder copy of the app's named volumes, taken
after the pull where the app stops anyway (decision 19, chosen by the
2026-09-23 bake-off). On a failed health check the box undoes: every copy
validated by its finished-marker first, volumes refilled, definition and pin
from the job's own pre-update copies, the old version checked with the OLD
.felhom.yml probe. It holds only if the undo fails, and the hold sentence
says so and what state the data is in. Bind-mounted folders are never
touched.

- R-637 built; R-638/R-640/R-641 do not arise with a folder copy; R-639
  (pre-update copies incl. .felhom.yml kept until the undo is over).
- journal phases copying/undoing with power-cut recovery.
- app.yaml last_update_undone + one line on the app page (hu/en).
- R-642: start/restart never answer "completed".
- Removal deletes kept undo copies.

MinAgent unchanged (0.131.0). Nine red-proofs in REPORT.md.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-23 11:12:49 +02:00

1021 lines
42 KiB
Go

package stacks
import (
"bufio"
"context"
"fmt"
"log"
"os"
"os/exec"
"path/filepath"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
)
// felhomDataDir matches backup.FelhomDataDir — duplicated to avoid circular import via StackDataProvider.
const felhomDataDir = "felhom-data"
// DeleteResponse holds the result of a stack deletion (orphan delete).
//
// R-442 (v0.236.0): HDDPathsRemoved / HDDPathsPreserved are ALWAYS non-nil — an empty list is `[]`,
// never `null`, because `null` was what a silently inert removal looked like for months and the two
// must be distinguishable. HDDPathsMissing lists folders the app recorded that were already gone from
// the drive (a fact, not a refusal). HDDNote is one customer-facing sentence about what was NOT
// found — empty when nothing needs saying.
type DeleteResponse struct {
Deleted string `json:"deleted"`
VolumesRemoved []string `json:"volumes_removed"`
HDDPathsRemoved []string `json:"hdd_paths_removed"`
HDDPathsPreserved []string `json:"hdd_paths_preserved"`
HDDPathsMissing []string `json:"hdd_paths_missing,omitempty"`
HDDNote string `json:"hdd_note,omitempty"`
}
// RemoveResponse holds the result of removing a deployed (non-orphaned) stack. Same R-442 shape as
// DeleteResponse, plus the backup half: BackupPathsRefused carries every backup path the removal
// declined to touch and why — until v0.236.0 that refusal existed only as a WARN log line.
type RemoveResponse struct {
Removed string `json:"removed"`
VolumesRemoved []string `json:"volumes_removed"`
HDDPathsRemoved []string `json:"hdd_paths_removed"`
HDDPathsPreserved []string `json:"hdd_paths_preserved"`
HDDPathsMissing []string `json:"hdd_paths_missing,omitempty"`
HDDNote string `json:"hdd_note,omitempty"`
BackupPathsRemoved []string `json:"backup_paths_removed,omitempty"`
BackupPathsRefused []string `json:"backup_paths_refused,omitempty"`
// Verified says the teardown was CHECKED, not just requested (R-626/R-633). False means a
// container carrying this project's compose label was still there after the watch window — the
// answer that used to be a silent 200.
Verified bool `json:"verified"`
ReappearedRemoved []string `json:"reappeared_removed,omitempty"`
}
// RemoveRefusedError is a removal REFUSED before anything was touched: the customer asked for the
// app's data to go with it and the box cannot honour that. It is a typed error so the API handler can
// map it to a non-2xx status and show Message verbatim (R-442). The app is NOT removed either — an app
// gone with its data left behind is unrecoverable from the UI (the customer cannot even re-run the
// removal). Same class as R-443: success is never reported over inaction.
type RemoveRefusedError struct {
Reason string // RefuseHDDUnresolved | RefuseDriveAbsent — for logs and tests
Message string // Hungarian, customer-facing, exact
}
func (e *RemoveRefusedError) Error() string { return e.Message }
// RemoveRefusedError reasons.
const (
RefuseHDDUnresolved = "hdd_unresolved" // data removal requested; compose binds a drive; app.yaml records none
RefuseDriveAbsent = "drive_absent" // data removal requested; the recorded drive is not mounted right now
)
// Customer-facing copy for the R-442 shapes. Exact strings — the live validation greps ASCII
// fragments of them (`llap` for the first, `nem el` for the second).
const (
msgHDDUnresolved = "Az alkalmazás adatainak helye nem állapítható meg, ezért semmit nem töröltünk. Az alkalmazás nem lett eltávolítva."
msgDriveAbsentFmt = "A(z) %s tárhely jelenleg nem elérhető — az alkalmazás nem távolítható el, amíg a meghajtó vissza nem csatlakozik."
noteNoDriveData = "Az alkalmazás nem tárolt saját adatot külső meghajtón, így ott nem volt mit törölni."
noteMissingFmt = "A következő adatmappa már nem volt a meghajtón: %s"
backupRefusedFmt = "%s — a mentés helye a várt mappán kívül esik, ezért nem töröltük"
)
// appHDDPath returns the data drive the named app RECORDED for itself at deploy time — app.yaml's
// HDD_PATH — and whether it recorded one at all.
//
// It implements, for the removal path, the rule 07-backup-architecture.md states under "[DESIGN]
// 2026-08-22 — the restore destination is resolved by the same rule as the capture destination"
// (~L437): "the drive if the app declares one (HDD_PATH), the system data path otherwise". Deploy
// (withPathVars), the start gate (api.startGatedByMissingDrive) and the backup destination
// (backup.GetAppDrivePath) all read the app's own record. Until v0.236.0 removal alone read the
// GLOBAL cfg.Paths.HDDPath — set on no box — so "delete my data" resolved zero mounts and reported
// success over 128 MB left on the drive (R-442, measured on demo-hp 2026-09-01).
//
// DELIBERATELY NO FALLBACK to m.cfg.Paths.HDDPath when the per-app value is empty. A single global
// drive is the assumption the storage arc removed (a customer can have several), and an empty answer
// must reach the caller as "not declared" so it can tell an SSD-only app (nothing to remove — a fact)
// from a removal it cannot honour (a refusal). A silent fallback is the exact path R-442 closes.
func (m *Manager) appHDDPath(name string) (string, bool) {
cfg := m.LoadAppConfigByName(name)
if cfg == nil {
return "", false
}
hdd := strings.TrimSpace(cfg.Env["HDD_PATH"])
if hdd == "" {
return "", false
}
return filepath.Clean(hdd), true
}
// composeBindsDrive reports whether the app's compose file binds anything under ${HDD_PATH} or
// ${USERDATA_PATH} — whether the app keeps data on a drive AT ALL. This is R-442's deduplication
// rule: "declares no drive" is a fact (an SSD-resident app — nothing to remove, empty list), while
// "binds a drive it cannot resolve" is a failure (refuse). Read through the ONE authoritative bind
// scanner; ParseComposeHDDMounts is unchanged.
func composeBindsDrive(composePath string) bool {
for _, b := range ParseComposeClassifiableBinds(composePath) {
if b.Root == appbackup.RootHDD || b.Root == appbackup.RootUserdata {
return true
}
}
return false
}
// hddPathForRemoval resolves the drive a removal acts on, or refuses — BEFORE anything is touched.
// Returns (path, declared, nil) to proceed; a *RemoveRefusedError to stop. Every refusal is logged at
// ERROR here AND returned to the caller, never one without the other. A removal that does not ask
// for the data is never refused on HDD grounds.
func (m *Manager) hddPathForRemoval(op, name, composePath string, removeHDDData bool) (string, bool, error) {
hddPath, declared := m.appHDDPath(name)
if !removeHDDData {
return hddPath, declared, nil
}
if !declared {
if composeBindsDrive(composePath) {
m.logger.Printf("[ERROR] [stacks] %s %s refused: data removal requested, the compose binds a drive path, but app.yaml records no HDD_PATH — nothing removed, app kept (R-442)", op, name)
return "", false, &RemoveRefusedError{Reason: RefuseHDDUnresolved, Message: msgHDDUnresolved}
}
return "", false, nil // SSD-resident: there is no drive data, and that is a fact
}
if !m.DriveLive(hddPath) {
m.logger.Printf("[ERROR] [stacks] %s %s refused: data removal requested but the drive recorded in HDD_PATH is not mounted — nothing removed, app kept (R-442)", op, name)
return hddPath, true, &RemoveRefusedError{Reason: RefuseDriveAbsent, Message: fmt.Sprintf(msgDriveAbsentFmt, hddPath)}
}
return hddPath, true, nil
}
// hddNoteFor composes the one-sentence HDDNote (see DeleteResponse). Only when the data was asked
// for: a kept-data removal has nothing to explain about what was not found.
func hddNoteFor(removeHDDData bool, mounts, missing []string) string {
switch {
case !removeHDDData:
return ""
case len(mounts) == 0:
return noteNoDriveData
case len(missing) > 0:
return fmt.Sprintf(noteMissingFmt, strings.Join(missing, ", "))
}
return ""
}
// BackupDataResponse holds information about backup data associated with a stack.
type BackupDataResponse struct {
Stack string `json:"stack"`
BackupPaths []HDDPath `json:"backup_paths"` // reuses HDDPath (path, size, exists)
HasBackups bool `json:"has_backups"`
}
// HDDDataResponse holds information about HDD data associated with a stack.
type HDDDataResponse struct {
Stack string `json:"stack"`
HDDPaths []HDDPath `json:"hdd_paths"`
HasHDDData bool `json:"has_hdd_data"`
}
// HDDPath represents a single HDD bind mount path and its status.
type HDDPath struct {
Path string `json:"path"`
SizeBytes int64 `json:"size_bytes"`
SizeHuman string `json:"size_human"`
Exists bool `json:"exists"`
}
// ProtectedHDDPaths returns the set of top-level HDD directories that must never be deleted.
func ProtectedHDDPaths(hddPath string) map[string]bool {
if hddPath == "" {
return nil
}
return map[string]bool{
// Model A: the in-guest drive mount IS the felhom-data namespace root, so backups/ and
// appdata/ sit directly under it (no felhom-data segment).
hddPath: true,
filepath.Join(hddPath, "appdata"): true,
filepath.Join(hddPath, "backups"): true,
filepath.Join(hddPath, "media"): true,
filepath.Join(hddPath, "Dokumentumok"): true,
// Legacy pre-Model-A double-nest location; kept protected so any leftover data there is
// never wiped by a removal.
filepath.Join(hddPath, felhomDataDir): true,
filepath.Join(hddPath, felhomDataDir, "appdata"): true,
filepath.Join(hddPath, felhomDataDir, "backups"): true,
}
}
// DeleteStack removes an orphaned stack: stops containers, removes volumes,
// optionally removes HDD data, and deletes the stack directory.
func (m *Manager) DeleteStack(name string, removeHDDData bool) (*DeleteResponse, error) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack called: name=%q, removeHDDData=%v", name, removeHDDData)
}
// Safety: never delete protected stacks
if m.cfg.IsProtectedStack(name) {
return nil, fmt.Errorf("stack %q is protected and cannot be deleted", name)
}
stack, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: state=%s, deployed=%v, orphaned=%v, deploying=%v",
name, stack.State, stack.Deployed, stack.Orphaned, stack.Deploying)
}
// Must be orphaned
if !stack.Orphaned {
return nil, fmt.Errorf("stack %q is not orphaned — only orphaned stacks can be deleted", name)
}
// Must not be deploying (H2 fix)
if stack.Deploying {
return nil, fmt.Errorf("stack %q is currently being deployed — wait for deployment to finish", name)
}
// Must be stopped (not running)
// StateDegraded (R-51) counts as running here: a degraded stack still has LIVE containers, and
// deleting its directory out from under them would leave orphans behind.
if stack.State == StateRunning || stack.State == StateStarting || stack.State == StateRestarting || stack.State == StateDegraded {
return nil, fmt.Errorf("stack %q is still running — stop it first before deleting", name)
}
stackDir := filepath.Dir(stack.ComposePath)
// R-442: the app's OWN recorded drive, never the global config — and a refusal here happens
// before compose down, so a refused removal has touched nothing.
hddPath, hddDeclared, err := m.hddPathForRemoval("DeleteStack", name, stack.ComposePath, removeHDDData)
if err != nil {
return nil, err
}
m.logger.Printf("[INFO] Deleting orphaned stack: %s (removeHDDData=%v, hddDeclared=%v)", name, removeHDDData, hddDeclared)
start := time.Now()
resp := &DeleteResponse{
Deleted: name,
HDDPathsRemoved: []string{},
HDDPathsPreserved: []string{},
}
// Step 1: Parse compose file for HDD bind mounts
hddMounts := ParseComposeHDDMounts(stack.ComposePath, hddPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: found %d HDD mounts from compose file", name, len(hddMounts))
for i, mount := range hddMounts {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: HDD mount[%d]=%s", name, i, mount)
}
}
// Step 2: Run docker compose down --rmi local --volumes
// H14: Return error if docker compose down fails — continuing would leave orphaned containers.
env := m.stackEnv(stackDir)
output, err := m.composeExecCustomEnv(stackDir, env, "down", "--rmi", "local", "--volumes")
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: compose down output: %s", name, truncateStr(output, 500))
}
if err != nil {
m.logger.Printf("[ERROR] docker compose down for %s failed: %v (output: %s)", name, err, truncateStr(output, 200))
return resp, fmt.Errorf("docker compose down failed for %s: %w", name, err)
}
// Step 3: Identify removed volumes from compose output
for _, line := range strings.Split(output, "\n") {
line = strings.TrimSpace(line)
if strings.Contains(line, "Removing volume") || strings.Contains(line, "Volume") {
resp.VolumesRemoved = append(resp.VolumesRemoved, line)
}
}
// Step 4: Handle HDD data
protected := ProtectedHDDPaths(hddPath)
for _, mount := range hddMounts {
// Safety: never delete protected top-level dirs
cleanPath := filepath.Clean(mount)
if protected != nil && protected[cleanPath] {
m.logger.Printf("[WARN] Refusing to delete protected HDD path: %s", cleanPath)
continue
}
if _, err := os.Stat(cleanPath); os.IsNotExist(err) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: HDD path does not exist, skipping: %s", name, cleanPath)
}
resp.HDDPathsMissing = append(resp.HDDPathsMissing, cleanPath) // R-442: stated, not a refusal
continue // path doesn't exist, nothing to do
}
if removeHDDData {
// Get size before removal
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: removing HDD path %s (%s)", name, cleanPath, sizeHuman)
}
if err := os.RemoveAll(cleanPath); err != nil {
m.logger.Printf("[ERROR] Failed to remove HDD data %s: %v", cleanPath, err)
} else {
m.logger.Printf("[INFO] Removed HDD data: %s (%s)", cleanPath, sizeHuman)
resp.HDDPathsRemoved = append(resp.HDDPathsRemoved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
} else {
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: preserving HDD path %s (%s)", name, cleanPath, sizeHuman)
}
resp.HDDPathsPreserved = append(resp.HDDPathsPreserved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
}
resp.HDDNote = hddNoteFor(removeHDDData, hddMounts, resp.HDDPathsMissing)
// Step 5: Remove stack directory
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: removing stack directory %s", name, stackDir)
}
if err := os.RemoveAll(stackDir); err != nil {
m.logger.Printf("[ERROR] Failed to remove stack directory %s: %v", stackDir, err)
return resp, fmt.Errorf("failed to remove stack directory: %w", err)
}
m.logger.Printf("[INFO] Stack %s deleted successfully (took %.1fs)", name, time.Since(start).Seconds())
// Step 6: Remove from in-memory map and rescan
m.mu.Lock()
delete(m.stacks, name)
m.mu.Unlock()
if err := m.ScanStacks(); err != nil {
m.logger.Printf("[WARN] Rescan after delete failed: %v", err)
}
return resp, nil
}
// GetStackHDDData returns information about HDD bind mounts for a stack.
func (m *Manager) GetStackHDDData(name string) (*HDDDataResponse, error) {
stack, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
// R-442: the app's own recorded HDD_PATH, not the global config (which no box sets).
hddPath, declared := m.appHDDPath(name)
resp := &HDDDataResponse{
Stack: name,
}
if !declared {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: app.yaml records no HDD_PATH, returning empty", name)
}
return resp, nil
}
mounts := ParseComposeHDDMounts(stack.ComposePath, hddPath)
protected := ProtectedHDDPaths(hddPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: found %d raw HDD mounts from compose", name, len(mounts))
}
for _, mount := range mounts {
cleanPath := filepath.Clean(mount)
// Skip protected top-level dirs
if protected != nil && protected[cleanPath] {
continue
}
hddItem := HDDPath{
Path: cleanPath,
}
info, err := os.Stat(cleanPath)
if err != nil {
hddItem.Exists = false
} else {
hddItem.Exists = true
if info.IsDir() {
hddItem.SizeBytes = getDirSizeBytes(cleanPath)
hddItem.SizeHuman = getDirSizeHuman(cleanPath)
}
}
resp.HDDPaths = append(resp.HDDPaths, hddItem)
}
resp.HasHDDData = len(resp.HDDPaths) > 0
if m.isDebug() {
for _, p := range resp.HDDPaths {
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: path=%s exists=%v size=%s", name, p.Path, p.Exists, p.SizeHuman)
}
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: hasHDDData=%v, %d paths returned", name, resp.HasHDDData, len(resp.HDDPaths))
}
return resp, nil
}
// RemoveStack removes a deployed (non-orphaned) stack: stops containers, removes
// volumes, optionally removes HDD data and backup data, then removes app.yaml
// so the stack reverts to "not deployed" state. The template files (docker-compose.yml,
// .felhom.yml) are preserved so the user can redeploy.
// removeVerifyWindow is how long the project is watched after `down` before the teardown is called
// verified. 20 s is the floor the brief sets; the measured re-creation happened at +2 s.
const removeVerifyWindow = 25 * time.Second
// projectContainersByLabel lists containers still carrying this compose project's label — including
// stopped ones, because a container that exists at all is one the household can still see.
func (m *Manager) projectContainersByLabel(project string) []string {
out, err := m.execCommand("docker", "ps", "-a", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Names}}")
if err != nil {
return nil
}
var names []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
names = append(names, l)
}
}
return names
}
// verifyTornDown watches the compose project after `down` and removes anything that comes back.
// Returns whether the project was clean at the end, and what had to be removed.
func (m *Manager) verifyTornDown(name string, window time.Duration) (bool, []string) {
deadline := m.now().Add(window)
var removed []string
for {
left := m.projectContainersByLabel(name)
for _, c := range left {
lbl, _ := m.execCommand("docker", "inspect", c, "--format", "{{json .Config.Labels}}")
m.logger.Printf("[WARN] [stacks] RemoveStack %s: container %q reappeared after `down` — removing it by name; its labels: %s", name, c, truncateStr(strings.TrimSpace(lbl), 300))
if out, err := m.execCommand("docker", "rm", "-f", c); err != nil {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s: could not remove the reappeared container %q: %v (%s)", name, c, err, truncateStr(out, 160))
} else {
removed = append(removed, c)
}
}
if !m.now().Before(deadline) {
break
}
time.Sleep(2 * time.Second)
}
still := m.projectContainersByLabel(name)
if len(still) > 0 {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s: NOT verified — %d container(s) still carry this project's label after %s: %v", name, len(still), window, still)
return false, removed
}
return true, removed
}
// RemoveBusyError is a removal refused because the backup side owns the app right now. It is a
// TYPED error, not a sentence the handler pattern-matches: the first live run of this guard answered
// **500** because the status mapping greps the error TEXT for "not deployed"/"still running" and the
// busy sentence contains neither. A 500 tells the UI something broke; this is a "wait a moment".
type RemoveBusyError struct {
Why string // the guard's own words, for logs — never shown to the household
}
func (e *RemoveBusyError) Error() string { return MsgRemoveBusyHU }
// Key lets the API localise it; same contract as util.MsgError.
func (e *RemoveBusyError) Key() string { return KeyRemoveBusy }
// MsgRemoveBusyHU is the default-language bytes, matching the bundle entry for KeyRemoveBusy.
const MsgRemoveBusyHU = "Az alkalmazáson mentés vagy visszaállítás fut. Várd meg, amíg befejeződik."
// KeyRemoveBusy is the one sentence R-633 adds: a remove refused because the backup side owns the app.
const KeyRemoveBusy = "err.stacks.az_alkalmazason_mentes_vagy_visszaallitas_fut"
// halfStateEvidence answers "does anything of this stack actually EXIST?" for a stack the record
// says is not deployed (R-634). Containers first, because that is the shape that hurt: an app
// serving traffic that no button could remove.
func (m *Manager) halfStateEvidence(name string, stack *Stack) (bool, string) {
if len(stack.Containers) > 0 {
return true, fmt.Sprintf("%d container(s) exist", len(stack.Containers))
}
if stack.ComposePath != "" {
if _, err := os.Stat(stack.ComposePath); err == nil {
dir := filepath.Dir(stack.ComposePath)
if _, err := os.Stat(filepath.Join(dir, "app.yaml")); err == nil {
return true, "a compose file and an app.yaml exist on disk"
}
return true, "a compose file exists on disk"
}
}
return false, ""
}
func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemove []string) (*RemoveResponse, error) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack called: name=%q, removeHDDData=%v, backupPathsToRemove=%d", name, removeHDDData, len(backupPathsToRemove))
}
// Safety: never remove protected stacks
if m.cfg.IsProtectedStack(name) {
return nil, fmt.Errorf("stack %q is protected and cannot be removed", name)
}
stack, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: state=%s, deployed=%v, orphaned=%v, deploying=%v",
name, stack.State, stack.Deployed, stack.Orphaned, stack.Deploying)
}
// R-634: `deployed` is a RECORD, and the record can be wrong while the machine is right. Three
// apps were measured running, healthy and serving with `deployed=false` — `outline` answering its
// own `/_health` with 200 and three containers up — and in that state BOTH remove calls answered
// `stack "x" is not deployed`, so the household had no button at all and a shell was the only
// exit. **The household must always be able to remove what the box shows them.** So the refusal
// now asks whether anything EXISTS, not whether a flag is set: containers, a compose file, or an
// app.yaml are each enough. Everything downstream already copes — the removal is driven by the
// compose file and the directory, not by the flag.
if !stack.Deployed {
half, why := m.halfStateEvidence(name, stack)
if !half {
return nil, fmt.Errorf("stack %q is not deployed", name)
}
m.logger.Printf("[WARN] [stacks] RemoveStack %s: deployed=false but %s — removing what exists (R-634)", name, why)
}
// Must not be deploying (H2 fix)
if stack.Deploying {
return nil, fmt.Errorf("stack %q is currently being deployed — wait for deployment to finish", name)
}
// R-633: the backup side owns operations this package cannot see. A remove sent while a RESTORE
// was in flight was measured tearing down what existed while the restore's own `compose up`
// re-created it — both calls returned success, the record said `deployed: false`, and a container
// went on restarting for hours with a live traefik route. The product already refuses exactly
// this clash for `update` and for `restore`, and names the blocking operation; `remove` did not
// consult it at all. Same guard, same adapter.
if g := m.guards(); g != nil {
if busy, why := g.Busy(name); busy {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s REFUSED (busy): %s", name, why)
return nil, &RemoveBusyError{Why: why}
}
}
if m.IsUpdating(name) {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s REFUSED (busy): a guarded update is in progress", name)
return nil, &RemoveBusyError{Why: "a guarded update is in progress"}
}
// Must be stopped (not running)
// StateDegraded (R-51) counts as running here: a degraded stack still has LIVE containers, and
// deleting its directory out from under them would leave orphans behind.
if stack.State == StateRunning || stack.State == StateStarting || stack.State == StateRestarting || stack.State == StateDegraded {
return nil, fmt.Errorf("stack %q is still running — stop it first before removing", name)
}
stackDir := filepath.Dir(stack.ComposePath)
// R-442: the app's OWN recorded drive, never the global config — and a refusal here happens
// before compose down, so a refused removal has touched nothing.
hddPath, hddDeclared, err := m.hddPathForRemoval("RemoveStack", name, stack.ComposePath, removeHDDData)
if err != nil {
return nil, err
}
m.logger.Printf("[INFO] Removing deployed stack: %s (removeHDDData=%v, hddDeclared=%v, backupPaths=%d)", name, removeHDDData, hddDeclared, len(backupPathsToRemove))
start := time.Now()
resp := &RemoveResponse{
Removed: name,
HDDPathsRemoved: []string{},
HDDPathsPreserved: []string{},
}
// Step 1: Parse compose file for HDD bind mounts
hddMounts := ParseComposeHDDMounts(stack.ComposePath, hddPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: found %d HDD mounts from compose file", name, len(hddMounts))
for i, mount := range hddMounts {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: HDD mount[%d]=%s", name, i, mount)
}
}
// Step 2: Run docker compose down --volumes (keep images for potential redeploy)
env := m.stackEnv(stackDir)
// R-489 (v0.242.0): the volumes are listed BEFORE and AFTER; the difference is what was removed.
// Parsing compose's progress output reported `null` over volumes it did remove — measured five
// times on demo-hp 2026-09-13 — because compose prints that progress to a TTY it does not have here.
volsBefore := m.projectVolumes(name)
output, err := m.composeExecCustomEnv(stackDir, env, "down", "--volumes")
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: compose down output: %s", name, truncateStr(output, 500))
}
if err != nil {
m.logger.Printf("[ERROR] docker compose down for %s failed: %v (output: %s)", name, err, truncateStr(output, 200))
return resp, fmt.Errorf("docker compose down failed for %s: %w", name, err)
}
// Step 2b: WATCH, then say so. R-626 recorded a removed `navidrome` coming back and could not
// diagnose it; R-633 caught the same shape with the window visible — a restore's own `compose
// up` re-creating what `down` had just torn out, seventeen seconds apart, while both calls
// returned success. `down` returning 0 is a request, not a result. So the project is watched for
// a bounded window and anything that reappears is removed BY NAME and logged with the label that
// created it, and the answer carries whether the check passed.
resp.Verified, resp.ReappearedRemoved = m.verifyTornDown(name, removeVerifyWindow)
// R-614: the app is going; its update record goes with it. Otherwise the NEXT install of the
// same name inherits a phase that belongs to an app that no longer exists.
m.ClearUpdateState(name)
// v0.263.0: an undo copy kept by a failed undo goes with the app — its volumes were just removed,
// and a copy of them must not outlive them.
if n := m.RemoveUndoCopies(name); n > 0 {
m.logger.Printf("[INFO] [stacks] RemoveStack %s: removed %d undo copy volume(s)", name, n)
}
// Step 3: the volumes that are gone now — `[]` when none, never null (R-489).
resp.VolumesRemoved = removedVolumes(volsBefore, m.projectVolumes(name))
if len(resp.VolumesRemoved) > 0 {
m.logger.Printf("[INFO] [stacks] RemoveStack %s: removed volume(s) %v", name, resp.VolumesRemoved)
}
// Step 4: Handle HDD data
protected := ProtectedHDDPaths(hddPath)
for _, mount := range hddMounts {
cleanPath := filepath.Clean(mount)
if protected != nil && protected[cleanPath] {
m.logger.Printf("[WARN] Refusing to delete protected HDD path: %s", cleanPath)
continue
}
if _, err := os.Stat(cleanPath); os.IsNotExist(err) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: HDD path does not exist, skipping: %s", name, cleanPath)
}
resp.HDDPathsMissing = append(resp.HDDPathsMissing, cleanPath) // R-442: stated, not a refusal
continue
}
if removeHDDData {
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: removing HDD path %s (%s)", name, cleanPath, sizeHuman)
}
if err := os.RemoveAll(cleanPath); err != nil {
m.logger.Printf("[ERROR] Failed to remove HDD data %s: %v", cleanPath, err)
} else {
m.logger.Printf("[INFO] Removed HDD data: %s (%s)", cleanPath, sizeHuman)
resp.HDDPathsRemoved = append(resp.HDDPathsRemoved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
} else {
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: preserving HDD path %s (%s)", name, cleanPath, sizeHuman)
}
resp.HDDPathsPreserved = append(resp.HDDPathsPreserved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
}
resp.HDDNote = hddNoteFor(removeHDDData, hddMounts, resp.HDDPathsMissing)
// Step 5: Handle backup data cleanup. Model A: backups/ sits directly under the app's felhom-data
// namespace root. R-442: that root is resolved by the SAME rule as the data half and as the
// router's AppNamespaceRoot that produced these paths — the app's own drive when it records one,
// the system data path otherwise (07-backup-architecture.md ~L437) — instead of the global
// cfg.Paths.HDDPath, under which every backup path was refused on every box. And the refusal now
// reaches the response (BackupPathsRefused), not only the log.
nsDrive := hddPath
if !hddDeclared {
nsDrive = m.sysDataPath
}
backupsBase := ""
if nsDrive != "" {
backupsBase = filepath.Join(appbackup.NamespaceRootFor(nsDrive, m.sysDataPath), "backups")
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: processing %d backup paths for removal (base=%s)", name, len(backupPathsToRemove), backupsBase)
}
for _, bkPath := range backupPathsToRemove {
cleanPath := filepath.Clean(bkPath)
// Validate path is under the expected backups directory
if backupsBase == "" || !strings.HasPrefix(cleanPath, backupsBase+string(filepath.Separator)) {
m.logger.Printf("[WARN] Refusing to remove backup path outside expected directory: %s (expected under %s)", cleanPath, backupsBase)
resp.BackupPathsRefused = append(resp.BackupPathsRefused, fmt.Sprintf(backupRefusedFmt, cleanPath))
continue
}
if _, err := os.Stat(cleanPath); os.IsNotExist(err) {
continue
}
sizeHuman := getDirSizeHuman(cleanPath)
if err := os.RemoveAll(cleanPath); err != nil {
m.logger.Printf("[ERROR] Failed to remove backup data %s: %v", cleanPath, err)
} else {
m.logger.Printf("[INFO] Removed backup data: %s (%s)", cleanPath, sizeHuman)
resp.BackupPathsRemoved = append(resp.BackupPathsRemoved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
}
// Step 6: Remove app.yaml only (keep template files for redeploy)
appYAMLPath := filepath.Join(stackDir, "app.yaml")
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: removing app.yaml at %s", name, appYAMLPath)
}
if err := os.Remove(appYAMLPath); err != nil && !os.IsNotExist(err) {
m.logger.Printf("[ERROR] Failed to remove %s: %v", appYAMLPath, err)
return resp, fmt.Errorf("failed to remove app.yaml: %w", err)
}
m.logger.Printf("[INFO] Stack %s removed successfully (took %.1fs)", name, time.Since(start).Seconds())
// Step 7: Update in-memory state and rescan
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Deployed = false
s.AppConfig = nil
}
m.mu.Unlock()
if err := m.ScanStacks(); err != nil {
m.logger.Printf("[WARN] Rescan after remove failed: %v", err)
}
return resp, nil
}
// GetStackBackupData returns information about backup data for a stack — what the remove dialog
// sizes "delete backups" from. drivePath is the app's namespace root; mirrorDirs are the app's
// Tier-2 mirror directories (backup.Manager.Tier2MirrorDirsForApp — stacks cannot import backup).
//
// R-485 (v0.240.0): until v0.239.0 this read `<ns>/backups/primary/<app>/db-dumps` and a pre-v2
// `<ns>/backups/secondary/<app>/rsync` path — both dead for an app whose backups are the recovery
// unit and a v2 mirror — and answered `has_backups:false` over 484 MB (measured on demo-hp
// 2026-09-13). It now sizes the whole recovery unit and every mirror.
func (m *Manager) GetStackBackupData(name string, drivePath string, mirrorDirs []string) (*BackupDataResponse, error) {
_, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
resp := &BackupDataResponse{
Stack: name,
}
if drivePath == "" {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackBackupData %s: no drive path provided, returning empty", name)
}
return resp, nil
}
// The recovery unit — definition, db-dumps and volume-dumps together. drivePath is the felhom-data
// namespace ROOT (Model A: the in-guest drive mount itself), so backups/ sits directly under it.
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(appbackup.RecoveryUnitPath(drivePath, name)))
for _, d := range mirrorDirs {
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(d))
}
if m.isDebug() {
for _, p := range resp.BackupPaths {
m.logger.Printf("[DEBUG] [stacks] GetStackBackupData %s: checked path=%s exists=%v size=%s", name, p.Path, p.Exists, p.SizeHuman)
}
}
for _, p := range resp.BackupPaths {
if p.Exists {
resp.HasBackups = true
break
}
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackBackupData %s: hasBackups=%v", name, resp.HasBackups)
}
return resp, nil
}
// buildPathInfo creates an HDDPath with size info for a given path.
func buildPathInfo(path string) HDDPath {
item := HDDPath{Path: path}
info, err := os.Stat(path)
if err != nil {
item.Exists = false
return item
}
item.Exists = true
if info.IsDir() {
item.SizeBytes = getDirSizeBytes(path)
item.SizeHuman = getDirSizeHuman(path)
}
return item
}
// ParseComposeUserdataMounts reads a docker-compose.yml and extracts the host bind-source paths that
// reference ${USERDATA_PATH} (resolved to userdataPath) — the dirs the deploy belt must pre-create
// with the userdata convention.
//
// R-75: this is now a thin RESOLVER over ParseComposeClassifiableBinds, which is the ONE authoritative
// compose-bind scanner. The two used to be byte-for-byte duplicate scanners (SPIKE §3) differing only
// in what they threw away, so a fix to one silently skipped the other; the classifier won because it
// is the richer of the two (it keeps the root and the :ro flag, both of which this function discards
// but the classification and derivation paths need).
//
// ONE deliberate behaviour drop, recorded rather than hidden: the old textual
// strings.ReplaceAll("${USERDATA_PATH}", …) + containment check also accepted a bind written as a
// LITERAL absolute path that happened to fall under userdataPath. The classifier matches the ${VAR}
// reference only. No catalog template has ever used the literal form (verified across all 53 in
// SPIKE §2 — every host token is a ${VAR}, a named volume, or the docker socket), and such a compose
// would be pinned to one machine's drive layout, so the capability was dead.
func ParseComposeUserdataMounts(composePath, userdataPath string) []string {
if userdataPath == "" {
return nil
}
var out []string
for _, b := range ParseComposeClassifiableBinds(composePath) {
if b.Root != appbackup.RootUserdata {
continue
}
out = append(out, filepath.Join(userdataPath, filepath.FromSlash(b.RelPath)))
}
return out
}
// ExportDataMounts returns the host directories a .fab export must capture for an app: the
// ${HDD_PATH}-referencing bind mounts PLUS — when the compose binds ${USERDATA_PATH} (the standard
// felhom convention; USERDATA_PATH = <HDD_PATH>/userdata, injected at deploy by withUserdataPath) —
// the userdata ROOT as a single entry. C6B-F1 (v0.130.0): ParseComposeHDDMounts alone never
// resolves ${USERDATA_PATH}, so 12/13 needs_hdd catalog apps exported ZERO userdata (a silent
// hollow bundle that passed the v0.125.0 guard).
//
// The userdata subtree is deliberately captured at its ROOT, not per-bind: the .fab manifest keys
// HDD tars by basename, and the import side maps a basename either to a resolved ${HDD_PATH} mount
// or to <HDD_PATH>/<basename> — "userdata" round-trips through that mapping exactly, while a
// nested bind like ${USERDATA_PATH}/media/tv would base to "tv" and restore to the wrong place.
// The root also covers sibling dirs the app created beyond its declared binds (same philosophy as
// the tier-2 namespace-wholesale copy).
//
// Dedupe is containment-aware in both directions: the userdata root is skipped when an HDD mount
// already covers it (an app binding ${HDD_PATH} itself), and HDD mounts inside the userdata root
// are dropped when the root is added (a literal ${HDD_PATH}/userdata/x bind would otherwise
// double-tar and basename-collide with the root).
// R-203 — nsRoot is the app's felhom-data NAMESPACE ROOT, which is what UserdataDir takes. It is
// NOT hddPath: identical on an enrolled drive, one segment shorter on the system-data fallback.
//
// SCOPE NOTE, because this function lives in delete.go and that is misleading: it is EXPORT-only.
// Its single production caller is the .fab export adapter (cmd/controller/main.go, exportAdapter.
// GetStackDataMounts). Nothing deletes based on this result. The delete path's own guard,
// ProtectedHDDPaths above, is layout-agnostic by construction — it protects BOTH <hdd>/... and
// <hdd>/felhom-data/... — so it was never affected by the namespace-root defect.
func ExportDataMounts(composePath, hddPath, nsRoot string) []string {
if hddPath == "" {
return nil
}
hddMounts := ParseComposeHDDMounts(composePath, hddPath)
if nsRoot == "" {
nsRoot = hddPath // an enrolled drive, or a caller with nothing better — the pre-R-203 shape
}
ud := appbackup.UserdataDir(filepath.Clean(nsRoot))
if len(ParseComposeUserdataMounts(composePath, ud)) == 0 {
return hddMounts
}
for _, m := range hddMounts {
if m == ud || strings.HasPrefix(ud, m+string(filepath.Separator)) {
// an HDD mount already covers the userdata root — nothing to add
return hddMounts
}
}
mounts := make([]string, 0, len(hddMounts)+1)
for _, m := range hddMounts {
if strings.HasPrefix(m, ud+string(filepath.Separator)) {
continue // inside the userdata root — the root tar covers it
}
mounts = append(mounts, m)
}
return append(mounts, ud)
}
// ParseComposeHDDMounts reads a docker-compose.yml and extracts host paths
// that reference the HDD path from volume bind mounts.
func ParseComposeHDDMounts(composePath, hddPath string) []string {
if hddPath == "" {
return nil
}
data, err := os.ReadFile(composePath)
if err != nil {
return nil
}
var mounts []string
seen := make(map[string]bool)
scanner := bufio.NewScanner(strings.NewReader(string(data)))
inVolumes := false
for scanner.Scan() {
line := strings.TrimSpace(scanner.Text())
// Track when we're in a volumes section (service-level, not top-level)
if strings.HasPrefix(line, "volumes:") {
inVolumes = true
continue
}
if inVolumes && !strings.HasPrefix(line, "-") && !strings.HasPrefix(line, "#") && line != "" {
inVolumes = false
}
if !inVolumes || !strings.HasPrefix(line, "- ") {
continue
}
// Parse bind mount: "- /host/path:/container/path:options"
mountStr := strings.TrimPrefix(line, "- ")
mountStr = strings.Trim(mountStr, "\"'")
parts := strings.SplitN(mountStr, ":", 3)
if len(parts) < 2 {
continue
}
hostPath := parts[0]
// Resolve ${HDD_PATH} variable reference
hostPath = strings.ReplaceAll(hostPath, "${HDD_PATH}", hddPath)
// C10: Clean path BEFORE prefix check to prevent traversal like ${HDD_PATH}/../../etc/passwd.
cleanPath := filepath.Clean(hostPath)
cleanHDD := filepath.Clean(hddPath)
// Check if this is an HDD mount (must be cleanHDD itself or a direct subpath)
if cleanPath != cleanHDD && !strings.HasPrefix(cleanPath, cleanHDD+string(filepath.Separator)) {
continue
}
if !seen[cleanPath] {
seen[cleanPath] = true
mounts = append(mounts, cleanPath)
}
}
log.Printf("[INFO] [stacks] ParseComposeHDDMounts: found %d HDD mounts for %s", len(mounts), composePath)
return mounts
}
// getDirSizeHuman returns a human-readable size string for a directory using du.
func getDirSizeHuman(path string) string {
cmd := exec.Command("du", "-sh", path)
output, err := cmd.Output()
if err != nil {
return "unknown"
}
fields := strings.Fields(string(output))
if len(fields) > 0 {
return fields[0]
}
return "unknown"
}
// getDirSizeBytes returns the total size in bytes for a directory.
func getDirSizeBytes(path string) int64 {
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
cmd := exec.CommandContext(ctx, "du", "-sb", path)
output, err := cmd.Output()
if err != nil {
return 0
}
fields := strings.Fields(string(output))
if len(fields) > 0 {
var size int64
if n, _ := fmt.Sscanf(fields[0], "%d", &size); n != 1 {
return 0
}
return size
}
return 0
}
// projectVolumes lists the named volumes Docker holds for a compose project (by its project label).
// A listing failure reads as no volumes, so a removal never fails on bookkeeping.
func (m *Manager) projectVolumes(project string) []string {
out, err := m.execCommand("docker", "volume", "ls", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Name}}")
if err != nil {
return nil
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols
}
// removedVolumes is before minus after, as a non-nil slice (the JSON must read `[]`, not `null`).
func removedVolumes(before, after []string) []string {
still := map[string]bool{}
for _, v := range after {
still[v] = true
}
removed := []string{}
for _, v := range before {
if !still[v] {
removed = append(removed, v)
}
}
return removed
}