Files
felhom-controller/controller/internal/stacks/delete.go
T

1063 lines
44 KiB
Go

package stacks
import (
"bufio"
"context"
"fmt"
"log"
"os"
"os/exec"
"path/filepath"
"sort"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
)
// felhomDataDir matches backup.FelhomDataDir — duplicated to avoid circular import via StackDataProvider.
const felhomDataDir = "felhom-data"
// DeleteResponse holds the result of a stack deletion (orphan delete).
//
// R-442 (v0.236.0): HDDPathsRemoved / HDDPathsPreserved are ALWAYS non-nil — an empty list is `[]`,
// never `null`, because `null` was what a silently inert removal looked like for months and the two
// must be distinguishable. HDDPathsMissing lists folders the app recorded that were already gone from
// the drive (a fact, not a refusal). HDDNote is one customer-facing sentence about what was NOT
// found — empty when nothing needs saying.
type DeleteResponse struct {
Deleted string `json:"deleted"`
VolumesRemoved []string `json:"volumes_removed"`
HDDPathsRemoved []string `json:"hdd_paths_removed"`
HDDPathsPreserved []string `json:"hdd_paths_preserved"`
HDDPathsMissing []string `json:"hdd_paths_missing,omitempty"`
HDDNote string `json:"hdd_note,omitempty"`
}
// RemoveResponse holds the result of removing a deployed (non-orphaned) stack. Same R-442 shape as
// DeleteResponse, plus the backup half: BackupPathsRefused carries every backup path the removal
// declined to touch and why — until v0.236.0 that refusal existed only as a WARN log line.
type RemoveResponse struct {
Removed string `json:"removed"`
VolumesRemoved []string `json:"volumes_removed"`
HDDPathsRemoved []string `json:"hdd_paths_removed"`
HDDPathsPreserved []string `json:"hdd_paths_preserved"`
HDDPathsMissing []string `json:"hdd_paths_missing,omitempty"`
HDDNote string `json:"hdd_note,omitempty"`
BackupPathsRemoved []string `json:"backup_paths_removed,omitempty"`
BackupPathsRefused []string `json:"backup_paths_refused,omitempty"`
// Verified says the teardown was CHECKED, not just requested (R-626/R-633). False means a
// container carrying this project's compose label was still there after the watch window — the
// answer that used to be a silent 200.
Verified bool `json:"verified"`
ReappearedRemoved []string `json:"reappeared_removed,omitempty"`
}
// RemoveRefusedError is a removal REFUSED before anything was touched: the customer asked for the
// app's data to go with it and the box cannot honour that. It is a typed error so the API handler can
// map it to a non-2xx status and show Message verbatim (R-442). The app is NOT removed either — an app
// gone with its data left behind is unrecoverable from the UI (the customer cannot even re-run the
// removal). Same class as R-443: success is never reported over inaction.
type RemoveRefusedError struct {
Reason string // RefuseHDDUnresolved | RefuseDriveAbsent — for logs and tests
Message string // Hungarian, customer-facing, exact
}
func (e *RemoveRefusedError) Error() string { return e.Message }
// RemoveRefusedError reasons.
const (
RefuseHDDUnresolved = "hdd_unresolved" // data removal requested; compose binds a drive; app.yaml records none
RefuseDriveAbsent = "drive_absent" // data removal requested; the recorded drive is not mounted right now
)
// Customer-facing copy for the R-442 shapes. Exact strings — the live validation greps ASCII
// fragments of them (`llap` for the first, `nem el` for the second).
const (
msgHDDUnresolved = "Az alkalmazás adatainak helye nem állapítható meg, ezért semmit nem töröltünk. Az alkalmazás nem lett eltávolítva."
msgDriveAbsentFmt = "A(z) %s tárhely jelenleg nem elérhető — az alkalmazás nem távolítható el, amíg a meghajtó vissza nem csatlakozik."
noteNoDriveData = "Az alkalmazás nem tárolt saját adatot külső meghajtón, így ott nem volt mit törölni."
noteMissingFmt = "A következő adatmappa már nem volt a meghajtón: %s"
backupRefusedFmt = "%s — a mentés helye a várt mappán kívül esik, ezért nem töröltük"
)
// appHDDPath returns the data drive the named app RECORDED for itself at deploy time — app.yaml's
// HDD_PATH — and whether it recorded one at all.
//
// It implements, for the removal path, the rule 07-backup-architecture.md states under "[DESIGN]
// 2026-08-22 — the restore destination is resolved by the same rule as the capture destination"
// (~L437): "the drive if the app declares one (HDD_PATH), the system data path otherwise". Deploy
// (withPathVars), the start gate (api.startGatedByMissingDrive) and the backup destination
// (backup.GetAppDrivePath) all read the app's own record. Until v0.236.0 removal alone read the
// GLOBAL cfg.Paths.HDDPath — set on no box — so "delete my data" resolved zero mounts and reported
// success over 128 MB left on the drive (R-442, measured on demo-hp 2026-09-01).
//
// DELIBERATELY NO FALLBACK to m.cfg.Paths.HDDPath when the per-app value is empty. A single global
// drive is the assumption the storage arc removed (a customer can have several), and an empty answer
// must reach the caller as "not declared" so it can tell an SSD-only app (nothing to remove — a fact)
// from a removal it cannot honour (a refusal). A silent fallback is the exact path R-442 closes.
func (m *Manager) appHDDPath(name string) (string, bool) {
cfg := m.LoadAppConfigByName(name)
if cfg == nil {
return "", false
}
hdd := strings.TrimSpace(cfg.Env["HDD_PATH"])
if hdd == "" {
return "", false
}
return filepath.Clean(hdd), true
}
// composeBindsDrive reports whether the app's compose file binds anything under ${HDD_PATH} or
// ${USERDATA_PATH} — whether the app keeps data on a drive AT ALL. This is R-442's deduplication
// rule: "declares no drive" is a fact (an SSD-resident app — nothing to remove, empty list), while
// "binds a drive it cannot resolve" is a failure (refuse). Read through the ONE authoritative bind
// scanner; ParseComposeHDDMounts is unchanged.
func composeBindsDrive(composePath string) bool {
for _, b := range ParseComposeClassifiableBinds(composePath) {
if b.Root == appbackup.RootHDD || b.Root == appbackup.RootUserdata {
return true
}
}
return false
}
// hddPathForRemoval resolves the drive a removal acts on, or refuses — BEFORE anything is touched.
// Returns (path, declared, nil) to proceed; a *RemoveRefusedError to stop. Every refusal is logged at
// ERROR here AND returned to the caller, never one without the other. A removal that does not ask
// for the data is never refused on HDD grounds.
func (m *Manager) hddPathForRemoval(op, name, composePath string, removeHDDData bool) (string, bool, error) {
hddPath, declared := m.appHDDPath(name)
if !removeHDDData {
return hddPath, declared, nil
}
if !declared {
if composeBindsDrive(composePath) {
m.logger.Printf("[ERROR] [stacks] %s %s refused: data removal requested, the compose binds a drive path, but app.yaml records no HDD_PATH — nothing removed, app kept (R-442)", op, name)
return "", false, &RemoveRefusedError{Reason: RefuseHDDUnresolved, Message: msgHDDUnresolved}
}
return "", false, nil // SSD-resident: there is no drive data, and that is a fact
}
if !m.DriveLive(hddPath) {
m.logger.Printf("[ERROR] [stacks] %s %s refused: data removal requested but the drive recorded in HDD_PATH is not mounted — nothing removed, app kept (R-442)", op, name)
return hddPath, true, &RemoveRefusedError{Reason: RefuseDriveAbsent, Message: fmt.Sprintf(msgDriveAbsentFmt, hddPath)}
}
return hddPath, true, nil
}
// hddNoteFor composes the one-sentence HDDNote (see DeleteResponse). Only when the data was asked
// for: a kept-data removal has nothing to explain about what was not found.
func hddNoteFor(removeHDDData bool, mounts, missing []string) string {
switch {
case !removeHDDData:
return ""
case len(mounts) == 0:
return noteNoDriveData
case len(missing) > 0:
return fmt.Sprintf(noteMissingFmt, strings.Join(missing, ", "))
}
return ""
}
// BackupDataResponse holds information about backup data associated with a stack.
type BackupDataResponse struct {
Stack string `json:"stack"`
BackupPaths []HDDPath `json:"backup_paths"` // reuses HDDPath (path, size, exists)
HasBackups bool `json:"has_backups"`
}
// HDDDataResponse holds information about HDD data associated with a stack.
type HDDDataResponse struct {
Stack string `json:"stack"`
HDDPaths []HDDPath `json:"hdd_paths"`
HasHDDData bool `json:"has_hdd_data"`
}
// HDDPath represents a single HDD bind mount path and its status.
type HDDPath struct {
Path string `json:"path"`
SizeBytes int64 `json:"size_bytes"`
SizeHuman string `json:"size_human"`
Exists bool `json:"exists"`
}
// ProtectedHDDPaths returns the set of top-level HDD directories that must never be deleted.
func ProtectedHDDPaths(hddPath string) map[string]bool {
if hddPath == "" {
return nil
}
return map[string]bool{
// Model A: the in-guest drive mount IS the felhom-data namespace root, so backups/ and
// appdata/ sit directly under it (no felhom-data segment).
hddPath: true,
filepath.Join(hddPath, "appdata"): true,
filepath.Join(hddPath, "backups"): true,
filepath.Join(hddPath, "media"): true,
filepath.Join(hddPath, "Dokumentumok"): true,
// Legacy pre-Model-A double-nest location; kept protected so any leftover data there is
// never wiped by a removal.
filepath.Join(hddPath, felhomDataDir): true,
filepath.Join(hddPath, felhomDataDir, "appdata"): true,
filepath.Join(hddPath, felhomDataDir, "backups"): true,
}
}
// DeleteStack removes an orphaned stack: stops containers, removes volumes,
// optionally removes HDD data, and deletes the stack directory.
func (m *Manager) DeleteStack(name string, removeHDDData bool) (*DeleteResponse, error) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack called: name=%q, removeHDDData=%v", name, removeHDDData)
}
// Safety: never delete protected stacks
if m.cfg.IsProtectedStack(name) {
return nil, fmt.Errorf("stack %q is protected and cannot be deleted", name)
}
stack, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: state=%s, deployed=%v, orphaned=%v, deploying=%v",
name, stack.State, stack.Deployed, stack.Orphaned, stack.Deploying)
}
// Must be orphaned
if !stack.Orphaned {
return nil, fmt.Errorf("stack %q is not orphaned — only orphaned stacks can be deleted", name)
}
// Must not be deploying (H2 fix)
if stack.Deploying {
return nil, fmt.Errorf("stack %q is currently being deployed — wait for deployment to finish", name)
}
// Must be stopped (not running)
// StateDegraded (R-51) counts as running here: a degraded stack still has LIVE containers, and
// deleting its directory out from under them would leave orphans behind.
if stack.State == StateRunning || stack.State == StateStarting || stack.State == StateRestarting || stack.State == StateDegraded {
return nil, fmt.Errorf("stack %q is still running — stop it first before deleting", name)
}
stackDir := filepath.Dir(stack.ComposePath)
// R-442: the app's OWN recorded drive, never the global config — and a refusal here happens
// before compose down, so a refused removal has touched nothing.
hddPath, hddDeclared, err := m.hddPathForRemoval("DeleteStack", name, stack.ComposePath, removeHDDData)
if err != nil {
return nil, err
}
m.logger.Printf("[INFO] Deleting orphaned stack: %s (removeHDDData=%v, hddDeclared=%v)", name, removeHDDData, hddDeclared)
start := time.Now()
resp := &DeleteResponse{
Deleted: name,
HDDPathsRemoved: []string{},
HDDPathsPreserved: []string{},
}
// Step 1: Parse compose file for HDD bind mounts
hddMounts := ParseComposeHDDMounts(stack.ComposePath, hddPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: found %d HDD mounts from compose file", name, len(hddMounts))
for i, mount := range hddMounts {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: HDD mount[%d]=%s", name, i, mount)
}
}
// Step 2: Run docker compose down --rmi local --volumes
// H14: Return error if docker compose down fails — continuing would leave orphaned containers.
env := m.stackEnv(stackDir)
output, err := m.composeExecCustomEnv(stackDir, env, "down", "--rmi", "local", "--volumes")
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: compose down output: %s", name, truncateStr(output, 500))
}
if err != nil {
m.logger.Printf("[ERROR] docker compose down for %s failed: %v (output: %s)", name, err, truncateStr(output, 200))
return resp, fmt.Errorf("docker compose down failed for %s: %w", name, err)
}
// Step 3: Identify removed volumes from compose output
for _, line := range strings.Split(output, "\n") {
line = strings.TrimSpace(line)
if strings.Contains(line, "Removing volume") || strings.Contains(line, "Volume") {
resp.VolumesRemoved = append(resp.VolumesRemoved, line)
}
}
// Step 4: Handle HDD data
protected := ProtectedHDDPaths(hddPath)
for _, mount := range hddMounts {
// Safety: never delete protected top-level dirs
cleanPath := filepath.Clean(mount)
if protected != nil && protected[cleanPath] {
m.logger.Printf("[WARN] Refusing to delete protected HDD path: %s", cleanPath)
continue
}
if _, err := os.Stat(cleanPath); os.IsNotExist(err) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: HDD path does not exist, skipping: %s", name, cleanPath)
}
resp.HDDPathsMissing = append(resp.HDDPathsMissing, cleanPath) // R-442: stated, not a refusal
continue // path doesn't exist, nothing to do
}
if removeHDDData {
// Get size before removal
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: removing HDD path %s (%s)", name, cleanPath, sizeHuman)
}
if err := os.RemoveAll(cleanPath); err != nil {
m.logger.Printf("[ERROR] Failed to remove HDD data %s: %v", cleanPath, err)
} else {
m.logger.Printf("[INFO] Removed HDD data: %s (%s)", cleanPath, sizeHuman)
resp.HDDPathsRemoved = append(resp.HDDPathsRemoved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
} else {
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: preserving HDD path %s (%s)", name, cleanPath, sizeHuman)
}
resp.HDDPathsPreserved = append(resp.HDDPathsPreserved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
}
resp.HDDNote = hddNoteFor(removeHDDData, hddMounts, resp.HDDPathsMissing)
// Step 5: Remove stack directory
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] DeleteStack %s: removing stack directory %s", name, stackDir)
}
if err := os.RemoveAll(stackDir); err != nil {
m.logger.Printf("[ERROR] Failed to remove stack directory %s: %v", stackDir, err)
return resp, fmt.Errorf("failed to remove stack directory: %w", err)
}
m.logger.Printf("[INFO] Stack %s deleted successfully (took %.1fs)", name, time.Since(start).Seconds())
// Step 6: Remove from in-memory map and rescan
m.mu.Lock()
delete(m.stacks, name)
m.mu.Unlock()
if err := m.ScanStacks(); err != nil {
m.logger.Printf("[WARN] Rescan after delete failed: %v", err)
}
return resp, nil
}
// GetStackHDDData returns information about HDD bind mounts for a stack.
func (m *Manager) GetStackHDDData(name string) (*HDDDataResponse, error) {
stack, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
// R-442: the app's own recorded HDD_PATH, not the global config (which no box sets).
hddPath, declared := m.appHDDPath(name)
resp := &HDDDataResponse{
Stack: name,
}
if !declared {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: app.yaml records no HDD_PATH, returning empty", name)
}
return resp, nil
}
mounts := ParseComposeHDDMounts(stack.ComposePath, hddPath)
protected := ProtectedHDDPaths(hddPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: found %d raw HDD mounts from compose", name, len(mounts))
}
for _, mount := range mounts {
cleanPath := filepath.Clean(mount)
// Skip protected top-level dirs
if protected != nil && protected[cleanPath] {
continue
}
hddItem := HDDPath{
Path: cleanPath,
}
info, err := os.Stat(cleanPath)
if err != nil {
hddItem.Exists = false
} else {
hddItem.Exists = true
if info.IsDir() {
hddItem.SizeBytes = getDirSizeBytes(cleanPath)
hddItem.SizeHuman = getDirSizeHuman(cleanPath)
}
}
resp.HDDPaths = append(resp.HDDPaths, hddItem)
}
resp.HasHDDData = len(resp.HDDPaths) > 0
if m.isDebug() {
for _, p := range resp.HDDPaths {
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: path=%s exists=%v size=%s", name, p.Path, p.Exists, p.SizeHuman)
}
m.logger.Printf("[DEBUG] [stacks] GetStackHDDData %s: hasHDDData=%v, %d paths returned", name, resp.HasHDDData, len(resp.HDDPaths))
}
return resp, nil
}
// RemoveStack removes a deployed (non-orphaned) stack: stops containers, removes
// volumes, optionally removes HDD data and backup data, then removes app.yaml
// so the stack reverts to "not deployed" state. The template files (docker-compose.yml,
// .felhom.yml) are preserved so the user can redeploy.
// removeVerifyWindow is how long the project is watched after `down` before the teardown is called
// verified. 20 s is the floor the brief sets; the measured re-creation happened at +2 s.
const removeVerifyWindow = 25 * time.Second
// projectContainersByLabel lists containers still carrying this compose project's label — including
// stopped ones, because a container that exists at all is one the household can still see.
func (m *Manager) projectContainersByLabel(project string) []string {
out, err := m.execCommand("docker", "ps", "-a", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Names}}")
if err != nil {
return nil
}
var names []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
names = append(names, l)
}
}
return names
}
// verifyTornDown watches the compose project after `down` and removes anything that comes back.
// Returns whether the project was clean at the end, and what had to be removed.
func (m *Manager) verifyTornDown(name string, window time.Duration) (bool, []string) {
deadline := m.now().Add(window)
var removed []string
for {
left := m.projectContainersByLabel(name)
for _, c := range left {
lbl, _ := m.execCommand("docker", "inspect", c, "--format", "{{json .Config.Labels}}")
m.logger.Printf("[WARN] [stacks] RemoveStack %s: container %q reappeared after `down` — removing it by name; its labels: %s", name, c, truncateStr(strings.TrimSpace(lbl), 300))
if out, err := m.execCommand("docker", "rm", "-f", c); err != nil {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s: could not remove the reappeared container %q: %v (%s)", name, c, err, truncateStr(out, 160))
} else {
removed = append(removed, c)
}
}
if !m.now().Before(deadline) {
break
}
time.Sleep(2 * time.Second)
}
still := m.projectContainersByLabel(name)
if len(still) > 0 {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s: NOT verified — %d container(s) still carry this project's label after %s: %v", name, len(still), window, still)
return false, removed
}
return true, removed
}
// RemoveBusyError is a removal refused because the backup side owns the app right now. It is a
// TYPED error, not a sentence the handler pattern-matches: the first live run of this guard answered
// **500** because the status mapping greps the error TEXT for "not deployed"/"still running" and the
// busy sentence contains neither. A 500 tells the UI something broke; this is a "wait a moment".
type RemoveBusyError struct {
Why string // the guard's own words, for logs — never shown to the household
}
func (e *RemoveBusyError) Error() string { return MsgRemoveBusyHU }
// Key lets the API localise it; same contract as util.MsgError.
func (e *RemoveBusyError) Key() string { return KeyRemoveBusy }
// MsgRemoveBusyHU is the default-language bytes, matching the bundle entry for KeyRemoveBusy.
const MsgRemoveBusyHU = "Az alkalmazáson mentés vagy visszaállítás fut. Várd meg, amíg befejeződik."
// KeyRemoveBusy is the one sentence R-633 adds: a remove refused because the backup side owns the app.
const KeyRemoveBusy = "err.stacks.az_alkalmazason_mentes_vagy_visszaallitas_fut"
// halfStateEvidence answers "does anything of this stack actually EXIST?" for a stack the record
// says is not deployed (R-634). Containers first, because that is the shape that hurt: an app
// serving traffic that no button could remove.
func (m *Manager) halfStateEvidence(name string, stack *Stack) (bool, string) {
if len(stack.Containers) > 0 {
return true, fmt.Sprintf("%d container(s) exist", len(stack.Containers))
}
if stack.ComposePath != "" {
if _, err := os.Stat(stack.ComposePath); err == nil {
dir := filepath.Dir(stack.ComposePath)
if _, err := os.Stat(filepath.Join(dir, "app.yaml")); err == nil {
return true, "a compose file and an app.yaml exist on disk"
}
return true, "a compose file exists on disk"
}
}
return false, ""
}
func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemove []string) (*RemoveResponse, error) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack called: name=%q, removeHDDData=%v, backupPathsToRemove=%d", name, removeHDDData, len(backupPathsToRemove))
}
// Safety: never remove protected stacks
if m.cfg.IsProtectedStack(name) {
return nil, fmt.Errorf("stack %q is protected and cannot be removed", name)
}
stack, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: state=%s, deployed=%v, orphaned=%v, deploying=%v",
name, stack.State, stack.Deployed, stack.Orphaned, stack.Deploying)
}
// R-634: `deployed` is a RECORD, and the record can be wrong while the machine is right. Three
// apps were measured running, healthy and serving with `deployed=false` — `outline` answering its
// own `/_health` with 200 and three containers up — and in that state BOTH remove calls answered
// `stack "x" is not deployed`, so the household had no button at all and a shell was the only
// exit. **The household must always be able to remove what the box shows them.** So the refusal
// now asks whether anything EXISTS, not whether a flag is set: containers, a compose file, or an
// app.yaml are each enough. Everything downstream already copes — the removal is driven by the
// compose file and the directory, not by the flag.
if !stack.Deployed {
half, why := m.halfStateEvidence(name, stack)
if !half {
return nil, fmt.Errorf("stack %q is not deployed", name)
}
m.logger.Printf("[WARN] [stacks] RemoveStack %s: deployed=false but %s — removing what exists (R-634)", name, why)
}
// Must not be deploying (H2 fix)
if stack.Deploying {
return nil, fmt.Errorf("stack %q is currently being deployed — wait for deployment to finish", name)
}
// R-633: the backup side owns operations this package cannot see. A remove sent while a RESTORE
// was in flight was measured tearing down what existed while the restore's own `compose up`
// re-created it — both calls returned success, the record said `deployed: false`, and a container
// went on restarting for hours with a live traefik route. The product already refuses exactly
// this clash for `update` and for `restore`, and names the blocking operation; `remove` did not
// consult it at all. Same guard, same adapter.
if g := m.guards(); g != nil {
if busy, why := g.Busy(name); busy {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s REFUSED (busy): %s", name, why)
return nil, &RemoveBusyError{Why: why}
}
}
if m.IsUpdating(name) {
m.logger.Printf("[ERROR] [stacks] RemoveStack %s REFUSED (busy): a guarded update is in progress", name)
return nil, &RemoveBusyError{Why: "a guarded update is in progress"}
}
// Must be stopped (not running)
// StateDegraded (R-51) counts as running here: a degraded stack still has LIVE containers, and
// deleting its directory out from under them would leave orphans behind.
if stack.State == StateRunning || stack.State == StateStarting || stack.State == StateRestarting || stack.State == StateDegraded {
return nil, fmt.Errorf("stack %q is still running — stop it first before removing", name)
}
stackDir := filepath.Dir(stack.ComposePath)
// R-442: the app's OWN recorded drive, never the global config — and a refusal here happens
// before compose down, so a refused removal has touched nothing.
hddPath, hddDeclared, err := m.hddPathForRemoval("RemoveStack", name, stack.ComposePath, removeHDDData)
if err != nil {
return nil, err
}
m.logger.Printf("[INFO] Removing deployed stack: %s (removeHDDData=%v, hddDeclared=%v, backupPaths=%d)", name, removeHDDData, hddDeclared, len(backupPathsToRemove))
start := time.Now()
resp := &RemoveResponse{
Removed: name,
HDDPathsRemoved: []string{},
HDDPathsPreserved: []string{},
}
// Step 1: Parse compose file for HDD bind mounts
hddMounts := ParseComposeHDDMounts(stack.ComposePath, hddPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: found %d HDD mounts from compose file", name, len(hddMounts))
for i, mount := range hddMounts {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: HDD mount[%d]=%s", name, i, mount)
}
}
// Step 2: Run docker compose down --volumes (keep images for potential redeploy)
env := m.stackEnv(stackDir)
// R-489 (v0.242.0): the volumes are listed BEFORE and AFTER; the difference is what was removed.
// Parsing compose's progress output reported `null` over volumes it did remove — measured five
// times on demo-hp 2026-09-13 — because compose prints that progress to a TTY it does not have here.
volsBefore := m.appVolumeSet(name, stackDir)
output, err := m.composeExecCustomEnv(stackDir, env, "down", "--volumes")
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: compose down output: %s", name, truncateStr(output, 500))
}
if err != nil {
m.logger.Printf("[ERROR] docker compose down for %s failed: %v (output: %s)", name, err, truncateStr(output, 200))
return resp, fmt.Errorf("docker compose down failed for %s: %w", name, err)
}
// Step 2b: WATCH, then say so. R-626 recorded a removed `navidrome` coming back and could not
// diagnose it; R-633 caught the same shape with the window visible — a restore's own `compose
// up` re-creating what `down` had just torn out, seventeen seconds apart, while both calls
// returned success. `down` returning 0 is a request, not a result. So the project is watched for
// a bounded window and anything that reappears is removed BY NAME and logged with the label that
// created it, and the answer carries whether the check passed.
resp.Verified, resp.ReappearedRemoved = m.verifyTornDown(name, removeVerifyWindow)
// R-614: the app is going; its update record goes with it. Otherwise the NEXT install of the
// same name inherits a phase that belongs to an app that no longer exists.
m.ClearUpdateState(name)
// v0.263.0: an undo copy kept by a failed undo goes with the app — its volumes were just removed,
// and a copy of them must not outlive them.
if n := m.RemoveUndoCopies(name); n > 0 {
m.logger.Printf("[INFO] [stacks] RemoveStack %s: removed %d undo copy volume(s)", name, n)
}
// Step 3: the volumes that are gone now — `[]` when none, never null (R-489).
resp.VolumesRemoved = removedVolumes(volsBefore, m.appVolumeSet(name, stackDir))
if len(resp.VolumesRemoved) > 0 {
m.logger.Printf("[INFO] [stacks] RemoveStack %s: removed volume(s) %v", name, resp.VolumesRemoved)
}
// Step 4: Handle HDD data
protected := ProtectedHDDPaths(hddPath)
for _, mount := range hddMounts {
cleanPath := filepath.Clean(mount)
if protected != nil && protected[cleanPath] {
m.logger.Printf("[WARN] Refusing to delete protected HDD path: %s", cleanPath)
continue
}
if _, err := os.Stat(cleanPath); os.IsNotExist(err) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: HDD path does not exist, skipping: %s", name, cleanPath)
}
resp.HDDPathsMissing = append(resp.HDDPathsMissing, cleanPath) // R-442: stated, not a refusal
continue
}
if removeHDDData {
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: removing HDD path %s (%s)", name, cleanPath, sizeHuman)
}
if err := os.RemoveAll(cleanPath); err != nil {
m.logger.Printf("[ERROR] Failed to remove HDD data %s: %v", cleanPath, err)
} else {
m.logger.Printf("[INFO] Removed HDD data: %s (%s)", cleanPath, sizeHuman)
resp.HDDPathsRemoved = append(resp.HDDPathsRemoved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
} else {
sizeHuman := getDirSizeHuman(cleanPath)
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: preserving HDD path %s (%s)", name, cleanPath, sizeHuman)
}
resp.HDDPathsPreserved = append(resp.HDDPathsPreserved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
}
resp.HDDNote = hddNoteFor(removeHDDData, hddMounts, resp.HDDPathsMissing)
// Step 5: Handle backup data cleanup. Model A: backups/ sits directly under the app's felhom-data
// namespace root. R-442: that root is resolved by the SAME rule as the data half and as the
// router's AppNamespaceRoot that produced these paths — the app's own drive when it records one,
// the system data path otherwise (07-backup-architecture.md ~L437) — instead of the global
// cfg.Paths.HDDPath, under which every backup path was refused on every box. And the refusal now
// reaches the response (BackupPathsRefused), not only the log.
nsDrive := hddPath
if !hddDeclared {
nsDrive = m.sysDataPath
}
backupsBase := ""
if nsDrive != "" {
backupsBase = filepath.Join(appbackup.NamespaceRootFor(nsDrive, m.sysDataPath), "backups")
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: processing %d backup paths for removal (base=%s)", name, len(backupPathsToRemove), backupsBase)
}
for _, bkPath := range backupPathsToRemove {
cleanPath := filepath.Clean(bkPath)
// Validate path is under the expected backups directory
if backupsBase == "" || !strings.HasPrefix(cleanPath, backupsBase+string(filepath.Separator)) {
m.logger.Printf("[WARN] Refusing to remove backup path outside expected directory: %s (expected under %s)", cleanPath, backupsBase)
resp.BackupPathsRefused = append(resp.BackupPathsRefused, fmt.Sprintf(backupRefusedFmt, cleanPath))
continue
}
if _, err := os.Stat(cleanPath); os.IsNotExist(err) {
continue
}
sizeHuman := getDirSizeHuman(cleanPath)
if err := os.RemoveAll(cleanPath); err != nil {
m.logger.Printf("[ERROR] Failed to remove backup data %s: %v", cleanPath, err)
} else {
m.logger.Printf("[INFO] Removed backup data: %s (%s)", cleanPath, sizeHuman)
resp.BackupPathsRemoved = append(resp.BackupPathsRemoved, fmt.Sprintf("%s (%s)", cleanPath, sizeHuman))
}
}
// Step 6: Remove app.yaml only (keep template files for redeploy)
appYAMLPath := filepath.Join(stackDir, "app.yaml")
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RemoveStack %s: removing app.yaml at %s", name, appYAMLPath)
}
if err := os.Remove(appYAMLPath); err != nil && !os.IsNotExist(err) {
m.logger.Printf("[ERROR] Failed to remove %s: %v", appYAMLPath, err)
return resp, fmt.Errorf("failed to remove app.yaml: %w", err)
}
// R-651 (v0.268.0): the pinned definition and the pinned version's .felhom.yml go with the app.
// MEASURED 2026-09-23 night on 9202: every remove left `applied-compose.yml` and `applied-meta/`
// behind, so a reinstall of the same name started beside the removed install's record — which the
// undo reads as "the version to go back to". The rest of the directory is the catalog mirror and
// stays. Pinned by TestR651_RemoveDeletesTheAppliedRecord.
for _, p := range []string{AppliedComposePath(stackDir), filepath.Join(stackDir, appliedMetaDir)} {
if err := os.RemoveAll(p); err != nil {
m.logger.Printf("[WARN] [stacks] RemoveStack %s: could not remove %s: %v", name, p, err)
}
}
m.logger.Printf("[INFO] Stack %s removed successfully (took %.1fs)", name, time.Since(start).Seconds())
// Step 7: Update in-memory state and rescan
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Deployed = false
s.AppConfig = nil
}
m.mu.Unlock()
if err := m.ScanStacks(); err != nil {
m.logger.Printf("[WARN] Rescan after remove failed: %v", err)
}
return resp, nil
}
// GetStackBackupData returns information about backup data for a stack — what the remove dialog
// sizes "delete backups" from. drivePath is the app's namespace root; mirrorDirs are the app's
// Tier-2 mirror directories (backup.Manager.Tier2MirrorDirsForApp — stacks cannot import backup).
//
// R-485 (v0.240.0): until v0.239.0 this read `<ns>/backups/primary/<app>/db-dumps` and a pre-v2
// `<ns>/backups/secondary/<app>/rsync` path — both dead for an app whose backups are the recovery
// unit and a v2 mirror — and answered `has_backups:false` over 484 MB (measured on demo-hp
// 2026-09-13). It now sizes the whole recovery unit and every mirror.
func (m *Manager) GetStackBackupData(name string, drivePath string, mirrorDirs []string) (*BackupDataResponse, error) {
_, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
}
resp := &BackupDataResponse{
Stack: name,
}
if drivePath == "" {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackBackupData %s: no drive path provided, returning empty", name)
}
return resp, nil
}
// The recovery unit — definition, db-dumps and volume-dumps together. drivePath is the felhom-data
// namespace ROOT (Model A: the in-guest drive mount itself), so backups/ sits directly under it.
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(appbackup.RecoveryUnitPath(drivePath, name)))
for _, d := range mirrorDirs {
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(d))
}
if m.isDebug() {
for _, p := range resp.BackupPaths {
m.logger.Printf("[DEBUG] [stacks] GetStackBackupData %s: checked path=%s exists=%v size=%s", name, p.Path, p.Exists, p.SizeHuman)
}
}
for _, p := range resp.BackupPaths {
if p.Exists {
resp.HasBackups = true
break
}
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] GetStackBackupData %s: hasBackups=%v", name, resp.HasBackups)
}
return resp, nil
}
// buildPathInfo creates an HDDPath with size info for a given path.
func buildPathInfo(path string) HDDPath {
item := HDDPath{Path: path}
info, err := os.Stat(path)
if err != nil {
item.Exists = false
return item
}
item.Exists = true
if info.IsDir() {
item.SizeBytes = getDirSizeBytes(path)
item.SizeHuman = getDirSizeHuman(path)
}
return item
}
// ParseComposeUserdataMounts reads a docker-compose.yml and extracts the host bind-source paths that
// reference ${USERDATA_PATH} (resolved to userdataPath) — the dirs the deploy belt must pre-create
// with the userdata convention.
//
// R-75: this is now a thin RESOLVER over ParseComposeClassifiableBinds, which is the ONE authoritative
// compose-bind scanner. The two used to be byte-for-byte duplicate scanners (SPIKE §3) differing only
// in what they threw away, so a fix to one silently skipped the other; the classifier won because it
// is the richer of the two (it keeps the root and the :ro flag, both of which this function discards
// but the classification and derivation paths need).
//
// ONE deliberate behaviour drop, recorded rather than hidden: the old textual
// strings.ReplaceAll("${USERDATA_PATH}", …) + containment check also accepted a bind written as a
// LITERAL absolute path that happened to fall under userdataPath. The classifier matches the ${VAR}
// reference only. No catalog template has ever used the literal form (verified across all 53 in
// SPIKE §2 — every host token is a ${VAR}, a named volume, or the docker socket), and such a compose
// would be pinned to one machine's drive layout, so the capability was dead.
func ParseComposeUserdataMounts(composePath, userdataPath string) []string {
if userdataPath == "" {
return nil
}
var out []string
for _, b := range ParseComposeClassifiableBinds(composePath) {
if b.Root != appbackup.RootUserdata {
continue
}
out = append(out, filepath.Join(userdataPath, filepath.FromSlash(b.RelPath)))
}
return out
}
// ExportDataMounts returns the host directories a .fab export must capture for an app: the
// ${HDD_PATH}-referencing bind mounts PLUS — when the compose binds ${USERDATA_PATH} (the standard
// felhom convention; USERDATA_PATH = <HDD_PATH>/userdata, injected at deploy by withUserdataPath) —
// the userdata ROOT as a single entry. C6B-F1 (v0.130.0): ParseComposeHDDMounts alone never
// resolves ${USERDATA_PATH}, so 12/13 needs_hdd catalog apps exported ZERO userdata (a silent
// hollow bundle that passed the v0.125.0 guard).
//
// The userdata subtree is deliberately captured at its ROOT, not per-bind: the .fab manifest keys
// HDD tars by basename, and the import side maps a basename either to a resolved ${HDD_PATH} mount
// or to <HDD_PATH>/<basename> — "userdata" round-trips through that mapping exactly, while a
// nested bind like ${USERDATA_PATH}/media/tv would base to "tv" and restore to the wrong place.
// The root also covers sibling dirs the app created beyond its declared binds (same philosophy as
// the tier-2 namespace-wholesale copy).
//
// Dedupe is containment-aware in both directions: the userdata root is skipped when an HDD mount
// already covers it (an app binding ${HDD_PATH} itself), and HDD mounts inside the userdata root
// are dropped when the root is added (a literal ${HDD_PATH}/userdata/x bind would otherwise
// double-tar and basename-collide with the root).
// R-203 — nsRoot is the app's felhom-data NAMESPACE ROOT, which is what UserdataDir takes. It is
// NOT hddPath: identical on an enrolled drive, one segment shorter on the system-data fallback.
//
// SCOPE NOTE, because this function lives in delete.go and that is misleading: it is EXPORT-only.
// Its single production caller is the .fab export adapter (cmd/controller/main.go, exportAdapter.
// GetStackDataMounts). Nothing deletes based on this result. The delete path's own guard,
// ProtectedHDDPaths above, is layout-agnostic by construction — it protects BOTH <hdd>/... and
// <hdd>/felhom-data/... — so it was never affected by the namespace-root defect.
func ExportDataMounts(composePath, hddPath, nsRoot string) []string {
if hddPath == "" {
return nil
}
hddMounts := ParseComposeHDDMounts(composePath, hddPath)
if nsRoot == "" {
nsRoot = hddPath // an enrolled drive, or a caller with nothing better — the pre-R-203 shape
}
ud := appbackup.UserdataDir(filepath.Clean(nsRoot))
if len(ParseComposeUserdataMounts(composePath, ud)) == 0 {
return hddMounts
}
for _, m := range hddMounts {
if m == ud || strings.HasPrefix(ud, m+string(filepath.Separator)) {
// an HDD mount already covers the userdata root — nothing to add
return hddMounts
}
}
mounts := make([]string, 0, len(hddMounts)+1)
for _, m := range hddMounts {
if strings.HasPrefix(m, ud+string(filepath.Separator)) {
continue // inside the userdata root — the root tar covers it
}
mounts = append(mounts, m)
}
return append(mounts, ud)
}
// ParseComposeHDDMounts reads a docker-compose.yml and extracts host paths
// that reference the HDD path from volume bind mounts.
func ParseComposeHDDMounts(composePath, hddPath string) []string {
if hddPath == "" {
return nil
}
data, err := os.ReadFile(composePath)
if err != nil {
return nil
}
var mounts []string
seen := make(map[string]bool)
scanner := bufio.NewScanner(strings.NewReader(string(data)))
inVolumes := false
for scanner.Scan() {
line := strings.TrimSpace(scanner.Text())
// Track when we're in a volumes section (service-level, not top-level)
if strings.HasPrefix(line, "volumes:") {
inVolumes = true
continue
}
if inVolumes && !strings.HasPrefix(line, "-") && !strings.HasPrefix(line, "#") && line != "" {
inVolumes = false
}
if !inVolumes || !strings.HasPrefix(line, "- ") {
continue
}
// Parse bind mount: "- /host/path:/container/path:options"
mountStr := strings.TrimPrefix(line, "- ")
mountStr = strings.Trim(mountStr, "\"'")
parts := strings.SplitN(mountStr, ":", 3)
if len(parts) < 2 {
continue
}
hostPath := parts[0]
// Resolve ${HDD_PATH} variable reference
hostPath = strings.ReplaceAll(hostPath, "${HDD_PATH}", hddPath)
// C10: Clean path BEFORE prefix check to prevent traversal like ${HDD_PATH}/../../etc/passwd.
cleanPath := filepath.Clean(hostPath)
cleanHDD := filepath.Clean(hddPath)
// Check if this is an HDD mount (must be cleanHDD itself or a direct subpath)
if cleanPath != cleanHDD && !strings.HasPrefix(cleanPath, cleanHDD+string(filepath.Separator)) {
continue
}
if !seen[cleanPath] {
seen[cleanPath] = true
mounts = append(mounts, cleanPath)
}
}
log.Printf("[INFO] [stacks] ParseComposeHDDMounts: found %d HDD mounts for %s", len(mounts), composePath)
return mounts
}
// getDirSizeHuman returns a human-readable size string for a directory using du.
func getDirSizeHuman(path string) string {
cmd := exec.Command("du", "-sh", path)
output, err := cmd.Output()
if err != nil {
return "unknown"
}
fields := strings.Fields(string(output))
if len(fields) > 0 {
return fields[0]
}
return "unknown"
}
// getDirSizeBytes returns the total size in bytes for a directory.
func getDirSizeBytes(path string) int64 {
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
cmd := exec.CommandContext(ctx, "du", "-sb", path)
output, err := cmd.Output()
if err != nil {
return 0
}
fields := strings.Fields(string(output))
if len(fields) > 0 {
var size int64
if n, _ := fmt.Sscanf(fields[0], "%d", &size); n != 1 {
return 0
}
return size
}
return 0
}
// projectVolumes lists the named volumes Docker holds for a compose project (by its project label).
// A listing failure reads as no volumes, so a removal never fails on bookkeeping.
func (m *Manager) projectVolumes(project string) []string {
out, err := m.execCommand("docker", "volume", "ls", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Name}}")
if err != nil {
return nil
}
var vols []string
for _, l := range strings.Split(out, "\n") {
if l = strings.TrimSpace(l); l != "" {
vols = append(vols, l)
}
}
return vols
}
// appVolumeSet is every volume the removal accounts for: the ones carrying the project label AND the
// ones the app's definition declares that Docker holds by name (R-658, v0.268.0). A restore before
// v0.268.0 recreated volumes without the label, and the remove then answered `volumes_removed: []`
// over volumes compose's `down --volumes` did remove (measured at the 2026-09-23 night's teardown).
// A listing failure reads as nothing, so a removal never fails on bookkeeping.
func (m *Manager) appVolumeSet(project, stackDir string) []string {
set := map[string]bool{}
for _, v := range m.projectVolumes(project) {
set[v] = true
}
if declared, _, err := DeclaredVolumeNames(ComposePathIn(stackDir)); err == nil && len(declared) > 0 {
if out, lerr := m.execCommand("docker", "volume", "ls", "-q"); lerr == nil {
have := map[string]bool{}
for _, l := range strings.Split(out, "\n") {
have[strings.TrimSpace(l)] = true
}
for _, d := range declared {
if have[d] {
set[d] = true
}
}
}
}
out := make([]string, 0, len(set))
for v := range set {
out = append(out, v)
}
sort.Strings(out)
return out
}
// removedVolumes is before minus after, as a non-nil slice (the JSON must read `[]`, not `null`).
func removedVolumes(before, after []string) []string {
still := map[string]bool{}
for _, v := range after {
still[v] = true
}
removed := []string{}
for _, v := range before {
if !still[v] {
removed = append(removed, v)
}
}
return removed
}