v0.262.0: six defects two drill nights found in the update, remove and hold paths
gates / gates (push) Successful in 26s
gates / gates (push) Successful in 26s
R-630 (P1): waitUpdateHealthy kept the probe inside `if hc != nil && len(hc.Checks) > 0`, and when findProbeContainer returned "" its else set last="no probe container" and LOOPED - the settle path sat in the outer else, unreachable. So verifying could only time out and failAndHold then stopped a working app. Measured on paperless-ngx: three containers healthy, failed at +313.0s, front door 404 after. It now falls through to the same settle path with a WARN naming the candidates. The probe target is decidable now: HealthCheckConfig.Container plus findProbeContainerMeta resolve by exact stack name -> explicit container -> a UNIQUE prefix -> nothing with the candidates returned. The old rule took the FIRST prefix match. A skipped stack records why instead of silence. R-634 (half): RemoveStack refused on the !Deployed FLAG while the machine had containers, a compose file and an app.yaml. It now asks whether anything EXISTS. The mechanism producing the bad record is still not diagnosed and R-634 stays open for it. R-633/R-626: RemoveStack consults UpdateGuards.Busy and IsUpdating and refuses with the app's own sentence - the product already refused this clash for update and for restore. And because `down` returning 0 is a request not a result, the project is watched for 25s afterwards, anything carrying its label is removed by name with its labels logged, and the answer carries `verified`. R-621: failAndHold writes compose logs --tail 400 into <stackdir>/hold-logs/<ts>/ BEFORE the down that destroys them. Two existing tests pin the compose sequence and correctly caught the new step; their expectations are updated with the reason that the ORDER is the assertion. R-614: RemoveStack calls ClearUpdateState. NOT in this release: R-625 (a held app still renders an Update button). Named, not half-done. Three new sentences, each born as a key in both bundles. Four red-proofs seen failing. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -12,6 +12,7 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
)
|
||||
|
||||
// felhomDataDir matches backup.FelhomDataDir — duplicated to avoid circular import via StackDataProvider.
|
||||
@@ -45,6 +46,11 @@ type RemoveResponse struct {
|
||||
HDDNote string `json:"hdd_note,omitempty"`
|
||||
BackupPathsRemoved []string `json:"backup_paths_removed,omitempty"`
|
||||
BackupPathsRefused []string `json:"backup_paths_refused,omitempty"`
|
||||
// Verified says the teardown was CHECKED, not just requested (R-626/R-633). False means a
|
||||
// container carrying this project's compose label was still there after the watch window — the
|
||||
// answer that used to be a silent 200.
|
||||
Verified bool `json:"verified"`
|
||||
ReappearedRemoved []string `json:"reappeared_removed,omitempty"`
|
||||
}
|
||||
|
||||
// RemoveRefusedError is a removal REFUSED before anything was touched: the customer asked for the
|
||||
@@ -413,6 +419,77 @@ func (m *Manager) GetStackHDDData(name string) (*HDDDataResponse, error) {
|
||||
// volumes, optionally removes HDD data and backup data, then removes app.yaml
|
||||
// so the stack reverts to "not deployed" state. The template files (docker-compose.yml,
|
||||
// .felhom.yml) are preserved so the user can redeploy.
|
||||
// removeVerifyWindow is how long the project is watched after `down` before the teardown is called
|
||||
// verified. 20 s is the floor the brief sets; the measured re-creation happened at +2 s.
|
||||
const removeVerifyWindow = 25 * time.Second
|
||||
|
||||
// projectContainersByLabel lists containers still carrying this compose project's label — including
|
||||
// stopped ones, because a container that exists at all is one the household can still see.
|
||||
func (m *Manager) projectContainersByLabel(project string) []string {
|
||||
out, err := m.execCommand("docker", "ps", "-a", "--filter", "label=com.docker.compose.project="+project, "--format", "{{.Names}}")
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var names []string
|
||||
for _, l := range strings.Split(out, "\n") {
|
||||
if l = strings.TrimSpace(l); l != "" {
|
||||
names = append(names, l)
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// verifyTornDown watches the compose project after `down` and removes anything that comes back.
|
||||
// Returns whether the project was clean at the end, and what had to be removed.
|
||||
func (m *Manager) verifyTornDown(name string, window time.Duration) (bool, []string) {
|
||||
deadline := m.now().Add(window)
|
||||
var removed []string
|
||||
for {
|
||||
left := m.projectContainersByLabel(name)
|
||||
for _, c := range left {
|
||||
lbl, _ := m.execCommand("docker", "inspect", c, "--format", "{{json .Config.Labels}}")
|
||||
m.logger.Printf("[WARN] [stacks] RemoveStack %s: container %q reappeared after `down` — removing it by name; its labels: %s", name, c, truncateStr(strings.TrimSpace(lbl), 300))
|
||||
if out, err := m.execCommand("docker", "rm", "-f", c); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] RemoveStack %s: could not remove the reappeared container %q: %v (%s)", name, c, err, truncateStr(out, 160))
|
||||
} else {
|
||||
removed = append(removed, c)
|
||||
}
|
||||
}
|
||||
if !m.now().Before(deadline) {
|
||||
break
|
||||
}
|
||||
time.Sleep(2 * time.Second)
|
||||
}
|
||||
still := m.projectContainersByLabel(name)
|
||||
if len(still) > 0 {
|
||||
m.logger.Printf("[ERROR] [stacks] RemoveStack %s: NOT verified — %d container(s) still carry this project's label after %s: %v", name, len(still), window, still)
|
||||
return false, removed
|
||||
}
|
||||
return true, removed
|
||||
}
|
||||
|
||||
// KeyRemoveBusy is the one sentence R-633 adds: a remove refused because the backup side owns the app.
|
||||
const KeyRemoveBusy = "err.stacks.az_alkalmazason_mentes_vagy_visszaallitas_fut"
|
||||
|
||||
// halfStateEvidence answers "does anything of this stack actually EXIST?" for a stack the record
|
||||
// says is not deployed (R-634). Containers first, because that is the shape that hurt: an app
|
||||
// serving traffic that no button could remove.
|
||||
func (m *Manager) halfStateEvidence(name string, stack *Stack) (bool, string) {
|
||||
if len(stack.Containers) > 0 {
|
||||
return true, fmt.Sprintf("%d container(s) exist", len(stack.Containers))
|
||||
}
|
||||
if stack.ComposePath != "" {
|
||||
if _, err := os.Stat(stack.ComposePath); err == nil {
|
||||
dir := filepath.Dir(stack.ComposePath)
|
||||
if _, err := os.Stat(filepath.Join(dir, "app.yaml")); err == nil {
|
||||
return true, "a compose file and an app.yaml exist on disk"
|
||||
}
|
||||
return true, "a compose file exists on disk"
|
||||
}
|
||||
}
|
||||
return false, ""
|
||||
}
|
||||
|
||||
func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemove []string) (*RemoveResponse, error) {
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] [stacks] RemoveStack called: name=%q, removeHDDData=%v, backupPathsToRemove=%d", name, removeHDDData, len(backupPathsToRemove))
|
||||
@@ -433,9 +510,20 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
|
||||
name, stack.State, stack.Deployed, stack.Orphaned, stack.Deploying)
|
||||
}
|
||||
|
||||
// Must be deployed
|
||||
// R-634: `deployed` is a RECORD, and the record can be wrong while the machine is right. Three
|
||||
// apps were measured running, healthy and serving with `deployed=false` — `outline` answering its
|
||||
// own `/_health` with 200 and three containers up — and in that state BOTH remove calls answered
|
||||
// `stack "x" is not deployed`, so the household had no button at all and a shell was the only
|
||||
// exit. **The household must always be able to remove what the box shows them.** So the refusal
|
||||
// now asks whether anything EXISTS, not whether a flag is set: containers, a compose file, or an
|
||||
// app.yaml are each enough. Everything downstream already copes — the removal is driven by the
|
||||
// compose file and the directory, not by the flag.
|
||||
if !stack.Deployed {
|
||||
return nil, fmt.Errorf("stack %q is not deployed", name)
|
||||
half, why := m.halfStateEvidence(name, stack)
|
||||
if !half {
|
||||
return nil, fmt.Errorf("stack %q is not deployed", name)
|
||||
}
|
||||
m.logger.Printf("[WARN] [stacks] RemoveStack %s: deployed=false but %s — removing what exists (R-634)", name, why)
|
||||
}
|
||||
|
||||
// Must not be deploying (H2 fix)
|
||||
@@ -443,6 +531,23 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
|
||||
return nil, fmt.Errorf("stack %q is currently being deployed — wait for deployment to finish", name)
|
||||
}
|
||||
|
||||
// R-633: the backup side owns operations this package cannot see. A remove sent while a RESTORE
|
||||
// was in flight was measured tearing down what existed while the restore's own `compose up`
|
||||
// re-created it — both calls returned success, the record said `deployed: false`, and a container
|
||||
// went on restarting for hours with a live traefik route. The product already refuses exactly
|
||||
// this clash for `update` and for `restore`, and names the blocking operation; `remove` did not
|
||||
// consult it at all. Same guard, same adapter.
|
||||
if g := m.guards(); g != nil {
|
||||
if busy, why := g.Busy(name); busy {
|
||||
m.logger.Printf("[ERROR] [stacks] RemoveStack %s REFUSED (busy): %s", name, why)
|
||||
return nil, util.MsgError(KeyRemoveBusy)
|
||||
}
|
||||
}
|
||||
if m.IsUpdating(name) {
|
||||
m.logger.Printf("[ERROR] [stacks] RemoveStack %s REFUSED (busy): a guarded update is in progress", name)
|
||||
return nil, util.MsgError(KeyRemoveBusy)
|
||||
}
|
||||
|
||||
// Must be stopped (not running)
|
||||
// StateDegraded (R-51) counts as running here: a degraded stack still has LIVE containers, and
|
||||
// deleting its directory out from under them would leave orphans behind.
|
||||
@@ -491,6 +596,18 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
|
||||
return resp, fmt.Errorf("docker compose down failed for %s: %w", name, err)
|
||||
}
|
||||
|
||||
// Step 2b: WATCH, then say so. R-626 recorded a removed `navidrome` coming back and could not
|
||||
// diagnose it; R-633 caught the same shape with the window visible — a restore's own `compose
|
||||
// up` re-creating what `down` had just torn out, seventeen seconds apart, while both calls
|
||||
// returned success. `down` returning 0 is a request, not a result. So the project is watched for
|
||||
// a bounded window and anything that reappears is removed BY NAME and logged with the label that
|
||||
// created it, and the answer carries whether the check passed.
|
||||
resp.Verified, resp.ReappearedRemoved = m.verifyTornDown(name, removeVerifyWindow)
|
||||
|
||||
// R-614: the app is going; its update record goes with it. Otherwise the NEXT install of the
|
||||
// same name inherits a phase that belongs to an app that no longer exists.
|
||||
m.ClearUpdateState(name)
|
||||
|
||||
// Step 3: the volumes that are gone now — `[]` when none, never null (R-489).
|
||||
resp.VolumesRemoved = removedVolumes(volsBefore, m.projectVolumes(name))
|
||||
if len(resp.VolumesRemoved) > 0 {
|
||||
|
||||
Reference in New Issue
Block a user