v0.262.0: six defects two drill nights found in the update, remove and hold paths
gates / gates (push) Successful in 26s
gates / gates (push) Successful in 26s
R-630 (P1): waitUpdateHealthy kept the probe inside `if hc != nil && len(hc.Checks) > 0`, and when findProbeContainer returned "" its else set last="no probe container" and LOOPED - the settle path sat in the outer else, unreachable. So verifying could only time out and failAndHold then stopped a working app. Measured on paperless-ngx: three containers healthy, failed at +313.0s, front door 404 after. It now falls through to the same settle path with a WARN naming the candidates. The probe target is decidable now: HealthCheckConfig.Container plus findProbeContainerMeta resolve by exact stack name -> explicit container -> a UNIQUE prefix -> nothing with the candidates returned. The old rule took the FIRST prefix match. A skipped stack records why instead of silence. R-634 (half): RemoveStack refused on the !Deployed FLAG while the machine had containers, a compose file and an app.yaml. It now asks whether anything EXISTS. The mechanism producing the bad record is still not diagnosed and R-634 stays open for it. R-633/R-626: RemoveStack consults UpdateGuards.Busy and IsUpdating and refuses with the app's own sentence - the product already refused this clash for update and for restore. And because `down` returning 0 is a request not a result, the project is watched for 25s afterwards, anything carrying its label is removed by name with its labels logged, and the answer carries `verified`. R-621: failAndHold writes compose logs --tail 400 into <stackdir>/hold-logs/<ts>/ BEFORE the down that destroys them. Two existing tests pin the compose sequence and correctly caught the new step; their expectations are updated with the reason that the ORDER is the assertion. R-614: RemoveStack calls ClearUpdateState. NOT in this release: R-625 (a held app still renders an Update button). Named, not half-done. Three new sentences, each born as a key in both bundles. Four red-proofs seen failing. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -178,6 +178,26 @@ func (m *Manager) currentDeployTime(name string) time.Time {
|
||||
return t
|
||||
}
|
||||
|
||||
// ClearUpdateState wipes everything a finished-or-abandoned update left on a stack. R-614: a fresh
|
||||
// install of an app that had previously failed an update showed the OLD phase — „Frissítve" on a
|
||||
// deploy that had just happened — because a remove cleared the directory and not the in-memory
|
||||
// record. The name is the only thing the next install shares with the last one, so the record has to
|
||||
// go when the app does.
|
||||
func (m *Manager) ClearUpdateState(name string) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.stacks[name]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
s.Updating = false
|
||||
s.UpdatePhase = ""
|
||||
s.UpdatePhaseLabel = ""
|
||||
s.UpdateError = ""
|
||||
s.updateHeld = false
|
||||
s.HealthProbe = nil
|
||||
}
|
||||
|
||||
func (m *Manager) markUpdateHeld(name string) {
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
@@ -717,9 +737,40 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
|
||||
m.logger.Printf("[INFO] [stacks] update %s: DONE in %s", name, m.now().Sub(start).Round(time.Second))
|
||||
}
|
||||
|
||||
// holdLogTailLines is how much of each service's log the hold keeps. 400 lines is enough to hold a
|
||||
// migration run and a startup failure, and small enough that a hold never fills a disk.
|
||||
const holdLogTailLines = "400"
|
||||
|
||||
// captureHoldLogs writes each service's log into <stackdir>/hold-logs/<ts>/ before the app is
|
||||
// stopped. Best-effort by design (R-621): the hold itself must happen either way.
|
||||
func (m *Manager) captureHoldLogs(name, dir string, env []string) {
|
||||
ts := m.now().UTC().Format("20060102T150405Z")
|
||||
outDir := filepath.Join(dir, "hold-logs", ts)
|
||||
if err := os.MkdirAll(outDir, 0o755); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: cannot create the hold-log directory %s: %v — the hold still proceeds", name, outDir, err)
|
||||
return
|
||||
}
|
||||
out, err := m.updateCompose(dir, env, "logs", "--no-color", "--tail", holdLogTailLines)
|
||||
if err != nil {
|
||||
m.logger.Printf("[WARN] [stacks] update %s: `compose logs` before the hold failed: %v — writing what came back anyway", name, err)
|
||||
}
|
||||
path := filepath.Join(outDir, "compose-logs.txt")
|
||||
if werr := os.WriteFile(path, []byte(out), 0o644); werr != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: could not write %s: %v", name, path, werr)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: kept %d bytes of the app's own log at %s before stopping it (R-621)", name, len(out), path)
|
||||
}
|
||||
|
||||
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
|
||||
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
|
||||
// R-621: capture the app's own logs BEFORE the `down`, because the `down` destroys them. Two
|
||||
// drill nights lost the only evidence of WHY an update failed this way — `adventurelog` ran nine
|
||||
// migrations and then never bound its port, and the log that would have said so was gone by the
|
||||
// time anyone looked. The capture is bounded and best-effort: a hold must never fail because its
|
||||
// evidence could not be written.
|
||||
m.captureHoldLogs(name, dir, env)
|
||||
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
||||
}
|
||||
@@ -769,12 +820,23 @@ func (m *Manager) removePreUpdateCopies(dir string) {
|
||||
_ = os.Remove(filepath.Join(dir, preUpdateAppliedFile))
|
||||
}
|
||||
|
||||
// settleReason names WHY the wait fell back to container state, so the journal and the log do not
|
||||
// have to be read together to tell "this app declares no check" from "this app declares one that
|
||||
// resolves to no container" (R-630).
|
||||
func settleReason(meta Metadata) string {
|
||||
if hc := meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
||||
return "no probe container — settled on container state"
|
||||
}
|
||||
return "no health check declared"
|
||||
}
|
||||
|
||||
// waitUpdateHealthy is the production health wait: the app's own .felhom.yml health check through the
|
||||
// existing probe, or — for an app with none — every container running and none restarting for
|
||||
// updateSettleWindow. NEVER the compose exit code, and never logPostStartStatus's delayed log line.
|
||||
func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout time.Duration) (bool, string) {
|
||||
deadline := m.now().Add(timeout)
|
||||
var runningSince time.Time
|
||||
warnedNoProbe := false
|
||||
last := "no observation yet"
|
||||
for {
|
||||
_ = m.RefreshStatus()
|
||||
@@ -784,8 +846,21 @@ func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout ti
|
||||
last = "stack vanished"
|
||||
runningSince = time.Time{}
|
||||
case st.State == StateRunning:
|
||||
// A declared health check is only usable if it resolves to a container. When it does
|
||||
// not, the app is judged the same way an app with NO declared check is judged —
|
||||
// settling on container state — and the log says so.
|
||||
//
|
||||
// R-630, and this `else` is the whole defect: the old code set `last = "no probe
|
||||
// container"` and looped, so `verifying` spent the FULL `update.health_timeout` and
|
||||
// `failAndHold` then STOPPED an app whose containers were all healthy. Measured on
|
||||
// paperless-ngx 2026-09-22: `done` was never reachable, `failed` at +313.0 s, front door
|
||||
// 404 afterwards. A stack with no probe is not "healthy" and it is not "failing" — it is
|
||||
// SETTLED ON CONTAINER STATE (`09` §3), and never a reason to stop a running app.
|
||||
usable := false
|
||||
if hc := st.Meta.HealthCheck; hc != nil && len(hc.Checks) > 0 {
|
||||
if c := findProbeContainer(name, st.Containers); c != "" {
|
||||
c, candidates := findProbeContainerMeta(name, &st.Meta, st.Containers)
|
||||
if c != "" {
|
||||
usable = true
|
||||
res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
@@ -796,15 +871,19 @@ func (m *Manager) waitUpdateHealthy(ctx context.Context, name string, timeout ti
|
||||
return true, "the app's health check passed"
|
||||
}
|
||||
last = "health check failing"
|
||||
} else {
|
||||
last = "no probe container"
|
||||
} else if !warnedNoProbe {
|
||||
warnedNoProbe = true
|
||||
m.logger.Printf("[WARN] [stacks] update %s: no probe container for %s — settling on container state instead; candidates: %v",
|
||||
name, name, candidates)
|
||||
}
|
||||
} else {
|
||||
}
|
||||
if !usable {
|
||||
if runningSince.IsZero() {
|
||||
runningSince = m.now()
|
||||
}
|
||||
if m.now().Sub(runningSince) >= updateSettleWindow {
|
||||
return true, fmt.Sprintf("all containers running, none restarting, for %s (no health check declared)", updateSettleWindow)
|
||||
return true, fmt.Sprintf("all containers running, none restarting, for %s (%s)",
|
||||
updateSettleWindow, settleReason(st.Meta))
|
||||
}
|
||||
last = "running, settling"
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user