v0.263.1: the undo asks the old probe even on an app marked unhealthy (R-637)
gates / gates (push) Successful in 27s
gates / gates (push) Successful in 27s
Found live on 9202: the periodic probe (current .felhom.yml, new port) flips the app to unhealthy, and the update's health wait probed only 'running' apps - so the undo's old probe was never asked and a serving old version was judged "did not start". With the undo's override, an unhealthy app is probed and the old check decides; never settled on container state. New seam probeRunFn; the test drives the real wait loop and reproduces the live message when the fix is switched off. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -918,11 +918,22 @@ func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeou
|
||||
for {
|
||||
_ = m.RefreshStatus()
|
||||
st, ok := m.GetStack(name)
|
||||
// v0.263.0 — THE UNDO'S PROBE MUST BE ALLOWED TO RUN. The periodic health probe judges the app
|
||||
// with the CURRENT .felhom.yml and flips a running app to StateUnhealthy when that check fails —
|
||||
// which is exactly the undo's situation when the new version brought a probe the old one does
|
||||
// not answer. Gating on StateRunning alone meant the old probe was never asked and the undo was
|
||||
// reported "did not start" after the full timeout. MEASURED LIVE on 9202 2026-09-23 (docmost,
|
||||
// `last: state unhealthy` for 90 s while the old version served). So with an override that
|
||||
// declares checks, an Unhealthy app is PROBED — and the override's own check decides. It never
|
||||
// settles an Unhealthy app on container state (below: the settle path requires StateRunning).
|
||||
// Pinned by TestUndo_OldProbeRunsOnAnAppTheCurrentProbeMarkedUnhealthy.
|
||||
undoProbe := ok && override != nil && st.State == StateUnhealthy &&
|
||||
override.HealthCheck != nil && len(override.HealthCheck.Checks) > 0
|
||||
switch {
|
||||
case !ok:
|
||||
last = "stack vanished"
|
||||
runningSince = time.Time{}
|
||||
case st.State == StateRunning:
|
||||
case st.State == StateRunning || undoProbe:
|
||||
// A declared health check is only usable if it resolves to a container. When it does
|
||||
// not, the app is judged the same way an app with NO declared check is judged —
|
||||
// settling on container state — and the log says so.
|
||||
@@ -942,7 +953,7 @@ func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeou
|
||||
c, candidates := findProbeContainerMeta(name, &meta, st.Containers)
|
||||
if c != "" {
|
||||
usable = true
|
||||
res := m.runChecks(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
||||
res := m.probeRun(probeTarget{stackName: name, containerName: c, checks: hc.Checks})
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.HealthProbe = res
|
||||
@@ -958,7 +969,9 @@ func (m *Manager) waitUpdateHealthyMeta(ctx context.Context, name string, timeou
|
||||
name, name, candidates)
|
||||
}
|
||||
}
|
||||
if !usable {
|
||||
if !usable && st.State != StateRunning {
|
||||
last = "unhealthy, and the check resolves to no container"
|
||||
} else if !usable {
|
||||
if runningSince.IsZero() {
|
||||
runningSince = m.now()
|
||||
}
|
||||
@@ -1213,3 +1226,13 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
||||
}
|
||||
return len(names)
|
||||
}
|
||||
|
||||
// probeRun is the network half of the update's health wait — its own seam, so a test can drive the
|
||||
// REAL wait loop (the state gate, the container resolution, the settle rule) with only the HTTP/TCP
|
||||
// probe faked. nil ⇒ runChecks.
|
||||
func (m *Manager) probeRun(t probeTarget) *HealthProbeResult {
|
||||
if m.probeRunFn != nil {
|
||||
return m.probeRunFn(t)
|
||||
}
|
||||
return m.runChecks(t)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user