R-772 (found live on 9202 with v0.286.0): a stopped probe container is seen on the next tick, not when the last healthy record's 5-minute interval runs out
Finding the container costs no network call, so it now runs before the interval check. Red-proof RP-D1b. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -38,6 +38,16 @@ func (m *Manager) RunHealthProbes() error {
|
|||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// R-772 (measured live on 9202, v0.286.0): a container that stops is seen at once, not when the interval of the
|
||||||
|
// last HEALTHY result runs out (up to 5 minutes later) — finding the container costs no network call, so it is
|
||||||
|
// done before the interval check, and a stack with nothing to probe is recorded "not checked" on this tick.
|
||||||
|
containerName, candidates := findProbeContainerMeta(name, &stack.Meta, stack.Containers)
|
||||||
|
if containerName == "" {
|
||||||
|
skippedNoContainer++
|
||||||
|
noProbe = append(noProbe, noProbeStack{name: name, candidates: candidates})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
// Check if interval has elapsed since last probe.
|
// Check if interval has elapsed since last probe.
|
||||||
// Fast 10s probes until healthy, then normal interval (default 5m).
|
// Fast 10s probes until healthy, then normal interval (default 5m).
|
||||||
// When HealthProbe is nil (just started/restarted), probe immediately.
|
// When HealthProbe is nil (just started/restarted), probe immediately.
|
||||||
@@ -58,16 +68,8 @@ func (m *Manager) RunHealthProbes() error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Find the main container to probe. R-630: a stack whose check resolves to no container
|
// R-630: a stack whose check resolves to no container used to be skipped SILENTLY — paperless-ngx's probe had
|
||||||
// used to be skipped SILENTLY — paperless-ngx's probe had therefore never run on any box,
|
// therefore never run on any box. It gets a RESULT that says why (above, before the interval check).
|
||||||
// and nothing on any screen could say so. It now gets a RESULT that says why, so the app
|
|
||||||
// page can tell "checked and healthy" from "never checked".
|
|
||||||
containerName, candidates := findProbeContainerMeta(name, &stack.Meta, stack.Containers)
|
|
||||||
if containerName == "" {
|
|
||||||
skippedNoContainer++
|
|
||||||
noProbe = append(noProbe, noProbeStack{name: name, candidates: candidates})
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
targets = append(targets, probeTarget{
|
targets = append(targets, probeTarget{
|
||||||
stackName: name,
|
stackName: name,
|
||||||
|
|||||||
@@ -78,3 +78,16 @@ func TestProbeSaysUnhealthy_NotCheckedLeavesTheStateAlone(t *testing.T) {
|
|||||||
t.Fatal("nil / healthy must not override")
|
t.Fatal("nil / healthy must not override")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Measured live on 9202 (v0.286.0): with a HEALTHY record 2 minutes old, stopping the probe's container left the record
|
||||||
|
// "healthy" until its 5-minute interval ran out. The missing container is now seen on the very next tick.
|
||||||
|
func TestRunHealthProbes_AStoppedContainerIsSeenAtOnce(t *testing.T) {
|
||||||
|
m := r772Manager(time.Now())
|
||||||
|
st := r772Stack()
|
||||||
|
st.HealthProbe = &HealthProbeResult{Healthy: true, LastCheck: time.Now().Add(-2 * time.Minute)}
|
||||||
|
m.stacks["karakeep"] = st
|
||||||
|
_ = m.RunHealthProbes()
|
||||||
|
if hp := m.stacks["karakeep"].HealthProbe; hp.Healthy || !hp.NotChecked {
|
||||||
|
t.Fatalf("a stopped probe container must be recorded not checked on the next tick, got %+v", hp)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user