R-772: a health probe that could not run records not_checked + NOT healthy, is looked at again on the 10 s cycle, and the app page says so

The state stays the containers' (probeSaysUnhealthy skips a not-checked record), so R-630 holds.
Red-proof RP-D1 — felhom.eu audits/visitors-2026-10-01/D.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-01 21:01:42 +02:00
parent 1e8d045815
commit b53721db34
7 changed files with 721 additions and 5 deletions
+7 -3
View File
@@ -82,13 +82,17 @@ func (m *Manager) RunHealthProbes() error {
for _, np := range noProbe {
m.mu.Lock()
if s, ok := m.stacks[np.name]; ok {
// R-772: NOT healthy — no check ran. R-630's point still holds through NotChecked: the state is read from
// the containers, never forced to unhealthy by this record. Not healthy also means the 10-second fast
// cycle, so a container that comes back is checked within seconds, not 5 minutes later.
s.HealthProbe = &HealthProbeResult{
Healthy: true, // never RED on our own inability to look — R-630's whole point
LastCheck: m.now(),
Healthy: false,
NotChecked: true,
LastCheck: m.now(),
Details: []HealthCheckDetail{{
Type: "none",
Target: np.name,
Healthy: true,
Healthy: false,
Error: MsgHealthNoProbeContainer,
MessageKey: KeyHealthNoProbeContainer,
}},
+12 -1
View File
@@ -114,10 +114,21 @@ type ContainerInfo struct {
}
// HealthProbeResult holds the latest controller-side health probe result.
// probeSaysUnhealthy: a probe that RAN and failed overrides a running state to unhealthy; a not-checked record (R-772)
// does not — the containers decide. Pinned by TestProbeSaysUnhealthy_NotCheckedLeavesTheStateAlone.
func probeSaysUnhealthy(hp *HealthProbeResult) bool {
return hp != nil && !hp.Healthy && !hp.NotChecked
}
type HealthProbeResult struct {
Healthy bool `json:"healthy"`
LastCheck time.Time `json:"last_check"`
Details []HealthCheckDetail `json:"details"`
// NotChecked: the declared check could not run — no container to probe (R-772). Healthy is then FALSE: a record of a
// check that did not happen never says healthy (presence is not success). The stack's STATE is still read from its
// containers (manager.go's override skips a not-checked record), so a running app is never painted unhealthy
// because the box could not look. Pinned by TestRunHealthProbes_NoContainerIsNotHealthy.
NotChecked bool `json:"not_checked,omitempty"`
}
// HealthCheckDetail holds the result of a single health check item.
@@ -812,7 +823,7 @@ func (m *Manager) refreshStatusLocked() error {
// Re-apply controller-side health probe results: if the last probe
// failed and Docker thinks the container is running, override to unhealthy.
if stack.State == StateRunning && stack.HealthProbe != nil && !stack.HealthProbe.Healthy {
if stack.State == StateRunning && probeSaysUnhealthy(stack.HealthProbe) {
stack.State = StateUnhealthy
}
@@ -0,0 +1,80 @@
package stacks
import (
"io"
"log"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
)
// R-772 — a health probe that finds NO container to probe recorded `healthy: true` and then waited 5 minutes (measured
// 2026-10-01 on 9202, Karakeep with its container stopped). A record of a check that did not run must never say healthy
// (presence is not success), and the box must look again soon.
func r772Manager(now time.Time) *Manager {
m := &Manager{cfg: &config.Config{}, logger: log.New(io.Discard, "", 0), stacks: map[string]*Stack{}}
m.updateNowFn = func() time.Time { return now }
return m
}
func r772Stack() *Stack {
return &Stack{
Name: "karakeep",
State: StateDegraded,
Meta: Metadata{HealthCheck: &HealthCheckConfig{Interval: "5m",
Checks: []HealthCheckItem{{Type: "http", Port: 3000, Path: "/api/health"}}}},
// the app container stopped; the others still run — nothing the probe may use
Containers: []ContainerInfo{
{Name: "karakeep", State: StateExited},
{Name: "karakeep-meilisearch", State: StateRunning},
{Name: "karakeep-chrome", State: StateRunning},
},
}
}
func TestRunHealthProbes_NoContainerIsNotHealthy(t *testing.T) {
m := r772Manager(time.Now())
m.stacks["karakeep"] = r772Stack()
if err := m.RunHealthProbes(); err != nil {
t.Fatal(err)
}
hp := m.stacks["karakeep"].HealthProbe
if hp == nil {
t.Fatal("no record written")
}
if hp.Healthy || !hp.NotChecked || len(hp.Details) != 1 || hp.Details[0].Healthy || hp.Details[0].MessageKey != KeyHealthNoProbeContainer {
t.Fatalf("a check that did not run must read NOT healthy + not_checked, got %+v", hp)
}
if m.stacks["karakeep"].State != StateDegraded {
t.Fatalf("the record must not change the state the containers gave, got %s", m.stacks["karakeep"].State)
}
}
// The box looks again on the FAST cycle (10 s), not after the 5-minute interval: 20 s later a second run re-records.
// Pre-fix (healthy: true) the second run skipped it for 5 minutes.
func TestRunHealthProbes_NotCheckedIsLookedAtAgainSoon(t *testing.T) {
first := time.Now().Add(-20 * time.Second)
m := r772Manager(first)
m.stacks["karakeep"] = r772Stack()
_ = m.RunHealthProbes()
m.updateNowFn = func() time.Time { return time.Now() }
_ = m.RunHealthProbes()
if got := m.stacks["karakeep"].HealthProbe.LastCheck; !got.After(first) {
t.Fatalf("a not-checked app must be looked at again within the 10-second cycle; last check still %v", got)
}
}
// R-630 stays true: a not-checked record never paints a RUNNING app unhealthy; a probe that ran and failed does.
func TestProbeSaysUnhealthy_NotCheckedLeavesTheStateAlone(t *testing.T) {
if probeSaysUnhealthy(&HealthProbeResult{Healthy: false, NotChecked: true}) {
t.Fatal("not checked must not override the state")
}
if !probeSaysUnhealthy(&HealthProbeResult{Healthy: false}) {
t.Fatal("a failed check must override running → unhealthy")
}
if probeSaysUnhealthy(nil) || probeSaysUnhealthy(&HealthProbeResult{Healthy: true}) {
t.Fatal("nil / healthy must not override")
}
}