b53721db34
The state stays the containers' (probeSaysUnhealthy skips a not-checked record), so R-630 holds. Red-proof RP-D1 — felhom.eu audits/visitors-2026-10-01/D. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
430 lines
13 KiB
Go
430 lines
13 KiB
Go
package stacks
|
|
|
|
import (
|
|
"fmt"
|
|
"io"
|
|
"net"
|
|
"net/http"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
)
|
|
|
|
// probeTarget holds the info needed to probe a single stack.
|
|
type probeTarget struct {
|
|
stackName string
|
|
containerName string
|
|
checks []HealthCheckItem
|
|
}
|
|
|
|
// RunHealthProbes runs controller-side health probes for all running stacks
|
|
// that have healthcheck configuration and whose interval has elapsed.
|
|
// Called by the scheduler every minute.
|
|
func (m *Manager) RunHealthProbes() error {
|
|
// Phase 1: collect targets (under lock)
|
|
var noProbe []noProbeStack
|
|
m.mu.RLock()
|
|
var targets []probeTarget
|
|
skippedNotDue := 0
|
|
skippedNoContainer := 0
|
|
for name, stack := range m.stacks {
|
|
// StateDegraded (R-51) is probed too: its live members still answer, and the probe result
|
|
// only ever overrides StateRunning below, so a degraded stack can never be masked as unhealthy.
|
|
if stack.State != StateRunning && stack.State != StateUnhealthy && stack.State != StateDegraded {
|
|
continue
|
|
}
|
|
hc := stack.Meta.HealthCheck
|
|
if hc == nil || len(hc.Checks) == 0 {
|
|
continue
|
|
}
|
|
|
|
// Check if interval has elapsed since last probe.
|
|
// Fast 10s probes until healthy, then normal interval (default 5m).
|
|
// When HealthProbe is nil (just started/restarted), probe immediately.
|
|
interval := parseInterval(hc.Interval)
|
|
if stack.HealthProbe != nil {
|
|
effectiveInterval := interval
|
|
if !stack.HealthProbe.Healthy {
|
|
effectiveInterval = 10 * time.Second
|
|
}
|
|
if time.Since(stack.HealthProbe.LastCheck) < effectiveInterval {
|
|
skippedNotDue++
|
|
if m.isDebug() {
|
|
sinceLastCheck := time.Since(stack.HealthProbe.LastCheck).Round(time.Second)
|
|
m.logger.Printf("[DEBUG] [stacks] RunHealthProbes: skipping %s — last check %s ago, effective interval %s, healthy=%v",
|
|
name, sinceLastCheck, effectiveInterval, stack.HealthProbe.Healthy)
|
|
}
|
|
continue
|
|
}
|
|
}
|
|
|
|
// Find the main container to probe. R-630: a stack whose check resolves to no container
|
|
// used to be skipped SILENTLY — paperless-ngx's probe had therefore never run on any box,
|
|
// and nothing on any screen could say so. It now gets a RESULT that says why, so the app
|
|
// page can tell "checked and healthy" from "never checked".
|
|
containerName, candidates := findProbeContainerMeta(name, &stack.Meta, stack.Containers)
|
|
if containerName == "" {
|
|
skippedNoContainer++
|
|
noProbe = append(noProbe, noProbeStack{name: name, candidates: candidates})
|
|
continue
|
|
}
|
|
|
|
targets = append(targets, probeTarget{
|
|
stackName: name,
|
|
containerName: containerName,
|
|
checks: hc.Checks,
|
|
})
|
|
}
|
|
m.mu.RUnlock()
|
|
|
|
// Record the "no probe container" stacks OUTSIDE the read lock, then write their results under
|
|
// the write lock — the same shape the probe results themselves use.
|
|
for _, np := range noProbe {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[np.name]; ok {
|
|
// R-772: NOT healthy — no check ran. R-630's point still holds through NotChecked: the state is read from
|
|
// the containers, never forced to unhealthy by this record. Not healthy also means the 10-second fast
|
|
// cycle, so a container that comes back is checked within seconds, not 5 minutes later.
|
|
s.HealthProbe = &HealthProbeResult{
|
|
Healthy: false,
|
|
NotChecked: true,
|
|
LastCheck: m.now(),
|
|
Details: []HealthCheckDetail{{
|
|
Type: "none",
|
|
Target: np.name,
|
|
Healthy: false,
|
|
Error: MsgHealthNoProbeContainer,
|
|
MessageKey: KeyHealthNoProbeContainer,
|
|
}},
|
|
}
|
|
}
|
|
m.mu.Unlock()
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] RunHealthProbes: %s has a health check but no container to probe — candidates: %v", np.name, np.candidates)
|
|
}
|
|
}
|
|
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] RunHealthProbes: collected %d targets (%d skipped not due, %d skipped no container)",
|
|
len(targets), skippedNotDue, skippedNoContainer)
|
|
}
|
|
|
|
if len(targets) == 0 {
|
|
return nil
|
|
}
|
|
|
|
// Phase 2: run all probes concurrently (no lock held)
|
|
type probeResult struct {
|
|
stackName string
|
|
result *HealthProbeResult
|
|
}
|
|
|
|
results := make([]probeResult, len(targets))
|
|
var wg sync.WaitGroup
|
|
|
|
for i, t := range targets {
|
|
wg.Add(1)
|
|
go func(idx int, t probeTarget) {
|
|
defer wg.Done()
|
|
result := m.runChecks(t)
|
|
results[idx] = probeResult{stackName: t.stackName, result: result}
|
|
}(i, t)
|
|
}
|
|
wg.Wait()
|
|
|
|
// Phase 3: apply results and log (under lock)
|
|
m.mu.Lock()
|
|
okCount, failCount := 0, 0
|
|
for _, pr := range results {
|
|
stack, ok := m.stacks[pr.stackName]
|
|
if !ok {
|
|
continue
|
|
}
|
|
stack.HealthProbe = pr.result
|
|
|
|
if pr.result.Healthy {
|
|
okCount++
|
|
// If Docker says running and probe is healthy, ensure state is running
|
|
// (clears a previous unhealthy override)
|
|
if stack.State == StateUnhealthy {
|
|
stack.State = StateRunning
|
|
}
|
|
} else {
|
|
failCount++
|
|
if stack.State == StateRunning {
|
|
stack.State = StateUnhealthy
|
|
}
|
|
}
|
|
}
|
|
m.mu.Unlock()
|
|
|
|
// Summary log
|
|
if failCount > 0 {
|
|
m.logger.Printf("[WARN] Health probes: %d ok, %d unhealthy (of %d probed)", okCount, failCount, len(targets))
|
|
} else {
|
|
m.logger.Printf("[INFO] Health probes: %d ok (of %d probed)", okCount, len(targets))
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// runChecks executes all health check items for a single stack target.
|
|
func (m *Manager) runChecks(t probeTarget) *HealthProbeResult {
|
|
result := &HealthProbeResult{
|
|
LastCheck: time.Now(),
|
|
Healthy: true,
|
|
}
|
|
|
|
for _, check := range t.checks {
|
|
detail := m.runSingleCheck(t.containerName, check)
|
|
result.Details = append(result.Details, detail)
|
|
|
|
if detail.Healthy {
|
|
if m.isDebug() {
|
|
if detail.Status > 0 {
|
|
m.logger.Printf("[DEBUG] Health probe %s: %s %s :%d%s → %d (%s)",
|
|
t.stackName, strings.ToUpper(check.Type), methodOrEmpty(check), check.Port, check.Path, detail.Status, detail.Latency)
|
|
} else {
|
|
m.logger.Printf("[DEBUG] Health probe %s: TCP :%d → ok (%s)",
|
|
t.stackName, check.Port, detail.Latency)
|
|
}
|
|
}
|
|
} else {
|
|
result.Healthy = false
|
|
m.logger.Printf("[WARN] Health probe %s: %s %s :%d%s → %s",
|
|
t.stackName, strings.ToUpper(check.Type), methodOrEmpty(check), check.Port, check.Path, detail.Error)
|
|
}
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// runSingleCheck executes one health check item and returns the result.
|
|
func (m *Manager) runSingleCheck(containerName string, check HealthCheckItem) HealthCheckDetail {
|
|
target := fmt.Sprintf(":%d%s", check.Port, check.Path)
|
|
|
|
switch check.Type {
|
|
case "tcp":
|
|
return m.probeTCP(containerName, check.Port, target)
|
|
case "http", "api":
|
|
return m.probeHTTP(containerName, check, target)
|
|
default:
|
|
return HealthCheckDetail{
|
|
Type: check.Type,
|
|
Target: target,
|
|
Healthy: false,
|
|
Error: fmt.Sprintf("unknown check type: %s", check.Type),
|
|
}
|
|
}
|
|
}
|
|
|
|
// probeTCP tests if a TCP port is reachable on the container.
|
|
func (m *Manager) probeTCP(containerName string, port int, target string) HealthCheckDetail {
|
|
start := time.Now()
|
|
addr := net.JoinHostPort(containerName, fmt.Sprintf("%d", port))
|
|
conn, err := net.DialTimeout("tcp", addr, 5*time.Second)
|
|
latency := time.Since(start)
|
|
|
|
detail := HealthCheckDetail{
|
|
Type: "tcp",
|
|
Target: target,
|
|
Latency: formatLatency(latency),
|
|
}
|
|
|
|
if err != nil {
|
|
detail.Healthy = false
|
|
detail.Error = err.Error()
|
|
} else {
|
|
conn.Close()
|
|
detail.Healthy = true
|
|
}
|
|
return detail
|
|
}
|
|
|
|
// probeHTTP makes an HTTP request to the container and evaluates the result.
|
|
// For "http" type: any response = healthy. For "api" type: validates expect rules.
|
|
func (m *Manager) probeHTTP(containerName string, check HealthCheckItem, target string) HealthCheckDetail {
|
|
url := fmt.Sprintf("http://%s:%d%s", containerName, check.Port, check.Path)
|
|
method := check.Method
|
|
if method == "" {
|
|
method = "GET"
|
|
}
|
|
|
|
start := time.Now()
|
|
|
|
client := &http.Client{
|
|
Timeout: 5 * time.Second,
|
|
CheckRedirect: func(req *http.Request, via []*http.Request) error {
|
|
return http.ErrUseLastResponse
|
|
},
|
|
}
|
|
|
|
req, err := http.NewRequest(method, url, nil)
|
|
if err != nil {
|
|
return HealthCheckDetail{
|
|
Type: check.Type,
|
|
Target: target,
|
|
Healthy: false,
|
|
Error: fmt.Sprintf("bad request: %v", err),
|
|
Latency: "0ms",
|
|
}
|
|
}
|
|
|
|
resp, err := client.Do(req)
|
|
latency := time.Since(start)
|
|
|
|
detail := HealthCheckDetail{
|
|
Type: check.Type,
|
|
Target: target,
|
|
Latency: formatLatency(latency),
|
|
}
|
|
|
|
if err != nil {
|
|
detail.Healthy = false
|
|
detail.Error = err.Error()
|
|
return detail
|
|
}
|
|
defer resp.Body.Close()
|
|
detail.Status = resp.StatusCode
|
|
|
|
// For "http" type, any response means the service is alive
|
|
if check.Type == "http" {
|
|
detail.Healthy = true
|
|
return detail
|
|
}
|
|
|
|
// For "api" type, validate expectations
|
|
if check.Expect == nil {
|
|
// No expectations = just check for a response (same as http)
|
|
detail.Healthy = true
|
|
return detail
|
|
}
|
|
|
|
// Check expected status code
|
|
if check.Expect.Status > 0 && resp.StatusCode != check.Expect.Status {
|
|
detail.Healthy = false
|
|
detail.Error = fmt.Sprintf("expected status %d, got %d", check.Expect.Status, resp.StatusCode)
|
|
return detail
|
|
}
|
|
|
|
// Check expected body content
|
|
if check.Expect.BodyContains != "" {
|
|
body, err := io.ReadAll(io.LimitReader(resp.Body, 8192)) // read up to 8KB
|
|
if err != nil {
|
|
detail.Healthy = false
|
|
detail.Error = fmt.Sprintf("reading body: %v", err)
|
|
return detail
|
|
}
|
|
if !strings.Contains(string(body), check.Expect.BodyContains) {
|
|
detail.Healthy = false
|
|
detail.Error = fmt.Sprintf("body missing expected string %q", check.Expect.BodyContains)
|
|
return detail
|
|
}
|
|
}
|
|
|
|
detail.Healthy = true
|
|
return detail
|
|
}
|
|
|
|
// The one sentence R-630 adds. Hungarian bytes by default, the key beside it — util.MsgError's
|
|
// contract, because HealthProbe is serialised raw to the API and has no localisation seam of its own.
|
|
const (
|
|
MsgHealthNoProbeContainer = "Nem futott egészségellenőrzés: nincs hozzá tartozó konténer."
|
|
KeyHealthNoProbeContainer = "health.no_probe_container"
|
|
)
|
|
|
|
// noProbeStack is a stack that declares a health check which resolves to no container.
|
|
type noProbeStack struct {
|
|
name string
|
|
candidates []string
|
|
}
|
|
|
|
// findProbeContainer returns the container to probe for a stack, and the candidates it rejected.
|
|
//
|
|
// Four rules, in order (R-630):
|
|
//
|
|
// 1. the container whose name EQUALS the stack name;
|
|
// 2. the container named by `healthcheck.container` in `.felhom.yml`;
|
|
// 3. a prefix match — but ONLY when exactly one running container matches. The old code took the
|
|
// FIRST prefix match, which for `immich` (four `immich-*` containers and no exact match) meant
|
|
// whichever the container list happened to yield, and which was seen live on `outline` probing
|
|
// `outline-postgres:3000` during startup because the exactly-named container was not up yet;
|
|
// 4. nothing — and the CANDIDATES come back with it, because "no container" with no list is the
|
|
// silence this whole row is about.
|
|
//
|
|
// `meta` may be nil; the caller is not required to have one.
|
|
func findProbeContainerMeta(stackName string, meta *Metadata, containers []ContainerInfo) (string, []string) {
|
|
probeable := func(c ContainerInfo) bool {
|
|
return c.State == StateRunning || c.State == StateUnhealthy
|
|
}
|
|
for _, c := range containers {
|
|
if c.Name == stackName && probeable(c) {
|
|
return c.Name, nil
|
|
}
|
|
}
|
|
if meta != nil && meta.HealthCheck != nil && meta.HealthCheck.Container != "" {
|
|
want := meta.HealthCheck.Container
|
|
for _, c := range containers {
|
|
if c.Name == want && probeable(c) {
|
|
return c.Name, nil
|
|
}
|
|
}
|
|
}
|
|
var prefix []string
|
|
for _, c := range containers {
|
|
if strings.HasPrefix(c.Name, stackName) && probeable(c) {
|
|
prefix = append(prefix, c.Name)
|
|
}
|
|
}
|
|
if len(prefix) == 1 {
|
|
return prefix[0], nil
|
|
}
|
|
// Ambiguous or empty: name what was seen so the log and the page can say why.
|
|
all := make([]string, 0, len(containers))
|
|
for _, c := range containers {
|
|
all = append(all, c.Name+"("+string(c.State)+")")
|
|
}
|
|
return "", all
|
|
}
|
|
|
|
// findProbeContainer is the one-value form kept for callers that only want the name.
|
|
func findProbeContainer(stackName string, containers []ContainerInfo) string {
|
|
n, _ := findProbeContainerMeta(stackName, nil, containers)
|
|
return n
|
|
}
|
|
|
|
// parseInterval parses a duration string like "5m", "30s", "1h".
|
|
// Returns 5 minutes as default if parsing fails.
|
|
func parseInterval(s string) time.Duration {
|
|
if s == "" {
|
|
return 5 * time.Minute
|
|
}
|
|
d, err := time.ParseDuration(s)
|
|
if err != nil {
|
|
return 5 * time.Minute
|
|
}
|
|
return d
|
|
}
|
|
|
|
// formatLatency formats a duration as a human-readable latency string.
|
|
func formatLatency(d time.Duration) string {
|
|
if d < time.Millisecond {
|
|
return fmt.Sprintf("%dµs", d.Microseconds())
|
|
}
|
|
if d < time.Second {
|
|
return fmt.Sprintf("%dms", d.Milliseconds())
|
|
}
|
|
return fmt.Sprintf("%.1fs", d.Seconds())
|
|
}
|
|
|
|
// methodOrEmpty returns the method string for logging, or empty for non-api checks.
|
|
func methodOrEmpty(check HealthCheckItem) string {
|
|
if check.Type == "api" && check.Method != "" {
|
|
return check.Method
|
|
}
|
|
if check.Type == "api" {
|
|
return "GET"
|
|
}
|
|
return "GET"
|
|
}
|