Files
felhom-controller/controller/internal/stacks/healthprobe.go
T
admin b793638484
gates / gates (push) Successful in 26s
v0.262.0: six defects two drill nights found in the update, remove and hold paths
R-630 (P1): waitUpdateHealthy kept the probe inside `if hc != nil && len(hc.Checks) > 0`, and when
findProbeContainer returned "" its else set last="no probe container" and LOOPED - the settle path
sat in the outer else, unreachable. So verifying could only time out and failAndHold then stopped a
working app. Measured on paperless-ngx: three containers healthy, failed at +313.0s, front door 404
after. It now falls through to the same settle path with a WARN naming the candidates.

The probe target is decidable now: HealthCheckConfig.Container plus findProbeContainerMeta resolve
by exact stack name -> explicit container -> a UNIQUE prefix -> nothing with the candidates
returned. The old rule took the FIRST prefix match. A skipped stack records why instead of silence.

R-634 (half): RemoveStack refused on the !Deployed FLAG while the machine had containers, a compose
file and an app.yaml. It now asks whether anything EXISTS. The mechanism producing the bad record is
still not diagnosed and R-634 stays open for it.

R-633/R-626: RemoveStack consults UpdateGuards.Busy and IsUpdating and refuses with the app's own
sentence - the product already refused this clash for update and for restore. And because `down`
returning 0 is a request not a result, the project is watched for 25s afterwards, anything carrying
its label is removed by name with its labels logged, and the answer carries `verified`.

R-621: failAndHold writes compose logs --tail 400 into <stackdir>/hold-logs/<ts>/ BEFORE the down
that destroys them. Two existing tests pin the compose sequence and correctly caught the new step;
their expectations are updated with the reason that the ORDER is the assertion.

R-614: RemoveStack calls ClearUpdateState.

NOT in this release: R-625 (a held app still renders an Update button). Named, not half-done.

Three new sentences, each born as a key in both bundles. Four red-proofs seen failing.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-22 20:46:50 +02:00

426 lines
12 KiB
Go

package stacks
import (
"fmt"
"io"
"net"
"net/http"
"strings"
"sync"
"time"
)
// probeTarget holds the info needed to probe a single stack.
type probeTarget struct {
stackName string
containerName string
checks []HealthCheckItem
}
// RunHealthProbes runs controller-side health probes for all running stacks
// that have healthcheck configuration and whose interval has elapsed.
// Called by the scheduler every minute.
func (m *Manager) RunHealthProbes() error {
// Phase 1: collect targets (under lock)
var noProbe []noProbeStack
m.mu.RLock()
var targets []probeTarget
skippedNotDue := 0
skippedNoContainer := 0
for name, stack := range m.stacks {
// StateDegraded (R-51) is probed too: its live members still answer, and the probe result
// only ever overrides StateRunning below, so a degraded stack can never be masked as unhealthy.
if stack.State != StateRunning && stack.State != StateUnhealthy && stack.State != StateDegraded {
continue
}
hc := stack.Meta.HealthCheck
if hc == nil || len(hc.Checks) == 0 {
continue
}
// Check if interval has elapsed since last probe.
// Fast 10s probes until healthy, then normal interval (default 5m).
// When HealthProbe is nil (just started/restarted), probe immediately.
interval := parseInterval(hc.Interval)
if stack.HealthProbe != nil {
effectiveInterval := interval
if !stack.HealthProbe.Healthy {
effectiveInterval = 10 * time.Second
}
if time.Since(stack.HealthProbe.LastCheck) < effectiveInterval {
skippedNotDue++
if m.isDebug() {
sinceLastCheck := time.Since(stack.HealthProbe.LastCheck).Round(time.Second)
m.logger.Printf("[DEBUG] [stacks] RunHealthProbes: skipping %s — last check %s ago, effective interval %s, healthy=%v",
name, sinceLastCheck, effectiveInterval, stack.HealthProbe.Healthy)
}
continue
}
}
// Find the main container to probe. R-630: a stack whose check resolves to no container
// used to be skipped SILENTLY — paperless-ngx's probe had therefore never run on any box,
// and nothing on any screen could say so. It now gets a RESULT that says why, so the app
// page can tell "checked and healthy" from "never checked".
containerName, candidates := findProbeContainerMeta(name, &stack.Meta, stack.Containers)
if containerName == "" {
skippedNoContainer++
noProbe = append(noProbe, noProbeStack{name: name, candidates: candidates})
continue
}
targets = append(targets, probeTarget{
stackName: name,
containerName: containerName,
checks: hc.Checks,
})
}
m.mu.RUnlock()
// Record the "no probe container" stacks OUTSIDE the read lock, then write their results under
// the write lock — the same shape the probe results themselves use.
for _, np := range noProbe {
m.mu.Lock()
if s, ok := m.stacks[np.name]; ok {
s.HealthProbe = &HealthProbeResult{
Healthy: true, // never RED on our own inability to look — R-630's whole point
LastCheck: m.now(),
Details: []HealthCheckDetail{{
Type: "none",
Target: np.name,
Healthy: true,
Error: MsgHealthNoProbeContainer,
MessageKey: KeyHealthNoProbeContainer,
}},
}
}
m.mu.Unlock()
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RunHealthProbes: %s has a health check but no container to probe — candidates: %v", np.name, np.candidates)
}
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] RunHealthProbes: collected %d targets (%d skipped not due, %d skipped no container)",
len(targets), skippedNotDue, skippedNoContainer)
}
if len(targets) == 0 {
return nil
}
// Phase 2: run all probes concurrently (no lock held)
type probeResult struct {
stackName string
result *HealthProbeResult
}
results := make([]probeResult, len(targets))
var wg sync.WaitGroup
for i, t := range targets {
wg.Add(1)
go func(idx int, t probeTarget) {
defer wg.Done()
result := m.runChecks(t)
results[idx] = probeResult{stackName: t.stackName, result: result}
}(i, t)
}
wg.Wait()
// Phase 3: apply results and log (under lock)
m.mu.Lock()
okCount, failCount := 0, 0
for _, pr := range results {
stack, ok := m.stacks[pr.stackName]
if !ok {
continue
}
stack.HealthProbe = pr.result
if pr.result.Healthy {
okCount++
// If Docker says running and probe is healthy, ensure state is running
// (clears a previous unhealthy override)
if stack.State == StateUnhealthy {
stack.State = StateRunning
}
} else {
failCount++
if stack.State == StateRunning {
stack.State = StateUnhealthy
}
}
}
m.mu.Unlock()
// Summary log
if failCount > 0 {
m.logger.Printf("[WARN] Health probes: %d ok, %d unhealthy (of %d probed)", okCount, failCount, len(targets))
} else {
m.logger.Printf("[INFO] Health probes: %d ok (of %d probed)", okCount, len(targets))
}
return nil
}
// runChecks executes all health check items for a single stack target.
func (m *Manager) runChecks(t probeTarget) *HealthProbeResult {
result := &HealthProbeResult{
LastCheck: time.Now(),
Healthy: true,
}
for _, check := range t.checks {
detail := m.runSingleCheck(t.containerName, check)
result.Details = append(result.Details, detail)
if detail.Healthy {
if m.isDebug() {
if detail.Status > 0 {
m.logger.Printf("[DEBUG] Health probe %s: %s %s :%d%s → %d (%s)",
t.stackName, strings.ToUpper(check.Type), methodOrEmpty(check), check.Port, check.Path, detail.Status, detail.Latency)
} else {
m.logger.Printf("[DEBUG] Health probe %s: TCP :%d → ok (%s)",
t.stackName, check.Port, detail.Latency)
}
}
} else {
result.Healthy = false
m.logger.Printf("[WARN] Health probe %s: %s %s :%d%s → %s",
t.stackName, strings.ToUpper(check.Type), methodOrEmpty(check), check.Port, check.Path, detail.Error)
}
}
return result
}
// runSingleCheck executes one health check item and returns the result.
func (m *Manager) runSingleCheck(containerName string, check HealthCheckItem) HealthCheckDetail {
target := fmt.Sprintf(":%d%s", check.Port, check.Path)
switch check.Type {
case "tcp":
return m.probeTCP(containerName, check.Port, target)
case "http", "api":
return m.probeHTTP(containerName, check, target)
default:
return HealthCheckDetail{
Type: check.Type,
Target: target,
Healthy: false,
Error: fmt.Sprintf("unknown check type: %s", check.Type),
}
}
}
// probeTCP tests if a TCP port is reachable on the container.
func (m *Manager) probeTCP(containerName string, port int, target string) HealthCheckDetail {
start := time.Now()
addr := net.JoinHostPort(containerName, fmt.Sprintf("%d", port))
conn, err := net.DialTimeout("tcp", addr, 5*time.Second)
latency := time.Since(start)
detail := HealthCheckDetail{
Type: "tcp",
Target: target,
Latency: formatLatency(latency),
}
if err != nil {
detail.Healthy = false
detail.Error = err.Error()
} else {
conn.Close()
detail.Healthy = true
}
return detail
}
// probeHTTP makes an HTTP request to the container and evaluates the result.
// For "http" type: any response = healthy. For "api" type: validates expect rules.
func (m *Manager) probeHTTP(containerName string, check HealthCheckItem, target string) HealthCheckDetail {
url := fmt.Sprintf("http://%s:%d%s", containerName, check.Port, check.Path)
method := check.Method
if method == "" {
method = "GET"
}
start := time.Now()
client := &http.Client{
Timeout: 5 * time.Second,
CheckRedirect: func(req *http.Request, via []*http.Request) error {
return http.ErrUseLastResponse
},
}
req, err := http.NewRequest(method, url, nil)
if err != nil {
return HealthCheckDetail{
Type: check.Type,
Target: target,
Healthy: false,
Error: fmt.Sprintf("bad request: %v", err),
Latency: "0ms",
}
}
resp, err := client.Do(req)
latency := time.Since(start)
detail := HealthCheckDetail{
Type: check.Type,
Target: target,
Latency: formatLatency(latency),
}
if err != nil {
detail.Healthy = false
detail.Error = err.Error()
return detail
}
defer resp.Body.Close()
detail.Status = resp.StatusCode
// For "http" type, any response means the service is alive
if check.Type == "http" {
detail.Healthy = true
return detail
}
// For "api" type, validate expectations
if check.Expect == nil {
// No expectations = just check for a response (same as http)
detail.Healthy = true
return detail
}
// Check expected status code
if check.Expect.Status > 0 && resp.StatusCode != check.Expect.Status {
detail.Healthy = false
detail.Error = fmt.Sprintf("expected status %d, got %d", check.Expect.Status, resp.StatusCode)
return detail
}
// Check expected body content
if check.Expect.BodyContains != "" {
body, err := io.ReadAll(io.LimitReader(resp.Body, 8192)) // read up to 8KB
if err != nil {
detail.Healthy = false
detail.Error = fmt.Sprintf("reading body: %v", err)
return detail
}
if !strings.Contains(string(body), check.Expect.BodyContains) {
detail.Healthy = false
detail.Error = fmt.Sprintf("body missing expected string %q", check.Expect.BodyContains)
return detail
}
}
detail.Healthy = true
return detail
}
// The one sentence R-630 adds. Hungarian bytes by default, the key beside it — util.MsgError's
// contract, because HealthProbe is serialised raw to the API and has no localisation seam of its own.
const (
MsgHealthNoProbeContainer = "Nem futott egészségellenőrzés: nincs hozzá tartozó konténer."
KeyHealthNoProbeContainer = "health.no_probe_container"
)
// noProbeStack is a stack that declares a health check which resolves to no container.
type noProbeStack struct {
name string
candidates []string
}
// findProbeContainer returns the container to probe for a stack, and the candidates it rejected.
//
// Four rules, in order (R-630):
//
// 1. the container whose name EQUALS the stack name;
// 2. the container named by `healthcheck.container` in `.felhom.yml`;
// 3. a prefix match — but ONLY when exactly one running container matches. The old code took the
// FIRST prefix match, which for `immich` (four `immich-*` containers and no exact match) meant
// whichever the container list happened to yield, and which was seen live on `outline` probing
// `outline-postgres:3000` during startup because the exactly-named container was not up yet;
// 4. nothing — and the CANDIDATES come back with it, because "no container" with no list is the
// silence this whole row is about.
//
// `meta` may be nil; the caller is not required to have one.
func findProbeContainerMeta(stackName string, meta *Metadata, containers []ContainerInfo) (string, []string) {
probeable := func(c ContainerInfo) bool {
return c.State == StateRunning || c.State == StateUnhealthy
}
for _, c := range containers {
if c.Name == stackName && probeable(c) {
return c.Name, nil
}
}
if meta != nil && meta.HealthCheck != nil && meta.HealthCheck.Container != "" {
want := meta.HealthCheck.Container
for _, c := range containers {
if c.Name == want && probeable(c) {
return c.Name, nil
}
}
}
var prefix []string
for _, c := range containers {
if strings.HasPrefix(c.Name, stackName) && probeable(c) {
prefix = append(prefix, c.Name)
}
}
if len(prefix) == 1 {
return prefix[0], nil
}
// Ambiguous or empty: name what was seen so the log and the page can say why.
all := make([]string, 0, len(containers))
for _, c := range containers {
all = append(all, c.Name+"("+string(c.State)+")")
}
return "", all
}
// findProbeContainer is the one-value form kept for callers that only want the name.
func findProbeContainer(stackName string, containers []ContainerInfo) string {
n, _ := findProbeContainerMeta(stackName, nil, containers)
return n
}
// parseInterval parses a duration string like "5m", "30s", "1h".
// Returns 5 minutes as default if parsing fails.
func parseInterval(s string) time.Duration {
if s == "" {
return 5 * time.Minute
}
d, err := time.ParseDuration(s)
if err != nil {
return 5 * time.Minute
}
return d
}
// formatLatency formats a duration as a human-readable latency string.
func formatLatency(d time.Duration) string {
if d < time.Millisecond {
return fmt.Sprintf("%dµs", d.Microseconds())
}
if d < time.Second {
return fmt.Sprintf("%dms", d.Milliseconds())
}
return fmt.Sprintf("%.1fs", d.Seconds())
}
// methodOrEmpty returns the method string for logging, or empty for non-api checks.
func methodOrEmpty(check HealthCheckItem) string {
if check.Type == "api" && check.Method != "" {
return check.Method
}
if check.Type == "api" {
return "GET"
}
return "GET"
}