v0.293.0: the controller heals itself after the guest's Docker socket is re-created (R-860)
gates / gates (push) Successful in 29s

internal/sockheal: 60 s of refusals (never a timeout, only after Docker answered once) → exit 75 so
Docker's restart policy brings the controller back on the current socket; every 5 min it restarts any
other socket user (traefik) holding an older inode. Measured on 9202: only a docker.socket restart
re-creates the file; dockerd crash / docker.service restart keep it.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-04 19:48:29 +02:00
parent 09e634de31
commit 2a7f6c6c42
5 changed files with 480 additions and 0 deletions
+44
View File
@@ -7,6 +7,7 @@ import (
"flag"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
"gitea.dooplex.hu/admin/felhom-controller/internal/sockheal"
"io"
"log"
"net/http"
@@ -18,6 +19,7 @@ import (
"time"
"crypto/subtle"
"strconv"
"strings"
"sync"
@@ -847,6 +849,21 @@ func main() {
sched.Every("stack-scan", 2*time.Minute, func(ctx context.Context) error {
return stackMgr.ScanStacks()
})
// R-860: heal after the guest's Docker socket FILE is re-created (a docker.socket restart — a docker-ce upgrade by
// any route, or by hand). The controller exits after 60 s of refusals so Docker restarts it on the current socket;
// then it restarts any other socket user (traefik) that still holds the old one. A slow Docker never counts.
sockWatch := newSocketWatch(logger)
sched.Every("docker-socket-watch", 15*time.Second, sockWatch.Tick)
sched.Every("docker-socket-users", 5*time.Minute, sockWatch.CheckUsers)
go func() { // once soon after start: the restart that healed THIS controller must heal traefik too
select {
case <-ctx.Done():
case <-time.After(30 * time.Second):
if err := sockWatch.CheckUsers(ctx); err != nil {
logger.Printf("[WARN] [sockheal] start-up socket-user check: %v", err)
}
}
}()
sched.Every("health-probes", 10*time.Second, func(ctx context.Context) error {
return stackMgr.RunHealthProbes()
})
@@ -2461,6 +2478,33 @@ func sameBootFleet(a, b []bootFleetSample) bool {
// which is the whole point, since the pre-v0.190.0 bug was a candidate set derived too early.
//
// Called from main() in a goroutine.
// newSocketWatch wires sockheal to the real socket and the docker CLI (dockerexec: refused under go test, R-650).
func newSocketWatch(logger *log.Logger) *sockheal.Watch {
run := func(ctx context.Context, args ...string) (string, error) {
c, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
out, err := dockerexec.CommandContext(c, "docker", args...).Output()
return string(out), err
}
return &sockheal.Watch{
Ping: func(ctx context.Context) error { return sockheal.PingSocket(ctx, sockheal.SocketPath) },
Exit: os.Exit,
Now: time.Now,
Window: sockheal.DefaultWindow,
Logger: logger,
OwnInode: func() (uint64, error) { return sockheal.InodeOf(sockheal.SocketPath) },
SocketUsers: sockheal.DockerSocketUsers(run),
UserInode: func(ctx context.Context, u sockheal.User) (uint64, error) {
out, err := run(ctx, "exec", u.Name, "stat", "-c", "%i", u.Dest)
if err != nil {
return 0, err
}
return strconv.ParseUint(strings.TrimSpace(out), 10, 64)
},
Restart: func(ctx context.Context, name string) error { _, err := run(ctx, "restart", name); return err },
}
}
func runBootReconcile(ctx context.Context, mgr bootrecon.StackProvider, logger *log.Logger) {
select {
case <-ctx.Done():