v0.293.0: the controller heals itself after the guest's Docker socket is re-created (R-860)
gates / gates (push) Successful in 29s
gates / gates (push) Successful in 29s
internal/sockheal: 60 s of refusals (never a timeout, only after Docker answered once) → exit 75 so Docker's restart policy brings the controller back on the current socket; every 5 min it restarts any other socket user (traefik) holding an older inode. Measured on 9202: only a docker.socket restart re-creates the file; dockerd crash / docker.service restart keep it. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -7,6 +7,7 @@ import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/sockheal"
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
@@ -18,6 +19,7 @@ import (
|
||||
"time"
|
||||
|
||||
"crypto/subtle"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
@@ -847,6 +849,21 @@ func main() {
|
||||
sched.Every("stack-scan", 2*time.Minute, func(ctx context.Context) error {
|
||||
return stackMgr.ScanStacks()
|
||||
})
|
||||
// R-860: heal after the guest's Docker socket FILE is re-created (a docker.socket restart — a docker-ce upgrade by
|
||||
// any route, or by hand). The controller exits after 60 s of refusals so Docker restarts it on the current socket;
|
||||
// then it restarts any other socket user (traefik) that still holds the old one. A slow Docker never counts.
|
||||
sockWatch := newSocketWatch(logger)
|
||||
sched.Every("docker-socket-watch", 15*time.Second, sockWatch.Tick)
|
||||
sched.Every("docker-socket-users", 5*time.Minute, sockWatch.CheckUsers)
|
||||
go func() { // once soon after start: the restart that healed THIS controller must heal traefik too
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
case <-time.After(30 * time.Second):
|
||||
if err := sockWatch.CheckUsers(ctx); err != nil {
|
||||
logger.Printf("[WARN] [sockheal] start-up socket-user check: %v", err)
|
||||
}
|
||||
}
|
||||
}()
|
||||
sched.Every("health-probes", 10*time.Second, func(ctx context.Context) error {
|
||||
return stackMgr.RunHealthProbes()
|
||||
})
|
||||
@@ -2461,6 +2478,33 @@ func sameBootFleet(a, b []bootFleetSample) bool {
|
||||
// which is the whole point, since the pre-v0.190.0 bug was a candidate set derived too early.
|
||||
//
|
||||
// Called from main() in a goroutine.
|
||||
// newSocketWatch wires sockheal to the real socket and the docker CLI (dockerexec: refused under go test, R-650).
|
||||
func newSocketWatch(logger *log.Logger) *sockheal.Watch {
|
||||
run := func(ctx context.Context, args ...string) (string, error) {
|
||||
c, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
defer cancel()
|
||||
out, err := dockerexec.CommandContext(c, "docker", args...).Output()
|
||||
return string(out), err
|
||||
}
|
||||
return &sockheal.Watch{
|
||||
Ping: func(ctx context.Context) error { return sockheal.PingSocket(ctx, sockheal.SocketPath) },
|
||||
Exit: os.Exit,
|
||||
Now: time.Now,
|
||||
Window: sockheal.DefaultWindow,
|
||||
Logger: logger,
|
||||
OwnInode: func() (uint64, error) { return sockheal.InodeOf(sockheal.SocketPath) },
|
||||
SocketUsers: sockheal.DockerSocketUsers(run),
|
||||
UserInode: func(ctx context.Context, u sockheal.User) (uint64, error) {
|
||||
out, err := run(ctx, "exec", u.Name, "stat", "-c", "%i", u.Dest)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return strconv.ParseUint(strings.TrimSpace(out), 10, 64)
|
||||
},
|
||||
Restart: func(ctx context.Context, name string) error { _, err := run(ctx, "restart", name); return err },
|
||||
}
|
||||
}
|
||||
|
||||
func runBootReconcile(ctx context.Context, mgr bootrecon.StackProvider, logger *log.Logger) {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
|
||||
Reference in New Issue
Block a user