// Package sockheal lets the controller heal itself after the guest's Docker socket FILE is re-created (R-860, // generalising R-858). // // MEASURED 2026-10-04 on scratch guest 9202 (`felhom.eu/documentation/audits/r840-config-bundle-2026-10-04/partE/`): // - `systemctl restart docker` and a `kill -9` of dockerd keep the socket file (systemd's docker.socket holds it; // same inode): the controller and traefik keep working. Nothing to heal. // - a restart of docker.socket — what a docker-ce package upgrade does, or a by-hand `systemctl restart // docker.socket` — RE-CREATES the file (inode 144 → 7202). With live-restore every container keeps running, but // a container that bind-mounts the socket FILE keeps the deleted inode: the controller and traefik get // "connection refused" for ever, while their own health checks stay "healthy". // - when the controller's process exits, Docker's restart policy brings it back within ~1 s, and the restarted // container mounts the CURRENT socket file. // // So: Watch.Tick pings the socket; once Docker has answered in this process's life and then refuses for Window // (60 s) without a break, the controller exits (ExitCode) and Docker restarts it on the current socket. A timeout or // a slow daemon never counts — only a refused or missing socket, the stale-inode signature. Watch.CheckUsers then // restarts every OTHER container that mounts the socket and still holds an older inode than the controller's own // (traefik today), so the router follows too. package sockheal import ( "context" "errors" "fmt" "io" "log" "net" "os" "strings" "sync" "syscall" "time" ) // SocketPath is the guest's Docker socket as the controller mounts it. const SocketPath = "/var/run/docker.sock" // ExitCode is the controller's exit status when it leaves to be restarted on the current socket. const ExitCode = 75 // DefaultWindow is how long Docker must refuse, without a break, before the controller exits. const DefaultWindow = 60 * time.Second // Watch is the socket watch. Every field but Logger is a seam; New fills the real ones. type Watch struct { Ping func(ctx context.Context) error // nil = Docker answered Exit func(code int) Now func() time.Time Window time.Duration Logger *log.Logger // CheckUsers seams. OwnInode func() (uint64, error) // the socket inode THIS container sees SocketUsers func(ctx context.Context) ([]User, error) // OTHER running containers that mount the socket UserInode func(ctx context.Context, u User) (uint64, error) // the socket inode that container sees Restart func(ctx context.Context, name string) error mu sync.Mutex everWorked bool refusedSince time.Time } // Refused reports whether err is the stale-socket signature: connection refused, or no socket at all. func Refused(err error) bool { return errors.Is(err, syscall.ECONNREFUSED) || errors.Is(err, syscall.ENOENT) } // PingSocket dials the socket and asks Docker's /_ping. A refused dial returns the dial error unwrapped enough for // Refused to see ECONNREFUSED. func PingSocket(ctx context.Context, path string) error { d := net.Dialer{Timeout: 3 * time.Second} c, err := d.DialContext(ctx, "unix", path) if err != nil { return err } defer c.Close() _ = c.SetDeadline(time.Now().Add(5 * time.Second)) if _, err := io.WriteString(c, "GET /_ping HTTP/1.0\r\nHost: docker\r\n\r\n"); err != nil { return err } buf := make([]byte, 64) n, err := c.Read(buf) if n == 0 && err != nil { return err } if !strings.HasPrefix(string(buf[:n]), "HTTP/1.") || !strings.Contains(string(buf[:n]), " 200 ") { return fmt.Errorf("docker /_ping answered %q", strings.SplitN(string(buf[:n]), "\r\n", 2)[0]) } return nil } // Tick is one check (the scheduler calls it every 15 s). func (w *Watch) Tick(ctx context.Context) error { err := w.Ping(ctx) w.mu.Lock() defer w.mu.Unlock() now := w.Now() if err == nil { if !w.refusedSince.IsZero() { w.Logger.Printf("[INFO] [sockheal] Docker answers again after %s of refusals — no restart needed", now.Sub(w.refusedSince).Round(time.Second)) } w.everWorked, w.refusedSince = true, time.Time{} return nil } if !Refused(err) { // A timeout or a slow daemon (a heavy backup, an engine step in progress) is not the stale-socket case. w.Logger.Printf("[DEBUG] [sockheal] Docker ping failed, not counted (not a refusal): %v", err) return nil } if !w.everWorked { // Never reached Docker in this process: a restart would not help, and exiting would loop every Window. w.Logger.Printf("[WARN] [sockheal] Docker refuses (%v) and never answered since this controller started — not exiting", err) return nil } if w.refusedSince.IsZero() { w.refusedSince = now w.Logger.Printf("[WARN] [sockheal] Docker refuses the socket (%v) — exiting after %s of refusals so Docker restarts "+ "this controller on the current socket (R-860)", err, w.Window) return nil } if gone := now.Sub(w.refusedSince); gone >= w.Window { w.Logger.Printf("[ERROR] [sockheal] Docker has refused the socket for %s (%v) — the socket file was re-created and "+ "this container holds the old one; EXITING (code %d) so Docker's restart policy brings it back on the current "+ "socket (R-860)", gone.Round(time.Second), err, ExitCode) w.Exit(ExitCode) } return nil } // CheckUsers restarts every OTHER running container that mounts the socket and sees a different inode than this // controller — only while this controller itself reaches Docker (so its own inode is the current one). func (w *Watch) CheckUsers(ctx context.Context) error { if err := w.Ping(ctx); err != nil { return nil // Tick handles a controller that cannot reach Docker } own, err := w.OwnInode() if err != nil { return fmt.Errorf("sockheal: own socket inode: %w", err) } users, err := w.SocketUsers(ctx) if err != nil { return fmt.Errorf("sockheal: list socket users: %w", err) } for _, u := range users { name := u.Name ino, err := w.UserInode(ctx, u) if err != nil { w.Logger.Printf("[DEBUG] [sockheal] %s: cannot read its socket inode (no stat in the image?): %v — skipped", name, err) continue } if ino == own { continue } w.Logger.Printf("[WARN] [sockheal] %s holds an old docker socket (inode %d, current %d) — restarting it (R-860)", name, ino, own) if err := w.Restart(ctx, name); err != nil { w.Logger.Printf("[ERROR] [sockheal] restart %s failed: %v", name, err) continue } w.Logger.Printf("[INFO] [sockheal] %s restarted onto the current docker socket", name) } return nil } // User is a container that bind-mounts the Docker socket, and where. type User struct { Name string Dest string } // SelfName is the controller's own container, never in SocketUsers' answer. const SelfName = "felhom-controller" // DockerSocketUsers lists the OTHER running containers that mount the socket (docker ps + one docker inspect). func DockerSocketUsers(run func(ctx context.Context, args ...string) (string, error)) func(ctx context.Context) ([]User, error) { return func(ctx context.Context) ([]User, error) { ids, err := run(ctx, "ps", "-q", "--no-trunc") if err != nil { return nil, err } fields := strings.Fields(ids) if len(fields) == 0 { return nil, nil } out, err := run(ctx, append([]string{"inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}"}, fields...)...) if err != nil { return nil, err } var users []User for _, line := range strings.Split(strings.TrimSpace(out), "\n") { name, mounts, ok := strings.Cut(strings.TrimSpace(line), "|") name = strings.TrimPrefix(name, "/") if !ok || name == SelfName { continue } for _, m := range strings.Split(mounts, ";") { if m == "/var/run/docker.sock" || m == "/run/docker.sock" { users = append(users, User{Name: name, Dest: m}) break } } } return users, nil } } // InodeOf returns a path's inode number. func InodeOf(path string) (uint64, error) { fi, err := os.Stat(path) if err != nil { return 0, err } st, ok := fi.Sys().(*syscall.Stat_t) if !ok { return 0, fmt.Errorf("no inode for %s", path) } return st.Ino, nil }