v0.293.0: the controller heals itself after the guest's Docker socket is re-created (R-860)
gates / gates (push) Successful in 29s
gates / gates (push) Successful in 29s
internal/sockheal: 60 s of refusals (never a timeout, only after Docker answered once) → exit 75 so Docker's restart policy brings the controller back on the current socket; every 5 min it restarts any other socket user (traefik) holding an older inode. Measured on 9202: only a docker.socket restart re-creates the file; dockerd crash / docker.service restart keep it. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,218 @@
|
||||
// Package sockheal lets the controller heal itself after the guest's Docker socket FILE is re-created (R-860,
|
||||
// generalising R-858).
|
||||
//
|
||||
// MEASURED 2026-10-04 on scratch guest 9202 (`felhom.eu/documentation/audits/r840-config-bundle-2026-10-04/partE/`):
|
||||
// - `systemctl restart docker` and a `kill -9` of dockerd keep the socket file (systemd's docker.socket holds it;
|
||||
// same inode): the controller and traefik keep working. Nothing to heal.
|
||||
// - a restart of docker.socket — what a docker-ce package upgrade does, or a by-hand `systemctl restart
|
||||
// docker.socket` — RE-CREATES the file (inode 144 → 7202). With live-restore every container keeps running, but
|
||||
// a container that bind-mounts the socket FILE keeps the deleted inode: the controller and traefik get
|
||||
// "connection refused" for ever, while their own health checks stay "healthy".
|
||||
// - when the controller's process exits, Docker's restart policy brings it back within ~1 s, and the restarted
|
||||
// container mounts the CURRENT socket file.
|
||||
//
|
||||
// So: Watch.Tick pings the socket; once Docker has answered in this process's life and then refuses for Window
|
||||
// (60 s) without a break, the controller exits (ExitCode) and Docker restarts it on the current socket. A timeout or
|
||||
// a slow daemon never counts — only a refused or missing socket, the stale-inode signature. Watch.CheckUsers then
|
||||
// restarts every OTHER container that mounts the socket and still holds an older inode than the controller's own
|
||||
// (traefik today), so the router follows too.
|
||||
package sockheal
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"net"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// SocketPath is the guest's Docker socket as the controller mounts it.
|
||||
const SocketPath = "/var/run/docker.sock"
|
||||
|
||||
// ExitCode is the controller's exit status when it leaves to be restarted on the current socket.
|
||||
const ExitCode = 75
|
||||
|
||||
// DefaultWindow is how long Docker must refuse, without a break, before the controller exits.
|
||||
const DefaultWindow = 60 * time.Second
|
||||
|
||||
// Watch is the socket watch. Every field but Logger is a seam; New fills the real ones.
|
||||
type Watch struct {
|
||||
Ping func(ctx context.Context) error // nil = Docker answered
|
||||
Exit func(code int)
|
||||
Now func() time.Time
|
||||
Window time.Duration
|
||||
Logger *log.Logger
|
||||
|
||||
// CheckUsers seams.
|
||||
OwnInode func() (uint64, error) // the socket inode THIS container sees
|
||||
SocketUsers func(ctx context.Context) ([]User, error) // OTHER running containers that mount the socket
|
||||
UserInode func(ctx context.Context, u User) (uint64, error) // the socket inode that container sees
|
||||
Restart func(ctx context.Context, name string) error
|
||||
|
||||
mu sync.Mutex
|
||||
everWorked bool
|
||||
refusedSince time.Time
|
||||
}
|
||||
|
||||
// Refused reports whether err is the stale-socket signature: connection refused, or no socket at all.
|
||||
func Refused(err error) bool {
|
||||
return errors.Is(err, syscall.ECONNREFUSED) || errors.Is(err, syscall.ENOENT)
|
||||
}
|
||||
|
||||
// PingSocket dials the socket and asks Docker's /_ping. A refused dial returns the dial error unwrapped enough for
|
||||
// Refused to see ECONNREFUSED.
|
||||
func PingSocket(ctx context.Context, path string) error {
|
||||
d := net.Dialer{Timeout: 3 * time.Second}
|
||||
c, err := d.DialContext(ctx, "unix", path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer c.Close()
|
||||
_ = c.SetDeadline(time.Now().Add(5 * time.Second))
|
||||
if _, err := io.WriteString(c, "GET /_ping HTTP/1.0\r\nHost: docker\r\n\r\n"); err != nil {
|
||||
return err
|
||||
}
|
||||
buf := make([]byte, 64)
|
||||
n, err := c.Read(buf)
|
||||
if n == 0 && err != nil {
|
||||
return err
|
||||
}
|
||||
if !strings.HasPrefix(string(buf[:n]), "HTTP/1.") || !strings.Contains(string(buf[:n]), " 200 ") {
|
||||
return fmt.Errorf("docker /_ping answered %q", strings.SplitN(string(buf[:n]), "\r\n", 2)[0])
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Tick is one check (the scheduler calls it every 15 s).
|
||||
func (w *Watch) Tick(ctx context.Context) error {
|
||||
err := w.Ping(ctx)
|
||||
w.mu.Lock()
|
||||
defer w.mu.Unlock()
|
||||
now := w.Now()
|
||||
if err == nil {
|
||||
if !w.refusedSince.IsZero() {
|
||||
w.Logger.Printf("[INFO] [sockheal] Docker answers again after %s of refusals — no restart needed",
|
||||
now.Sub(w.refusedSince).Round(time.Second))
|
||||
}
|
||||
w.everWorked, w.refusedSince = true, time.Time{}
|
||||
return nil
|
||||
}
|
||||
if !Refused(err) {
|
||||
// A timeout or a slow daemon (a heavy backup, an engine step in progress) is not the stale-socket case.
|
||||
w.Logger.Printf("[DEBUG] [sockheal] Docker ping failed, not counted (not a refusal): %v", err)
|
||||
return nil
|
||||
}
|
||||
if !w.everWorked {
|
||||
// Never reached Docker in this process: a restart would not help, and exiting would loop every Window.
|
||||
w.Logger.Printf("[WARN] [sockheal] Docker refuses (%v) and never answered since this controller started — not exiting", err)
|
||||
return nil
|
||||
}
|
||||
if w.refusedSince.IsZero() {
|
||||
w.refusedSince = now
|
||||
w.Logger.Printf("[WARN] [sockheal] Docker refuses the socket (%v) — exiting after %s of refusals so Docker restarts "+
|
||||
"this controller on the current socket (R-860)", err, w.Window)
|
||||
return nil
|
||||
}
|
||||
if gone := now.Sub(w.refusedSince); gone >= w.Window {
|
||||
w.Logger.Printf("[ERROR] [sockheal] Docker has refused the socket for %s (%v) — the socket file was re-created and "+
|
||||
"this container holds the old one; EXITING (code %d) so Docker's restart policy brings it back on the current "+
|
||||
"socket (R-860)", gone.Round(time.Second), err, ExitCode)
|
||||
w.Exit(ExitCode)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// CheckUsers restarts every OTHER running container that mounts the socket and sees a different inode than this
|
||||
// controller — only while this controller itself reaches Docker (so its own inode is the current one).
|
||||
func (w *Watch) CheckUsers(ctx context.Context) error {
|
||||
if err := w.Ping(ctx); err != nil {
|
||||
return nil // Tick handles a controller that cannot reach Docker
|
||||
}
|
||||
own, err := w.OwnInode()
|
||||
if err != nil {
|
||||
return fmt.Errorf("sockheal: own socket inode: %w", err)
|
||||
}
|
||||
users, err := w.SocketUsers(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("sockheal: list socket users: %w", err)
|
||||
}
|
||||
for _, u := range users {
|
||||
name := u.Name
|
||||
ino, err := w.UserInode(ctx, u)
|
||||
if err != nil {
|
||||
w.Logger.Printf("[DEBUG] [sockheal] %s: cannot read its socket inode (no stat in the image?): %v — skipped", name, err)
|
||||
continue
|
||||
}
|
||||
if ino == own {
|
||||
continue
|
||||
}
|
||||
w.Logger.Printf("[WARN] [sockheal] %s holds an old docker socket (inode %d, current %d) — restarting it (R-860)", name, ino, own)
|
||||
if err := w.Restart(ctx, name); err != nil {
|
||||
w.Logger.Printf("[ERROR] [sockheal] restart %s failed: %v", name, err)
|
||||
continue
|
||||
}
|
||||
w.Logger.Printf("[INFO] [sockheal] %s restarted onto the current docker socket", name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// User is a container that bind-mounts the Docker socket, and where.
|
||||
type User struct {
|
||||
Name string
|
||||
Dest string
|
||||
}
|
||||
|
||||
// SelfName is the controller's own container, never in SocketUsers' answer.
|
||||
const SelfName = "felhom-controller"
|
||||
|
||||
// DockerSocketUsers lists the OTHER running containers that mount the socket (docker ps + one docker inspect).
|
||||
func DockerSocketUsers(run func(ctx context.Context, args ...string) (string, error)) func(ctx context.Context) ([]User, error) {
|
||||
return func(ctx context.Context) ([]User, error) {
|
||||
ids, err := run(ctx, "ps", "-q", "--no-trunc")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
fields := strings.Fields(ids)
|
||||
if len(fields) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
out, err := run(ctx, append([]string{"inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}"}, fields...)...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var users []User
|
||||
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
|
||||
name, mounts, ok := strings.Cut(strings.TrimSpace(line), "|")
|
||||
name = strings.TrimPrefix(name, "/")
|
||||
if !ok || name == SelfName {
|
||||
continue
|
||||
}
|
||||
for _, m := range strings.Split(mounts, ";") {
|
||||
if m == "/var/run/docker.sock" || m == "/run/docker.sock" {
|
||||
users = append(users, User{Name: name, Dest: m})
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
return users, nil
|
||||
}
|
||||
}
|
||||
|
||||
// InodeOf returns a path's inode number.
|
||||
func InodeOf(path string) (uint64, error) {
|
||||
fi, err := os.Stat(path)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
st, ok := fi.Sys().(*syscall.Stat_t)
|
||||
if !ok {
|
||||
return 0, fmt.Errorf("no inode for %s", path)
|
||||
}
|
||||
return st.Ino, nil
|
||||
}
|
||||
Reference in New Issue
Block a user