2a7f6c6c42
gates / gates (push) Successful in 29s
internal/sockheal: 60 s of refusals (never a timeout, only after Docker answered once) → exit 75 so Docker's restart policy brings the controller back on the current socket; every 5 min it restarts any other socket user (traefik) holding an older inode. Measured on 9202: only a docker.socket restart re-creates the file; dockerd crash / docker.service restart keep it. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
219 lines
8.1 KiB
Go
219 lines
8.1 KiB
Go
// Package sockheal lets the controller heal itself after the guest's Docker socket FILE is re-created (R-860,
|
|
// generalising R-858).
|
|
//
|
|
// MEASURED 2026-10-04 on scratch guest 9202 (`felhom.eu/documentation/audits/r840-config-bundle-2026-10-04/partE/`):
|
|
// - `systemctl restart docker` and a `kill -9` of dockerd keep the socket file (systemd's docker.socket holds it;
|
|
// same inode): the controller and traefik keep working. Nothing to heal.
|
|
// - a restart of docker.socket — what a docker-ce package upgrade does, or a by-hand `systemctl restart
|
|
// docker.socket` — RE-CREATES the file (inode 144 → 7202). With live-restore every container keeps running, but
|
|
// a container that bind-mounts the socket FILE keeps the deleted inode: the controller and traefik get
|
|
// "connection refused" for ever, while their own health checks stay "healthy".
|
|
// - when the controller's process exits, Docker's restart policy brings it back within ~1 s, and the restarted
|
|
// container mounts the CURRENT socket file.
|
|
//
|
|
// So: Watch.Tick pings the socket; once Docker has answered in this process's life and then refuses for Window
|
|
// (60 s) without a break, the controller exits (ExitCode) and Docker restarts it on the current socket. A timeout or
|
|
// a slow daemon never counts — only a refused or missing socket, the stale-inode signature. Watch.CheckUsers then
|
|
// restarts every OTHER container that mounts the socket and still holds an older inode than the controller's own
|
|
// (traefik today), so the router follows too.
|
|
package sockheal
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"log"
|
|
"net"
|
|
"os"
|
|
"strings"
|
|
"sync"
|
|
"syscall"
|
|
"time"
|
|
)
|
|
|
|
// SocketPath is the guest's Docker socket as the controller mounts it.
|
|
const SocketPath = "/var/run/docker.sock"
|
|
|
|
// ExitCode is the controller's exit status when it leaves to be restarted on the current socket.
|
|
const ExitCode = 75
|
|
|
|
// DefaultWindow is how long Docker must refuse, without a break, before the controller exits.
|
|
const DefaultWindow = 60 * time.Second
|
|
|
|
// Watch is the socket watch. Every field but Logger is a seam; New fills the real ones.
|
|
type Watch struct {
|
|
Ping func(ctx context.Context) error // nil = Docker answered
|
|
Exit func(code int)
|
|
Now func() time.Time
|
|
Window time.Duration
|
|
Logger *log.Logger
|
|
|
|
// CheckUsers seams.
|
|
OwnInode func() (uint64, error) // the socket inode THIS container sees
|
|
SocketUsers func(ctx context.Context) ([]User, error) // OTHER running containers that mount the socket
|
|
UserInode func(ctx context.Context, u User) (uint64, error) // the socket inode that container sees
|
|
Restart func(ctx context.Context, name string) error
|
|
|
|
mu sync.Mutex
|
|
everWorked bool
|
|
refusedSince time.Time
|
|
}
|
|
|
|
// Refused reports whether err is the stale-socket signature: connection refused, or no socket at all.
|
|
func Refused(err error) bool {
|
|
return errors.Is(err, syscall.ECONNREFUSED) || errors.Is(err, syscall.ENOENT)
|
|
}
|
|
|
|
// PingSocket dials the socket and asks Docker's /_ping. A refused dial returns the dial error unwrapped enough for
|
|
// Refused to see ECONNREFUSED.
|
|
func PingSocket(ctx context.Context, path string) error {
|
|
d := net.Dialer{Timeout: 3 * time.Second}
|
|
c, err := d.DialContext(ctx, "unix", path)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer c.Close()
|
|
_ = c.SetDeadline(time.Now().Add(5 * time.Second))
|
|
if _, err := io.WriteString(c, "GET /_ping HTTP/1.0\r\nHost: docker\r\n\r\n"); err != nil {
|
|
return err
|
|
}
|
|
buf := make([]byte, 64)
|
|
n, err := c.Read(buf)
|
|
if n == 0 && err != nil {
|
|
return err
|
|
}
|
|
if !strings.HasPrefix(string(buf[:n]), "HTTP/1.") || !strings.Contains(string(buf[:n]), " 200 ") {
|
|
return fmt.Errorf("docker /_ping answered %q", strings.SplitN(string(buf[:n]), "\r\n", 2)[0])
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// Tick is one check (the scheduler calls it every 15 s).
|
|
func (w *Watch) Tick(ctx context.Context) error {
|
|
err := w.Ping(ctx)
|
|
w.mu.Lock()
|
|
defer w.mu.Unlock()
|
|
now := w.Now()
|
|
if err == nil {
|
|
if !w.refusedSince.IsZero() {
|
|
w.Logger.Printf("[INFO] [sockheal] Docker answers again after %s of refusals — no restart needed",
|
|
now.Sub(w.refusedSince).Round(time.Second))
|
|
}
|
|
w.everWorked, w.refusedSince = true, time.Time{}
|
|
return nil
|
|
}
|
|
if !Refused(err) {
|
|
// A timeout or a slow daemon (a heavy backup, an engine step in progress) is not the stale-socket case.
|
|
w.Logger.Printf("[DEBUG] [sockheal] Docker ping failed, not counted (not a refusal): %v", err)
|
|
return nil
|
|
}
|
|
if !w.everWorked {
|
|
// Never reached Docker in this process: a restart would not help, and exiting would loop every Window.
|
|
w.Logger.Printf("[WARN] [sockheal] Docker refuses (%v) and never answered since this controller started — not exiting", err)
|
|
return nil
|
|
}
|
|
if w.refusedSince.IsZero() {
|
|
w.refusedSince = now
|
|
w.Logger.Printf("[WARN] [sockheal] Docker refuses the socket (%v) — exiting after %s of refusals so Docker restarts "+
|
|
"this controller on the current socket (R-860)", err, w.Window)
|
|
return nil
|
|
}
|
|
if gone := now.Sub(w.refusedSince); gone >= w.Window {
|
|
w.Logger.Printf("[ERROR] [sockheal] Docker has refused the socket for %s (%v) — the socket file was re-created and "+
|
|
"this container holds the old one; EXITING (code %d) so Docker's restart policy brings it back on the current "+
|
|
"socket (R-860)", gone.Round(time.Second), err, ExitCode)
|
|
w.Exit(ExitCode)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// CheckUsers restarts every OTHER running container that mounts the socket and sees a different inode than this
|
|
// controller — only while this controller itself reaches Docker (so its own inode is the current one).
|
|
func (w *Watch) CheckUsers(ctx context.Context) error {
|
|
if err := w.Ping(ctx); err != nil {
|
|
return nil // Tick handles a controller that cannot reach Docker
|
|
}
|
|
own, err := w.OwnInode()
|
|
if err != nil {
|
|
return fmt.Errorf("sockheal: own socket inode: %w", err)
|
|
}
|
|
users, err := w.SocketUsers(ctx)
|
|
if err != nil {
|
|
return fmt.Errorf("sockheal: list socket users: %w", err)
|
|
}
|
|
for _, u := range users {
|
|
name := u.Name
|
|
ino, err := w.UserInode(ctx, u)
|
|
if err != nil {
|
|
w.Logger.Printf("[DEBUG] [sockheal] %s: cannot read its socket inode (no stat in the image?): %v — skipped", name, err)
|
|
continue
|
|
}
|
|
if ino == own {
|
|
continue
|
|
}
|
|
w.Logger.Printf("[WARN] [sockheal] %s holds an old docker socket (inode %d, current %d) — restarting it (R-860)", name, ino, own)
|
|
if err := w.Restart(ctx, name); err != nil {
|
|
w.Logger.Printf("[ERROR] [sockheal] restart %s failed: %v", name, err)
|
|
continue
|
|
}
|
|
w.Logger.Printf("[INFO] [sockheal] %s restarted onto the current docker socket", name)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// User is a container that bind-mounts the Docker socket, and where.
|
|
type User struct {
|
|
Name string
|
|
Dest string
|
|
}
|
|
|
|
// SelfName is the controller's own container, never in SocketUsers' answer.
|
|
const SelfName = "felhom-controller"
|
|
|
|
// DockerSocketUsers lists the OTHER running containers that mount the socket (docker ps + one docker inspect).
|
|
func DockerSocketUsers(run func(ctx context.Context, args ...string) (string, error)) func(ctx context.Context) ([]User, error) {
|
|
return func(ctx context.Context) ([]User, error) {
|
|
ids, err := run(ctx, "ps", "-q", "--no-trunc")
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
fields := strings.Fields(ids)
|
|
if len(fields) == 0 {
|
|
return nil, nil
|
|
}
|
|
out, err := run(ctx, append([]string{"inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}"}, fields...)...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var users []User
|
|
for _, line := range strings.Split(strings.TrimSpace(out), "\n") {
|
|
name, mounts, ok := strings.Cut(strings.TrimSpace(line), "|")
|
|
name = strings.TrimPrefix(name, "/")
|
|
if !ok || name == SelfName {
|
|
continue
|
|
}
|
|
for _, m := range strings.Split(mounts, ";") {
|
|
if m == "/var/run/docker.sock" || m == "/run/docker.sock" {
|
|
users = append(users, User{Name: name, Dest: m})
|
|
break
|
|
}
|
|
}
|
|
}
|
|
return users, nil
|
|
}
|
|
}
|
|
|
|
// InodeOf returns a path's inode number.
|
|
func InodeOf(path string) (uint64, error) {
|
|
fi, err := os.Stat(path)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
st, ok := fi.Sys().(*syscall.Stat_t)
|
|
if !ok {
|
|
return 0, fmt.Errorf("no inode for %s", path)
|
|
}
|
|
return st.Ino, nil
|
|
}
|