v0.293.0: the controller heals itself after the guest's Docker socket is re-created (R-860)
gates / gates (push) Successful in 29s
gates / gates (push) Successful in 29s
internal/sockheal: 60 s of refusals (never a timeout, only after Docker answered once) → exit 75 so Docker's restart policy brings the controller back on the current socket; every 5 min it restarts any other socket user (traefik) holding an older inode. Measured on 9202: only a docker.socket restart re-creates the file; dockerd crash / docker.service restart keep it. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,193 @@
|
||||
package sockheal
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
type clock struct{ t time.Time }
|
||||
|
||||
func (c *clock) now() time.Time { return c.t }
|
||||
|
||||
func newWatch(pingErr *error) (*Watch, *clock, *[]int, *bytes.Buffer) {
|
||||
c := &clock{t: time.Date(2026, 10, 4, 17, 20, 0, 0, time.UTC)}
|
||||
var exits []int
|
||||
var buf bytes.Buffer
|
||||
w := &Watch{Ping: func(context.Context) error { return *pingErr }, Exit: func(code int) { exits = append(exits, code) },
|
||||
Now: c.now, Window: DefaultWindow, Logger: log.New(&buf, "", 0)}
|
||||
return w, c, &exits, &buf
|
||||
}
|
||||
|
||||
var refused = fmt.Errorf("dial unix /var/run/docker.sock: connect: %w", syscall.ECONNREFUSED)
|
||||
|
||||
// The consequence: a controller that reached Docker and then is refused for 60 s EXITS (so Docker restarts it on the
|
||||
// current socket) — not before 60 s, and with the agreed code.
|
||||
func TestTick_RefusedForTheWindowExits(t *testing.T) {
|
||||
var err error
|
||||
w, c, exits, _ := newWatch(&err)
|
||||
w.Tick(context.Background()) // worked once
|
||||
err = refused
|
||||
for i := 0; i < 4; i++ { // 0, 15, 30, 45 s
|
||||
w.Tick(context.Background())
|
||||
c.t = c.t.Add(15 * time.Second)
|
||||
}
|
||||
if len(*exits) != 0 {
|
||||
t.Fatalf("exited before the window: %v", *exits)
|
||||
}
|
||||
w.Tick(context.Background()) // 60 s
|
||||
if len(*exits) != 1 || (*exits)[0] != ExitCode {
|
||||
t.Fatalf("exits = %v, want one exit with %d", *exits, ExitCode)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTick_TimeoutsNeverCount(t *testing.T) {
|
||||
var err error
|
||||
w, c, exits, _ := newWatch(&err)
|
||||
w.Tick(context.Background())
|
||||
err = fmt.Errorf("read: %w", os.ErrDeadlineExceeded)
|
||||
for i := 0; i < 20; i++ {
|
||||
w.Tick(context.Background())
|
||||
c.t = c.t.Add(15 * time.Second)
|
||||
}
|
||||
if len(*exits) != 0 {
|
||||
t.Fatalf("a slow Docker must never restart the controller: %v", *exits)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTick_NeverWorkedNeverExits(t *testing.T) {
|
||||
err := refused
|
||||
w, c, exits, buf := newWatch(&err)
|
||||
for i := 0; i < 20; i++ {
|
||||
w.Tick(context.Background())
|
||||
c.t = c.t.Add(15 * time.Second)
|
||||
}
|
||||
if len(*exits) != 0 || !strings.Contains(buf.String(), "never answered") {
|
||||
t.Fatalf("a controller that never reached Docker must not loop: exits=%v", *exits)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTick_ASuccessResetsTheWindow(t *testing.T) {
|
||||
var err error
|
||||
w, c, exits, _ := newWatch(&err)
|
||||
w.Tick(context.Background())
|
||||
err = refused
|
||||
w.Tick(context.Background())
|
||||
c.t = c.t.Add(45 * time.Second)
|
||||
err = nil
|
||||
w.Tick(context.Background())
|
||||
err = refused
|
||||
c.t = c.t.Add(15 * time.Second)
|
||||
w.Tick(context.Background())
|
||||
c.t = c.t.Add(45 * time.Second)
|
||||
w.Tick(context.Background())
|
||||
if len(*exits) != 0 {
|
||||
t.Fatalf("the window must restart after Docker answered: %v", *exits)
|
||||
}
|
||||
}
|
||||
|
||||
func usersWatch(own uint64, inodes map[string]uint64, ping error) (*Watch, *[]string) {
|
||||
var restarted []string
|
||||
err := ping
|
||||
w, _, _, _ := newWatch(&err)
|
||||
w.OwnInode = func() (uint64, error) { return own, nil }
|
||||
w.SocketUsers = func(context.Context) ([]User, error) {
|
||||
var u []User
|
||||
for n := range inodes {
|
||||
u = append(u, User{Name: n, Dest: SocketPath})
|
||||
}
|
||||
return u, nil
|
||||
}
|
||||
w.UserInode = func(_ context.Context, u User) (uint64, error) {
|
||||
if inodes[u.Name] == 0 {
|
||||
return 0, errors.New("stat: not found")
|
||||
}
|
||||
return inodes[u.Name], nil
|
||||
}
|
||||
w.Restart = func(_ context.Context, name string) error { restarted = append(restarted, name); return nil }
|
||||
return w, &restarted
|
||||
}
|
||||
|
||||
// The measured case: the controller is on 7202, traefik still on 144 → traefik (only) is restarted.
|
||||
func TestCheckUsers_RestartsOnlyTheStaleOnes(t *testing.T) {
|
||||
w, restarted := usersWatch(7202, map[string]uint64{"traefik": 144, "dozzle": 7202, "nostat": 0}, nil)
|
||||
if err := w.CheckUsers(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(*restarted) != 1 || (*restarted)[0] != "traefik" {
|
||||
t.Fatalf("restarted = %v, want [traefik]", *restarted)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCheckUsers_NothingWhileTheControllerItselfIsBlind(t *testing.T) {
|
||||
w, restarted := usersWatch(144, map[string]uint64{"traefik": 7202}, refused)
|
||||
_ = w.CheckUsers(context.Background())
|
||||
if len(*restarted) != 0 {
|
||||
t.Fatalf("a blind controller's own inode is the OLD one; it must restart nothing: %v", *restarted)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDockerSocketUsers_ParsesAndSkipsItself(t *testing.T) {
|
||||
run := func(_ context.Context, args ...string) (string, error) {
|
||||
if args[0] == "ps" {
|
||||
return "aaa\nbbb\nccc\n", nil
|
||||
}
|
||||
return "/felhom-controller|/mnt;/var/run/docker.sock;\n/traefik|/etc/traefik/acme.json;/var/run/docker.sock;\n/paperless|/data;\n/dozzle|/run/docker.sock;\n", nil
|
||||
}
|
||||
u, err := DockerSocketUsers(run)(context.Background())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if fmt.Sprint(u) != "[{traefik /var/run/docker.sock} {dozzle /run/docker.sock}]" {
|
||||
t.Fatalf("users = %v", u)
|
||||
}
|
||||
}
|
||||
|
||||
// PingSocket against REAL unix sockets (a temp dir, no Docker): a live one answers, a stale file (no listener — what a
|
||||
// container keeps after docker.socket re-created the file) and a missing one are both Refused.
|
||||
func TestPingSocket_RealSockets(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
live := filepath.Join(dir, "live.sock")
|
||||
ln, err := net.Listen("unix", live)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer ln.Close()
|
||||
go func() {
|
||||
for {
|
||||
c, err := ln.Accept()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
buf := make([]byte, 256)
|
||||
_, _ = c.Read(buf)
|
||||
_, _ = c.Write([]byte("HTTP/1.0 200 OK\r\nContent-Length: 2\r\n\r\nOK"))
|
||||
c.Close()
|
||||
}
|
||||
}()
|
||||
if err := PingSocket(context.Background(), live); err != nil {
|
||||
t.Fatalf("live socket: %v", err)
|
||||
}
|
||||
stale := filepath.Join(dir, "stale.sock")
|
||||
sl, err := net.Listen("unix", stale)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sl.(*net.UnixListener).SetUnlinkOnClose(false)
|
||||
sl.Close()
|
||||
if err := PingSocket(context.Background(), stale); !Refused(err) {
|
||||
t.Fatalf("stale socket file: want a refusal, got %v", err)
|
||||
}
|
||||
if err := PingSocket(context.Background(), filepath.Join(dir, "missing.sock")); !Refused(err) {
|
||||
t.Fatalf("missing socket: want a refusal, got %v", err)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user