diff --git a/CHANGELOG.md b/CHANGELOG.md index 3fe514c..8ee6fa1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,19 @@ +## v0.293.0 — the controller heals itself after the guest's Docker socket is re-created (R-860, generalises R-858) (2026-10-04) + +**MinAgent: 0.131.0** (unchanged). No new household string. + +- **R-860.** Measured on scratch guest 9202: `systemctl restart docker` and a killed dockerd keep the socket file + (systemd's `docker.socket` holds it) — nothing breaks. A restart of `docker.socket` (what a docker-ce upgrade does, + by any route, or by hand) RE-CREATES the file: with live-restore every container keeps running, but the controller + and traefik keep the deleted inode and get "connection refused" for ever while their health stays "healthy". +- New `internal/sockheal`: every 15 s the controller pings `/var/run/docker.sock`; once Docker has answered in this + process's life and then REFUSES (ECONNREFUSED / ENOENT — never a timeout) for 60 s without a break, the controller + exits (code 75) and Docker's restart policy brings it back on the current socket in ~1 s (measured). Every 5 min + and 30 s after start it restarts any OTHER running container that mounts the socket and sees an older inode than + its own (traefik today). A controller that never reached Docker does not exit (no loop). +- Tests: `internal/sockheal/sockheal_test.go` (real unix sockets in a temp dir for the stale-file case; no Docker). + Red-proof (9 of 9): `felhom.eu/documentation/audits/r840-config-bundle-2026-10-04/partE/e6-redproof-controller.txt`. + ## v0.292.0 — cloudflared says whether the tunnel is CONNECTED (R-841) (2026-10-04) **MinAgent: 0.131.0** (unchanged). No new household string. Agent v0.141.0 reads the result; an older agent ignores it. diff --git a/controller/README.md b/controller/README.md index f381c88..eeba595 100644 --- a/controller/README.md +++ b/controller/README.md @@ -1040,6 +1040,15 @@ hard **404** at its URL even though the container is running. The `routeUnpublis (`funcmap.go`) drives a distinct "URL nem elérhető – útvonal nincs publikálva" indicator on the dashboard and stacks cards for such apps, so a dead URL isn't mistaken for a merely-degraded-but-reachable one. +#### Docker socket self-heal (`internal/sockheal/`, v0.293.0, R-860) + +The controller and traefik bind-mount the Docker socket FILE. When `docker.socket` restarts (a docker-ce upgrade by any +route, or by hand), the file is re-created and a running container keeps the deleted one: with live-restore nothing +restarts, and both are blind to Docker. Every 15 s the controller pings the socket; after 60 s of "connection refused" +(never a timeout, and only after Docker had answered in this process) it exits, and Docker's restart policy brings it +back on the current socket. Every 5 min (and 30 s after start) it restarts any other container that mounts the socket +and still sees an older inode (traefik). Logs: `[sockheal]`. + #### Controller-side Health Probes (`internal/stacks/healthprobe.go`) For apps that declare a `healthcheck:` section in `.felhom.yml`, the controller probes the container directly over the Docker network (both are on `traefik-public`). This complements Docker-level healthchecks and is the **only** health mechanism for distroless/scratch images that lack shell utilities. diff --git a/controller/cmd/controller/main.go b/controller/cmd/controller/main.go index c877025..67ed5cf 100644 --- a/controller/cmd/controller/main.go +++ b/controller/cmd/controller/main.go @@ -7,6 +7,7 @@ import ( "flag" "fmt" "gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec" + "gitea.dooplex.hu/admin/felhom-controller/internal/sockheal" "io" "log" "net/http" @@ -18,6 +19,7 @@ import ( "time" "crypto/subtle" + "strconv" "strings" "sync" @@ -847,6 +849,21 @@ func main() { sched.Every("stack-scan", 2*time.Minute, func(ctx context.Context) error { return stackMgr.ScanStacks() }) + // R-860: heal after the guest's Docker socket FILE is re-created (a docker.socket restart — a docker-ce upgrade by + // any route, or by hand). The controller exits after 60 s of refusals so Docker restarts it on the current socket; + // then it restarts any other socket user (traefik) that still holds the old one. A slow Docker never counts. + sockWatch := newSocketWatch(logger) + sched.Every("docker-socket-watch", 15*time.Second, sockWatch.Tick) + sched.Every("docker-socket-users", 5*time.Minute, sockWatch.CheckUsers) + go func() { // once soon after start: the restart that healed THIS controller must heal traefik too + select { + case <-ctx.Done(): + case <-time.After(30 * time.Second): + if err := sockWatch.CheckUsers(ctx); err != nil { + logger.Printf("[WARN] [sockheal] start-up socket-user check: %v", err) + } + } + }() sched.Every("health-probes", 10*time.Second, func(ctx context.Context) error { return stackMgr.RunHealthProbes() }) @@ -2461,6 +2478,33 @@ func sameBootFleet(a, b []bootFleetSample) bool { // which is the whole point, since the pre-v0.190.0 bug was a candidate set derived too early. // // Called from main() in a goroutine. +// newSocketWatch wires sockheal to the real socket and the docker CLI (dockerexec: refused under go test, R-650). +func newSocketWatch(logger *log.Logger) *sockheal.Watch { + run := func(ctx context.Context, args ...string) (string, error) { + c, cancel := context.WithTimeout(ctx, 30*time.Second) + defer cancel() + out, err := dockerexec.CommandContext(c, "docker", args...).Output() + return string(out), err + } + return &sockheal.Watch{ + Ping: func(ctx context.Context) error { return sockheal.PingSocket(ctx, sockheal.SocketPath) }, + Exit: os.Exit, + Now: time.Now, + Window: sockheal.DefaultWindow, + Logger: logger, + OwnInode: func() (uint64, error) { return sockheal.InodeOf(sockheal.SocketPath) }, + SocketUsers: sockheal.DockerSocketUsers(run), + UserInode: func(ctx context.Context, u sockheal.User) (uint64, error) { + out, err := run(ctx, "exec", u.Name, "stat", "-c", "%i", u.Dest) + if err != nil { + return 0, err + } + return strconv.ParseUint(strings.TrimSpace(out), 10, 64) + }, + Restart: func(ctx context.Context, name string) error { _, err := run(ctx, "restart", name); return err }, + } +} + func runBootReconcile(ctx context.Context, mgr bootrecon.StackProvider, logger *log.Logger) { select { case <-ctx.Done(): diff --git a/controller/internal/sockheal/sockheal.go b/controller/internal/sockheal/sockheal.go new file mode 100644 index 0000000..0764d38 --- /dev/null +++ b/controller/internal/sockheal/sockheal.go @@ -0,0 +1,218 @@ +// Package sockheal lets the controller heal itself after the guest's Docker socket FILE is re-created (R-860, +// generalising R-858). +// +// MEASURED 2026-10-04 on scratch guest 9202 (`felhom.eu/documentation/audits/r840-config-bundle-2026-10-04/partE/`): +// - `systemctl restart docker` and a `kill -9` of dockerd keep the socket file (systemd's docker.socket holds it; +// same inode): the controller and traefik keep working. Nothing to heal. +// - a restart of docker.socket — what a docker-ce package upgrade does, or a by-hand `systemctl restart +// docker.socket` — RE-CREATES the file (inode 144 → 7202). With live-restore every container keeps running, but +// a container that bind-mounts the socket FILE keeps the deleted inode: the controller and traefik get +// "connection refused" for ever, while their own health checks stay "healthy". +// - when the controller's process exits, Docker's restart policy brings it back within ~1 s, and the restarted +// container mounts the CURRENT socket file. +// +// So: Watch.Tick pings the socket; once Docker has answered in this process's life and then refuses for Window +// (60 s) without a break, the controller exits (ExitCode) and Docker restarts it on the current socket. A timeout or +// a slow daemon never counts — only a refused or missing socket, the stale-inode signature. Watch.CheckUsers then +// restarts every OTHER container that mounts the socket and still holds an older inode than the controller's own +// (traefik today), so the router follows too. +package sockheal + +import ( + "context" + "errors" + "fmt" + "io" + "log" + "net" + "os" + "strings" + "sync" + "syscall" + "time" +) + +// SocketPath is the guest's Docker socket as the controller mounts it. +const SocketPath = "/var/run/docker.sock" + +// ExitCode is the controller's exit status when it leaves to be restarted on the current socket. +const ExitCode = 75 + +// DefaultWindow is how long Docker must refuse, without a break, before the controller exits. +const DefaultWindow = 60 * time.Second + +// Watch is the socket watch. Every field but Logger is a seam; New fills the real ones. +type Watch struct { + Ping func(ctx context.Context) error // nil = Docker answered + Exit func(code int) + Now func() time.Time + Window time.Duration + Logger *log.Logger + + // CheckUsers seams. + OwnInode func() (uint64, error) // the socket inode THIS container sees + SocketUsers func(ctx context.Context) ([]User, error) // OTHER running containers that mount the socket + UserInode func(ctx context.Context, u User) (uint64, error) // the socket inode that container sees + Restart func(ctx context.Context, name string) error + + mu sync.Mutex + everWorked bool + refusedSince time.Time +} + +// Refused reports whether err is the stale-socket signature: connection refused, or no socket at all. +func Refused(err error) bool { + return errors.Is(err, syscall.ECONNREFUSED) || errors.Is(err, syscall.ENOENT) +} + +// PingSocket dials the socket and asks Docker's /_ping. A refused dial returns the dial error unwrapped enough for +// Refused to see ECONNREFUSED. +func PingSocket(ctx context.Context, path string) error { + d := net.Dialer{Timeout: 3 * time.Second} + c, err := d.DialContext(ctx, "unix", path) + if err != nil { + return err + } + defer c.Close() + _ = c.SetDeadline(time.Now().Add(5 * time.Second)) + if _, err := io.WriteString(c, "GET /_ping HTTP/1.0\r\nHost: docker\r\n\r\n"); err != nil { + return err + } + buf := make([]byte, 64) + n, err := c.Read(buf) + if n == 0 && err != nil { + return err + } + if !strings.HasPrefix(string(buf[:n]), "HTTP/1.") || !strings.Contains(string(buf[:n]), " 200 ") { + return fmt.Errorf("docker /_ping answered %q", strings.SplitN(string(buf[:n]), "\r\n", 2)[0]) + } + return nil +} + +// Tick is one check (the scheduler calls it every 15 s). +func (w *Watch) Tick(ctx context.Context) error { + err := w.Ping(ctx) + w.mu.Lock() + defer w.mu.Unlock() + now := w.Now() + if err == nil { + if !w.refusedSince.IsZero() { + w.Logger.Printf("[INFO] [sockheal] Docker answers again after %s of refusals — no restart needed", + now.Sub(w.refusedSince).Round(time.Second)) + } + w.everWorked, w.refusedSince = true, time.Time{} + return nil + } + if !Refused(err) { + // A timeout or a slow daemon (a heavy backup, an engine step in progress) is not the stale-socket case. + w.Logger.Printf("[DEBUG] [sockheal] Docker ping failed, not counted (not a refusal): %v", err) + return nil + } + if !w.everWorked { + // Never reached Docker in this process: a restart would not help, and exiting would loop every Window. + w.Logger.Printf("[WARN] [sockheal] Docker refuses (%v) and never answered since this controller started — not exiting", err) + return nil + } + if w.refusedSince.IsZero() { + w.refusedSince = now + w.Logger.Printf("[WARN] [sockheal] Docker refuses the socket (%v) — exiting after %s of refusals so Docker restarts "+ + "this controller on the current socket (R-860)", err, w.Window) + return nil + } + if gone := now.Sub(w.refusedSince); gone >= w.Window { + w.Logger.Printf("[ERROR] [sockheal] Docker has refused the socket for %s (%v) — the socket file was re-created and "+ + "this container holds the old one; EXITING (code %d) so Docker's restart policy brings it back on the current "+ + "socket (R-860)", gone.Round(time.Second), err, ExitCode) + w.Exit(ExitCode) + } + return nil +} + +// CheckUsers restarts every OTHER running container that mounts the socket and sees a different inode than this +// controller — only while this controller itself reaches Docker (so its own inode is the current one). +func (w *Watch) CheckUsers(ctx context.Context) error { + if err := w.Ping(ctx); err != nil { + return nil // Tick handles a controller that cannot reach Docker + } + own, err := w.OwnInode() + if err != nil { + return fmt.Errorf("sockheal: own socket inode: %w", err) + } + users, err := w.SocketUsers(ctx) + if err != nil { + return fmt.Errorf("sockheal: list socket users: %w", err) + } + for _, u := range users { + name := u.Name + ino, err := w.UserInode(ctx, u) + if err != nil { + w.Logger.Printf("[DEBUG] [sockheal] %s: cannot read its socket inode (no stat in the image?): %v — skipped", name, err) + continue + } + if ino == own { + continue + } + w.Logger.Printf("[WARN] [sockheal] %s holds an old docker socket (inode %d, current %d) — restarting it (R-860)", name, ino, own) + if err := w.Restart(ctx, name); err != nil { + w.Logger.Printf("[ERROR] [sockheal] restart %s failed: %v", name, err) + continue + } + w.Logger.Printf("[INFO] [sockheal] %s restarted onto the current docker socket", name) + } + return nil +} + +// User is a container that bind-mounts the Docker socket, and where. +type User struct { + Name string + Dest string +} + +// SelfName is the controller's own container, never in SocketUsers' answer. +const SelfName = "felhom-controller" + +// DockerSocketUsers lists the OTHER running containers that mount the socket (docker ps + one docker inspect). +func DockerSocketUsers(run func(ctx context.Context, args ...string) (string, error)) func(ctx context.Context) ([]User, error) { + return func(ctx context.Context) ([]User, error) { + ids, err := run(ctx, "ps", "-q", "--no-trunc") + if err != nil { + return nil, err + } + fields := strings.Fields(ids) + if len(fields) == 0 { + return nil, nil + } + out, err := run(ctx, append([]string{"inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}"}, fields...)...) + if err != nil { + return nil, err + } + var users []User + for _, line := range strings.Split(strings.TrimSpace(out), "\n") { + name, mounts, ok := strings.Cut(strings.TrimSpace(line), "|") + name = strings.TrimPrefix(name, "/") + if !ok || name == SelfName { + continue + } + for _, m := range strings.Split(mounts, ";") { + if m == "/var/run/docker.sock" || m == "/run/docker.sock" { + users = append(users, User{Name: name, Dest: m}) + break + } + } + } + return users, nil + } +} + +// InodeOf returns a path's inode number. +func InodeOf(path string) (uint64, error) { + fi, err := os.Stat(path) + if err != nil { + return 0, err + } + st, ok := fi.Sys().(*syscall.Stat_t) + if !ok { + return 0, fmt.Errorf("no inode for %s", path) + } + return st.Ino, nil +} diff --git a/controller/internal/sockheal/sockheal_test.go b/controller/internal/sockheal/sockheal_test.go new file mode 100644 index 0000000..a493da2 --- /dev/null +++ b/controller/internal/sockheal/sockheal_test.go @@ -0,0 +1,193 @@ +package sockheal + +import ( + "bytes" + "context" + "errors" + "fmt" + "log" + "net" + "os" + "path/filepath" + "strings" + "syscall" + "testing" + "time" +) + +type clock struct{ t time.Time } + +func (c *clock) now() time.Time { return c.t } + +func newWatch(pingErr *error) (*Watch, *clock, *[]int, *bytes.Buffer) { + c := &clock{t: time.Date(2026, 10, 4, 17, 20, 0, 0, time.UTC)} + var exits []int + var buf bytes.Buffer + w := &Watch{Ping: func(context.Context) error { return *pingErr }, Exit: func(code int) { exits = append(exits, code) }, + Now: c.now, Window: DefaultWindow, Logger: log.New(&buf, "", 0)} + return w, c, &exits, &buf +} + +var refused = fmt.Errorf("dial unix /var/run/docker.sock: connect: %w", syscall.ECONNREFUSED) + +// The consequence: a controller that reached Docker and then is refused for 60 s EXITS (so Docker restarts it on the +// current socket) — not before 60 s, and with the agreed code. +func TestTick_RefusedForTheWindowExits(t *testing.T) { + var err error + w, c, exits, _ := newWatch(&err) + w.Tick(context.Background()) // worked once + err = refused + for i := 0; i < 4; i++ { // 0, 15, 30, 45 s + w.Tick(context.Background()) + c.t = c.t.Add(15 * time.Second) + } + if len(*exits) != 0 { + t.Fatalf("exited before the window: %v", *exits) + } + w.Tick(context.Background()) // 60 s + if len(*exits) != 1 || (*exits)[0] != ExitCode { + t.Fatalf("exits = %v, want one exit with %d", *exits, ExitCode) + } +} + +func TestTick_TimeoutsNeverCount(t *testing.T) { + var err error + w, c, exits, _ := newWatch(&err) + w.Tick(context.Background()) + err = fmt.Errorf("read: %w", os.ErrDeadlineExceeded) + for i := 0; i < 20; i++ { + w.Tick(context.Background()) + c.t = c.t.Add(15 * time.Second) + } + if len(*exits) != 0 { + t.Fatalf("a slow Docker must never restart the controller: %v", *exits) + } +} + +func TestTick_NeverWorkedNeverExits(t *testing.T) { + err := refused + w, c, exits, buf := newWatch(&err) + for i := 0; i < 20; i++ { + w.Tick(context.Background()) + c.t = c.t.Add(15 * time.Second) + } + if len(*exits) != 0 || !strings.Contains(buf.String(), "never answered") { + t.Fatalf("a controller that never reached Docker must not loop: exits=%v", *exits) + } +} + +func TestTick_ASuccessResetsTheWindow(t *testing.T) { + var err error + w, c, exits, _ := newWatch(&err) + w.Tick(context.Background()) + err = refused + w.Tick(context.Background()) + c.t = c.t.Add(45 * time.Second) + err = nil + w.Tick(context.Background()) + err = refused + c.t = c.t.Add(15 * time.Second) + w.Tick(context.Background()) + c.t = c.t.Add(45 * time.Second) + w.Tick(context.Background()) + if len(*exits) != 0 { + t.Fatalf("the window must restart after Docker answered: %v", *exits) + } +} + +func usersWatch(own uint64, inodes map[string]uint64, ping error) (*Watch, *[]string) { + var restarted []string + err := ping + w, _, _, _ := newWatch(&err) + w.OwnInode = func() (uint64, error) { return own, nil } + w.SocketUsers = func(context.Context) ([]User, error) { + var u []User + for n := range inodes { + u = append(u, User{Name: n, Dest: SocketPath}) + } + return u, nil + } + w.UserInode = func(_ context.Context, u User) (uint64, error) { + if inodes[u.Name] == 0 { + return 0, errors.New("stat: not found") + } + return inodes[u.Name], nil + } + w.Restart = func(_ context.Context, name string) error { restarted = append(restarted, name); return nil } + return w, &restarted +} + +// The measured case: the controller is on 7202, traefik still on 144 → traefik (only) is restarted. +func TestCheckUsers_RestartsOnlyTheStaleOnes(t *testing.T) { + w, restarted := usersWatch(7202, map[string]uint64{"traefik": 144, "dozzle": 7202, "nostat": 0}, nil) + if err := w.CheckUsers(context.Background()); err != nil { + t.Fatal(err) + } + if len(*restarted) != 1 || (*restarted)[0] != "traefik" { + t.Fatalf("restarted = %v, want [traefik]", *restarted) + } +} + +func TestCheckUsers_NothingWhileTheControllerItselfIsBlind(t *testing.T) { + w, restarted := usersWatch(144, map[string]uint64{"traefik": 7202}, refused) + _ = w.CheckUsers(context.Background()) + if len(*restarted) != 0 { + t.Fatalf("a blind controller's own inode is the OLD one; it must restart nothing: %v", *restarted) + } +} + +func TestDockerSocketUsers_ParsesAndSkipsItself(t *testing.T) { + run := func(_ context.Context, args ...string) (string, error) { + if args[0] == "ps" { + return "aaa\nbbb\nccc\n", nil + } + return "/felhom-controller|/mnt;/var/run/docker.sock;\n/traefik|/etc/traefik/acme.json;/var/run/docker.sock;\n/paperless|/data;\n/dozzle|/run/docker.sock;\n", nil + } + u, err := DockerSocketUsers(run)(context.Background()) + if err != nil { + t.Fatal(err) + } + if fmt.Sprint(u) != "[{traefik /var/run/docker.sock} {dozzle /run/docker.sock}]" { + t.Fatalf("users = %v", u) + } +} + +// PingSocket against REAL unix sockets (a temp dir, no Docker): a live one answers, a stale file (no listener — what a +// container keeps after docker.socket re-created the file) and a missing one are both Refused. +func TestPingSocket_RealSockets(t *testing.T) { + dir := t.TempDir() + live := filepath.Join(dir, "live.sock") + ln, err := net.Listen("unix", live) + if err != nil { + t.Fatal(err) + } + defer ln.Close() + go func() { + for { + c, err := ln.Accept() + if err != nil { + return + } + buf := make([]byte, 256) + _, _ = c.Read(buf) + _, _ = c.Write([]byte("HTTP/1.0 200 OK\r\nContent-Length: 2\r\n\r\nOK")) + c.Close() + } + }() + if err := PingSocket(context.Background(), live); err != nil { + t.Fatalf("live socket: %v", err) + } + stale := filepath.Join(dir, "stale.sock") + sl, err := net.Listen("unix", stale) + if err != nil { + t.Fatal(err) + } + sl.(*net.UnixListener).SetUnlinkOnClose(false) + sl.Close() + if err := PingSocket(context.Background(), stale); !Refused(err) { + t.Fatalf("stale socket file: want a refusal, got %v", err) + } + if err := PingSocket(context.Background(), filepath.Join(dir, "missing.sock")); !Refused(err) { + t.Fatalf("missing socket: want a refusal, got %v", err) + } +}