R-858: after a Docker engine step the wrapper restarts ONLY the containers that mount the docker socket (controller, traefik); the health rule fails when the controller cannot reach Docker from inside its container (ruling 95)
gates / gates (push) Successful in 18s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-04 18:21:24 +02:00
parent 42af3ab9bc
commit 495003051b
4 changed files with 83 additions and 1 deletions
+28 -1
View File
@@ -86,6 +86,11 @@ SIG_NAMESPACE = "felhom-op-v1"
SIGNED_OP = "os_docker_step"
NONCE_FILE = "/var/lib/felhom-os-apply/nonces.json"
DAEMON_JSON = "/etc/docker/daemon.json"
# R-858 (v0.142.1): a Docker engine step restarts dockerd, which RECREATES the socket file. With live-restore the
# containers keep running — and one that bind-mounts the socket FILE keeps the deleted inode: measured 2026-10-04 on
# demo-felhom, the controller and traefik were blind to Docker for 1h44m. After a step that installed something, the
# wrapper restarts exactly the containers that mount one of these paths (never the apps, never the engine).
DOCKER_SOCKETS = ("/var/run/docker.sock", "/run/docker.sock")
CRASH_GUARD_STATE = "/var/lib/felhom-crash-guard/state.json"
@@ -575,7 +580,10 @@ class Apply:
if len(p) >= 4 and p[3]:
cont[p[0]]["id"] = p[3]
nrc, _, _ = self.g(["getent", "hosts", "deb.debian.org"], timeout=30)
return {"docker_ok": rc == 0, "containers": cont,
# R-858: the controller's own health check stayed "healthy" while it could not reach Docker at all — so ask
# the consequence directly: can the controller talk to the engine from inside its container?
crc, _, _ = self.g(["docker", "exec", "felhom-controller", "docker", "version", "--format", "{{.Server.Version}}"], timeout=60)
return {"docker_ok": rc == 0, "containers": cont, "controller_docker_ok": crc == 0,
"controller": cont.get("felhom-controller", {}).get("health", "absent"),
"network_ok": nrc == 0}
@@ -728,6 +736,23 @@ class Apply:
out.append({"name": p["name"], "version": p["to"], "origin": "Debian-Security" if "Debian-Security" in o else "Debian"})
return out
def restart_socket_users(self):
"""R-858: restart ONLY the containers that bind-mount the Docker socket, so they attach to the new one."""
rc, out, _ = self.g(["docker", "ps", "-q", "--no-trunc"], timeout=60)
users = []
for cid in out.split():
irc, iout, _ = self.g(["docker", "inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}", cid], timeout=60)
if irc != 0 or "|" not in iout:
continue
name, mounts = iout.strip().split("|", 1)
if any(m in DOCKER_SOCKETS for m in mounts.split(";")):
users.append(name.lstrip("/"))
users.sort()
if users:
rrc, _, rerr = self.g(["docker", "restart"] + users, timeout=300)
self.r.log(f"os-apply: SOCKET-USERS restarted={','.join(users)} rc={rrc} (R-858: they held the old docker socket)")
return users
def pending_docker(self):
"""Ring 0 (select pending-docker): the newest pending version of each INSTALLED Docker package, Docker origin."""
rc, pend, remv, _ = self.simulate(["dist-upgrade"])
@@ -837,6 +862,8 @@ class Apply:
return 3, None
self.report["upgraded"] = [{"name": n, "version": v} for n, v in upgrade]
self.report["seconds"] = round(secs, 1)
if self.layer == "docker":
self.report["socket_restarted"] = self.restart_socket_users()
procs, reboot = self.restart_needed()
self.report["restart_needed"] = procs
self.report["docker_restart_needed"] = any(p in ("dockerd", "containerd") for p in procs)
+28
View File
@@ -186,6 +186,14 @@ class Fake:
return 0, self.engine + "\n", ""
if cmd == "docker" and a[1:3] == ["ps", "-q"]:
return 0, "".join(i + "\n" for i in self.ids), ""
if cmd == "docker" and a[1] == "inspect":
mounts = {"aaa111": "/felhom-controller|/var/run/docker.sock;/app/data;", "bbb222": "/app|/data;"}
return 0, mounts.get(a[-1], "/other|;") + "\n", ""
if cmd == "docker" and a[1] == "restart":
self.restarted_containers = a[2:]
return 0, "", ""
if cmd == "docker" and a[1] == "exec":
return (1, "", "Cannot connect to the Docker daemon") if getattr(self, "controller_blind", False) else (0, self.engine + "\n", "")
if cmd == "docker":
ids = self.ids + ["x"] * 2
return 0, f"felhom-controller\trunning\tUp 1 hour (healthy)\t{ids[0]}\napp\trunning\tUp 1 hour (healthy)\t{ids[1]}\n", ""
@@ -804,6 +812,26 @@ class DockerLane(unittest.TestCase):
rc, rep = run(f)
self.assertEqual(rep["plan"]["upgrade"], 0, "an older version on an unsigned step is 'already', never installed")
def test_step_restarts_only_the_socket_users(self):
# R-858: after an engine step, ONLY the container that mounts the docker socket is restarted (here the controller)
f = docker_fake(signed=signed_job())
rc, rep = run(f)
self.assertEqual(rc, 0, rep)
self.assertEqual(f.restarted_containers, ["felhom-controller"])
self.assertEqual(rep["socket_restarted"], ["felhom-controller"])
def test_no_install_restarts_nothing(self):
f = docker_fake(signed=signed_job())
f.installed.update({"docker-ce": "5:29.8.2-1~debian.13~trixie", "containerd.io": "2.3.6-1~debian.13~trixie"})
rc, rep = run(f)
self.assertFalse(hasattr(f, "restarted_containers"), "nothing installed -> no container restart")
def test_health_says_when_the_controller_cannot_reach_docker(self):
f = docker_fake(signed=signed_job())
f.controller_blind = True
rc, rep = run(f)
self.assertFalse(rep["health_after"]["controller_docker_ok"])
def test_health_carries_container_ids(self):
f = docker_fake(signed=signed_job())
rc, rep = run(f)
+6
View File
@@ -75,6 +75,9 @@ type Health struct {
NetworkOK bool `json:"network_ok"`
Controller string `json:"controller"`
Containers map[string]Container `json:"containers"`
// ControllerDockerOK: the controller reaches the engine from INSIDE its container (R-858, wrapper ≥ v0.142.1;
// nil from an older wrapper = not checked). Its own health check stayed "healthy" while it was blind.
ControllerDockerOK *bool `json:"controller_docker_ok,omitempty"`
HostServices map[string]string `json:"host_services,omitempty"`
GuestRunning *bool `json:"guest_running,omitempty"`
Guest *Health `json:"guest,omitempty"`
@@ -240,6 +243,9 @@ func HealthVerdict(before, after *Health) (bool, string) {
if after.Controller != "healthy" {
return false, "the controller is " + after.Controller
}
if after.ControllerDockerOK != nil && !*after.ControllerDockerOK {
return false, "the controller cannot reach Docker (it holds an old socket — R-858)"
}
if before == nil {
return true, ""
}
+21
View File
@@ -465,3 +465,24 @@ func TestDocker_ChangedIDIsHealthFailed(t *testing.T) {
t.Fatalf("docker = %+v", p.Docker)
}
}
// R-858: a controller that cannot reach Docker fails the health rule even though its own check says healthy.
// Red-proof: drop the ControllerDockerOK check in HealthVerdict and this fails.
func TestHealthVerdict_ControllerBlindToDockerFails(t *testing.T) {
no, yes2 := false, true
after := guestOK()
after.ControllerDockerOK = &no
if ok, why := HealthVerdict(guestOK(), after); ok || !strings.Contains(why, "R-858") {
t.Fatalf("a blind controller passed: %v %q", ok, why)
}
if ok, _ := DockerHealthVerdict(guestOK(), after, "", ""); ok {
t.Fatal("the Docker rule passed a blind controller")
}
after.ControllerDockerOK = &yes2
if ok, why := HealthVerdict(guestOK(), after); !ok {
t.Fatalf("a seeing controller failed: %s", why)
}
if ok, _ := HealthVerdict(guestOK(), guestOK()); !ok {
t.Fatal("an older wrapper (no field) must not fail")
}
}