R-858: after a Docker engine step the wrapper restarts ONLY the containers that mount the docker socket (controller, traefik); the health rule fails when the controller cannot reach Docker from inside its container (ruling 95)
gates / gates (push) Successful in 18s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-04 18:21:24 +02:00
parent 42af3ab9bc
commit 495003051b
4 changed files with 83 additions and 1 deletions
+28 -1
View File
@@ -86,6 +86,11 @@ SIG_NAMESPACE = "felhom-op-v1"
SIGNED_OP = "os_docker_step"
NONCE_FILE = "/var/lib/felhom-os-apply/nonces.json"
DAEMON_JSON = "/etc/docker/daemon.json"
# R-858 (v0.142.1): a Docker engine step restarts dockerd, which RECREATES the socket file. With live-restore the
# containers keep running — and one that bind-mounts the socket FILE keeps the deleted inode: measured 2026-10-04 on
# demo-felhom, the controller and traefik were blind to Docker for 1h44m. After a step that installed something, the
# wrapper restarts exactly the containers that mount one of these paths (never the apps, never the engine).
DOCKER_SOCKETS = ("/var/run/docker.sock", "/run/docker.sock")
CRASH_GUARD_STATE = "/var/lib/felhom-crash-guard/state.json"
@@ -575,7 +580,10 @@ class Apply:
if len(p) >= 4 and p[3]:
cont[p[0]]["id"] = p[3]
nrc, _, _ = self.g(["getent", "hosts", "deb.debian.org"], timeout=30)
return {"docker_ok": rc == 0, "containers": cont,
# R-858: the controller's own health check stayed "healthy" while it could not reach Docker at all — so ask
# the consequence directly: can the controller talk to the engine from inside its container?
crc, _, _ = self.g(["docker", "exec", "felhom-controller", "docker", "version", "--format", "{{.Server.Version}}"], timeout=60)
return {"docker_ok": rc == 0, "containers": cont, "controller_docker_ok": crc == 0,
"controller": cont.get("felhom-controller", {}).get("health", "absent"),
"network_ok": nrc == 0}
@@ -728,6 +736,23 @@ class Apply:
out.append({"name": p["name"], "version": p["to"], "origin": "Debian-Security" if "Debian-Security" in o else "Debian"})
return out
def restart_socket_users(self):
"""R-858: restart ONLY the containers that bind-mount the Docker socket, so they attach to the new one."""
rc, out, _ = self.g(["docker", "ps", "-q", "--no-trunc"], timeout=60)
users = []
for cid in out.split():
irc, iout, _ = self.g(["docker", "inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}", cid], timeout=60)
if irc != 0 or "|" not in iout:
continue
name, mounts = iout.strip().split("|", 1)
if any(m in DOCKER_SOCKETS for m in mounts.split(";")):
users.append(name.lstrip("/"))
users.sort()
if users:
rrc, _, rerr = self.g(["docker", "restart"] + users, timeout=300)
self.r.log(f"os-apply: SOCKET-USERS restarted={','.join(users)} rc={rrc} (R-858: they held the old docker socket)")
return users
def pending_docker(self):
"""Ring 0 (select pending-docker): the newest pending version of each INSTALLED Docker package, Docker origin."""
rc, pend, remv, _ = self.simulate(["dist-upgrade"])
@@ -837,6 +862,8 @@ class Apply:
return 3, None
self.report["upgraded"] = [{"name": n, "version": v} for n, v in upgrade]
self.report["seconds"] = round(secs, 1)
if self.layer == "docker":
self.report["socket_restarted"] = self.restart_socket_users()
procs, reboot = self.restart_needed()
self.report["restart_needed"] = procs
self.report["docker_restart_needed"] = any(p in ("dockerd", "containerd") for p in procs)
+28
View File
@@ -186,6 +186,14 @@ class Fake:
return 0, self.engine + "\n", ""
if cmd == "docker" and a[1:3] == ["ps", "-q"]:
return 0, "".join(i + "\n" for i in self.ids), ""
if cmd == "docker" and a[1] == "inspect":
mounts = {"aaa111": "/felhom-controller|/var/run/docker.sock;/app/data;", "bbb222": "/app|/data;"}
return 0, mounts.get(a[-1], "/other|;") + "\n", ""
if cmd == "docker" and a[1] == "restart":
self.restarted_containers = a[2:]
return 0, "", ""
if cmd == "docker" and a[1] == "exec":
return (1, "", "Cannot connect to the Docker daemon") if getattr(self, "controller_blind", False) else (0, self.engine + "\n", "")
if cmd == "docker":
ids = self.ids + ["x"] * 2
return 0, f"felhom-controller\trunning\tUp 1 hour (healthy)\t{ids[0]}\napp\trunning\tUp 1 hour (healthy)\t{ids[1]}\n", ""
@@ -804,6 +812,26 @@ class DockerLane(unittest.TestCase):
rc, rep = run(f)
self.assertEqual(rep["plan"]["upgrade"], 0, "an older version on an unsigned step is 'already', never installed")
def test_step_restarts_only_the_socket_users(self):
# R-858: after an engine step, ONLY the container that mounts the docker socket is restarted (here the controller)
f = docker_fake(signed=signed_job())
rc, rep = run(f)
self.assertEqual(rc, 0, rep)
self.assertEqual(f.restarted_containers, ["felhom-controller"])
self.assertEqual(rep["socket_restarted"], ["felhom-controller"])
def test_no_install_restarts_nothing(self):
f = docker_fake(signed=signed_job())
f.installed.update({"docker-ce": "5:29.8.2-1~debian.13~trixie", "containerd.io": "2.3.6-1~debian.13~trixie"})
rc, rep = run(f)
self.assertFalse(hasattr(f, "restarted_containers"), "nothing installed -> no container restart")
def test_health_says_when_the_controller_cannot_reach_docker(self):
f = docker_fake(signed=signed_job())
f.controller_blind = True
rc, rep = run(f)
self.assertFalse(rep["health_after"]["controller_docker_ok"])
def test_health_carries_container_ids(self):
f = docker_fake(signed=signed_job())
rc, rep = run(f)