R-858: after a Docker engine step the wrapper restarts ONLY the containers that mount the docker socket (controller, traefik); the health rule fails when the controller cannot reach Docker from inside its container (ruling 95)
gates / gates (push) Successful in 18s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-04 18:21:24 +02:00
parent 42af3ab9bc
commit 495003051b
4 changed files with 83 additions and 1 deletions
+28 -1
View File
@@ -86,6 +86,11 @@ SIG_NAMESPACE = "felhom-op-v1"
SIGNED_OP = "os_docker_step"
NONCE_FILE = "/var/lib/felhom-os-apply/nonces.json"
DAEMON_JSON = "/etc/docker/daemon.json"
# R-858 (v0.142.1): a Docker engine step restarts dockerd, which RECREATES the socket file. With live-restore the
# containers keep running — and one that bind-mounts the socket FILE keeps the deleted inode: measured 2026-10-04 on
# demo-felhom, the controller and traefik were blind to Docker for 1h44m. After a step that installed something, the
# wrapper restarts exactly the containers that mount one of these paths (never the apps, never the engine).
DOCKER_SOCKETS = ("/var/run/docker.sock", "/run/docker.sock")
CRASH_GUARD_STATE = "/var/lib/felhom-crash-guard/state.json"
@@ -575,7 +580,10 @@ class Apply:
if len(p) >= 4 and p[3]:
cont[p[0]]["id"] = p[3]
nrc, _, _ = self.g(["getent", "hosts", "deb.debian.org"], timeout=30)
return {"docker_ok": rc == 0, "containers": cont,
# R-858: the controller's own health check stayed "healthy" while it could not reach Docker at all — so ask
# the consequence directly: can the controller talk to the engine from inside its container?
crc, _, _ = self.g(["docker", "exec", "felhom-controller", "docker", "version", "--format", "{{.Server.Version}}"], timeout=60)
return {"docker_ok": rc == 0, "containers": cont, "controller_docker_ok": crc == 0,
"controller": cont.get("felhom-controller", {}).get("health", "absent"),
"network_ok": nrc == 0}
@@ -728,6 +736,23 @@ class Apply:
out.append({"name": p["name"], "version": p["to"], "origin": "Debian-Security" if "Debian-Security" in o else "Debian"})
return out
def restart_socket_users(self):
"""R-858: restart ONLY the containers that bind-mount the Docker socket, so they attach to the new one."""
rc, out, _ = self.g(["docker", "ps", "-q", "--no-trunc"], timeout=60)
users = []
for cid in out.split():
irc, iout, _ = self.g(["docker", "inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}", cid], timeout=60)
if irc != 0 or "|" not in iout:
continue
name, mounts = iout.strip().split("|", 1)
if any(m in DOCKER_SOCKETS for m in mounts.split(";")):
users.append(name.lstrip("/"))
users.sort()
if users:
rrc, _, rerr = self.g(["docker", "restart"] + users, timeout=300)
self.r.log(f"os-apply: SOCKET-USERS restarted={','.join(users)} rc={rrc} (R-858: they held the old docker socket)")
return users
def pending_docker(self):
"""Ring 0 (select pending-docker): the newest pending version of each INSTALLED Docker package, Docker origin."""
rc, pend, remv, _ = self.simulate(["dist-upgrade"])
@@ -837,6 +862,8 @@ class Apply:
return 3, None
self.report["upgraded"] = [{"name": n, "version": v} for n, v in upgrade]
self.report["seconds"] = round(secs, 1)
if self.layer == "docker":
self.report["socket_restarted"] = self.restart_socket_users()
procs, reboot = self.restart_needed()
self.report["restart_needed"] = procs
self.report["docker_restart_needed"] = any(p in ("dockerd", "containerd") for p in procs)