R-858: after a Docker engine step the wrapper restarts ONLY the containers that mount the docker socket (controller, traefik); the health rule fails when the controller cannot reach Docker from inside its container (ruling 95)
gates / gates (push) Successful in 18s
gates / gates (push) Successful in 18s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
+28
-1
@@ -86,6 +86,11 @@ SIG_NAMESPACE = "felhom-op-v1"
|
||||
SIGNED_OP = "os_docker_step"
|
||||
NONCE_FILE = "/var/lib/felhom-os-apply/nonces.json"
|
||||
DAEMON_JSON = "/etc/docker/daemon.json"
|
||||
# R-858 (v0.142.1): a Docker engine step restarts dockerd, which RECREATES the socket file. With live-restore the
|
||||
# containers keep running — and one that bind-mounts the socket FILE keeps the deleted inode: measured 2026-10-04 on
|
||||
# demo-felhom, the controller and traefik were blind to Docker for 1h44m. After a step that installed something, the
|
||||
# wrapper restarts exactly the containers that mount one of these paths (never the apps, never the engine).
|
||||
DOCKER_SOCKETS = ("/var/run/docker.sock", "/run/docker.sock")
|
||||
CRASH_GUARD_STATE = "/var/lib/felhom-crash-guard/state.json"
|
||||
|
||||
|
||||
@@ -575,7 +580,10 @@ class Apply:
|
||||
if len(p) >= 4 and p[3]:
|
||||
cont[p[0]]["id"] = p[3]
|
||||
nrc, _, _ = self.g(["getent", "hosts", "deb.debian.org"], timeout=30)
|
||||
return {"docker_ok": rc == 0, "containers": cont,
|
||||
# R-858: the controller's own health check stayed "healthy" while it could not reach Docker at all — so ask
|
||||
# the consequence directly: can the controller talk to the engine from inside its container?
|
||||
crc, _, _ = self.g(["docker", "exec", "felhom-controller", "docker", "version", "--format", "{{.Server.Version}}"], timeout=60)
|
||||
return {"docker_ok": rc == 0, "containers": cont, "controller_docker_ok": crc == 0,
|
||||
"controller": cont.get("felhom-controller", {}).get("health", "absent"),
|
||||
"network_ok": nrc == 0}
|
||||
|
||||
@@ -728,6 +736,23 @@ class Apply:
|
||||
out.append({"name": p["name"], "version": p["to"], "origin": "Debian-Security" if "Debian-Security" in o else "Debian"})
|
||||
return out
|
||||
|
||||
def restart_socket_users(self):
|
||||
"""R-858: restart ONLY the containers that bind-mount the Docker socket, so they attach to the new one."""
|
||||
rc, out, _ = self.g(["docker", "ps", "-q", "--no-trunc"], timeout=60)
|
||||
users = []
|
||||
for cid in out.split():
|
||||
irc, iout, _ = self.g(["docker", "inspect", "-f", "{{.Name}}|{{range .Mounts}}{{.Destination}};{{end}}", cid], timeout=60)
|
||||
if irc != 0 or "|" not in iout:
|
||||
continue
|
||||
name, mounts = iout.strip().split("|", 1)
|
||||
if any(m in DOCKER_SOCKETS for m in mounts.split(";")):
|
||||
users.append(name.lstrip("/"))
|
||||
users.sort()
|
||||
if users:
|
||||
rrc, _, rerr = self.g(["docker", "restart"] + users, timeout=300)
|
||||
self.r.log(f"os-apply: SOCKET-USERS restarted={','.join(users)} rc={rrc} (R-858: they held the old docker socket)")
|
||||
return users
|
||||
|
||||
def pending_docker(self):
|
||||
"""Ring 0 (select pending-docker): the newest pending version of each INSTALLED Docker package, Docker origin."""
|
||||
rc, pend, remv, _ = self.simulate(["dist-upgrade"])
|
||||
@@ -837,6 +862,8 @@ class Apply:
|
||||
return 3, None
|
||||
self.report["upgraded"] = [{"name": n, "version": v} for n, v in upgrade]
|
||||
self.report["seconds"] = round(secs, 1)
|
||||
if self.layer == "docker":
|
||||
self.report["socket_restarted"] = self.restart_socket_users()
|
||||
procs, reboot = self.restart_needed()
|
||||
self.report["restart_needed"] = procs
|
||||
self.report["docker_restart_needed"] = any(p in ("dockerd", "containerd") for p in procs)
|
||||
|
||||
Reference in New Issue
Block a user