v0.84.0: ReassertNetworkMounts — NAS automount survives guest reboots (RCA fix 1)

Storage §8 decision table (stop + enable --now on idle triggers; active mounts untouched),
daemon leg at startup with per-running-guest visibility verify, guest-hook post-start leg
(root, direct systemctl, non-fatal). Red-proofs: always-rearm table FAIL; unwired hook FAIL.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PSK5g6qYLknKj8u3QAFEr6
This commit is contained in:
2026-07-11 20:46:59 +02:00
parent 0df72ea643
commit 474b858c0b
12 changed files with 715 additions and 27 deletions
+40 -14
View File
@@ -450,12 +450,43 @@ func (h *SudoHostOps) RemoveNetworkMount(ctx context.Context, name string) error
// timeout — it NEVER stat()s the (possibly EIO/D-state) mountpoint, so a black-holed NAS cannot wedge
// this call. Best-effort: an unreadable unit dir yields an empty list.
func (h *SudoHostOps) ListNetworkMounts(ctx context.Context) ([]NetworkMountStatus, error) {
entries, err := h.networkUnitEntries()
if err != nil {
return nil, err
}
mounts := h.mountedFSTypes()
var out []NetworkMountStatus
for _, e := range entries {
st := NetworkMountStatus{
Name: e.name,
Protocol: e.proto,
Server: e.server,
Export: e.export,
Where: e.where,
Configured: true,
Mounted: isNetworkMounted(mounts[e.where]),
Reachable: endpointReachable(netEndpoint(e.proto, e.server)),
}
st.Health = networkHealth(st.Reachable, st.Mounted)
out = append(out, st)
}
return out, nil
}
// networkUnitEntry is one installed network-storage unit pair, parsed from its .mount file.
type networkUnitEntry struct {
name, proto, server, export, where string
}
// networkUnitEntries enumerates the installed network-storage unit pairs from the (world-readable)
// unit dir — the shared core of ListNetworkMounts and ReassertNetworkAutomounts. No probing, no
// mountpoint access. Best-effort per file; an unreadable unit dir is the only error.
func (h *SudoHostOps) networkUnitEntries() ([]networkUnitEntry, error) {
entries, err := os.ReadDir(h.unitDir)
if err != nil {
return nil, fmt.Errorf("netmount: reading unit dir: %w", err)
}
mounts := h.mountedFSTypes()
var out []NetworkMountStatus
var out []networkUnitEntry
for _, e := range entries {
if !strings.HasSuffix(e.Name(), ".mount") {
continue // the .mount unit carries the What/Type; the .automount mirrors Where only
@@ -468,18 +499,13 @@ func (h *SudoHostOps) ListNetworkMounts(ctx context.Context) ([]NetworkMountStat
if !ok {
continue // not one of ours (or a drive by-uuid mount)
}
st := NetworkMountStatus{
Name: strings.TrimPrefix(where, NetworkMountRoot+"/"),
Protocol: proto,
Server: server,
Export: export,
Where: where,
Configured: true,
Mounted: isNetworkMounted(mounts[where]),
Reachable: endpointReachable(netEndpoint(proto, server)),
}
st.Health = networkHealth(st.Reachable, st.Mounted)
out = append(out, st)
out = append(out, networkUnitEntry{
name: strings.TrimPrefix(where, NetworkMountRoot+"/"),
proto: proto,
server: server,
export: export,
where: where,
})
}
return out, nil
}
+117
View File
@@ -0,0 +1,117 @@
package storage
import (
"context"
"fmt"
"strings"
)
// Network-mount guest-reboot reassert (RCA AUDIT-nas-cwa-rca-2026-07-11 fix 1).
//
// THE BUG IT FIXES: a fresh guest namespace inherits REAL submounts of the shared parent (ext4, or
// an actively-mounted nfs4) but NOT an idle autofs trigger — so after any guest reboot an idle NAS
// share silently degrades to a local stub directory inside the guest. The heal is host-side and
// host-global: re-creating the .automount unit emits a FRESH trigger-mount event, which propagates
// live into every running guest's slave bind (live-proven in the RCA remediation, 2026-07-11).
//
// The action uses the sudoers-granted verbs only (`systemctl stop -- *.automount` +
// `systemctl enable --now -- *.automount`; there is NO restart grant). Idempotent: re-arming an
// already-armed trigger just recreates it — same end state, and the fresh mount event is harmless.
// An ACTIVE real mount is never touched (stopping the automount of a live mount would churn it).
// Reassert actions (the §8 decision table, encoded).
const (
// NetReassertRearmed: the trigger was re-created (stop + enable --now) — the propagation heal.
NetReassertRearmed = "rearmed"
// NetReassertSkipActive: a real network fs is mounted at the path — inherited by fresh
// namespaces, nothing to do (verify only).
NetReassertSkipActive = "skip-active"
// NetReassertSkipNone: neither a real mount nor an armed trigger at the path — a removed or
// orphan state owned by the add/remove flows, not this reconcile.
NetReassertSkipNone = "skip-none"
)
// NetReassertResult is one share's outcome in a reassert pass.
type NetReassertResult struct {
Name string // share name (mount dir basename)
Where string // guest-visible mountpoint (/mnt/felhom-drives/<name>)
Action string // NetReassert* constant
Err error // set when the rearm action failed (skip rows never error)
}
// netReassertAction is the pure §8 decision: the /proc/mounts fstype at the share's mountpoint
// ("" = nothing mounted there) → the action to take.
func netReassertAction(fstype string) string {
switch {
case isNetworkMounted(fstype):
return NetReassertSkipActive
case fstype == "autofs":
return NetReassertRearmed
default:
return NetReassertSkipNone
}
}
// ReassertNetworkAutomounts runs the reassert pass over every configured network mount: for each
// installed pair, decide per netReassertAction and re-arm idle triggers. Returns one result per
// share so callers (agent startup / guest-hook post-start) can verify guest visibility. Errors on
// one share never stop the pass. Callers MUST NOT invoke this from periodic health paths — an idle
// trigger is healthy, and the pass is only needed after a guest (re)start or at agent startup.
func (h *SudoHostOps) ReassertNetworkAutomounts(ctx context.Context) []NetReassertResult {
entries, err := h.networkUnitEntries()
if err != nil {
h.logger.Warn("netreassert: unit enumeration failed", "err", err)
return nil
}
fstypes := map[string]string{}
if h.host != nil {
if mounts, merr := h.host.Mounts(); merr == nil {
for _, m := range mounts {
fstypes[m.MountPoint] = m.FSType
}
} else {
h.logger.Warn("netreassert: mount-table read failed — treating all shares as unmounted", "err", merr)
}
}
var out []NetReassertResult
for _, e := range entries {
res := NetReassertResult{Name: e.name, Where: e.where, Action: netReassertAction(fstypes[e.where])}
switch res.Action {
case NetReassertSkipActive:
h.logger.Debug("netreassert: share actively mounted — skip (fresh namespaces inherit real mounts)",
"name", e.name, "where", e.where)
case NetReassertSkipNone:
h.logger.Debug("netreassert: no mount and no armed trigger — skip (removed/orphan state owned elsewhere)",
"name", e.name, "where", e.where)
case NetReassertRearmed:
if err := h.rearmNetworkAutomount(ctx, e.where); err != nil {
res.Err = err
h.logger.Warn("netreassert: trigger re-arm failed", "name", e.name, "where", e.where, "err", err)
} else {
h.logger.Info("netreassert: automount trigger re-armed (fresh mount event propagates into running guests)",
"name", e.name, "where", e.where)
}
}
out = append(out, res)
}
return out
}
// rearmNetworkAutomount stops then re-enables+starts the .automount for a mountpoint. The stop is
// tolerated failing (unit not loaded); the enable --now is the action that must succeed. Both verbs
// are the existing FELHOM_NETMOUNT sudoers grants.
func (h *SudoHostOps) rearmNetworkAutomount(ctx context.Context, where string) error {
mountUnit, err := UnitNameForMount(where)
if err != nil {
return fmt.Errorf("netreassert: unit name for %s: %w", where, err)
}
automountUnit := strings.TrimSuffix(mountUnit, ".mount") + ".automount"
if err := h.run(ctx, h.bins.Systemctl, "stop", "--", automountUnit); err != nil {
// Tolerated: a not-loaded unit still enables cleanly below.
h.logger.Debug("netreassert: automount stop tolerated failure", "unit", automountUnit, "err", err)
}
if err := h.run(ctx, h.bins.Systemctl, "enable", "--now", "--", automountUnit); err != nil {
return fmt.Errorf("netreassert: enabling automount %s: %w", automountUnit, err)
}
return nil
}
+160
View File
@@ -0,0 +1,160 @@
package storage
import (
"context"
"os"
"path/filepath"
"runtime"
"strings"
"testing"
)
// The §8 decision table, encoded exactly (RCA AUDIT-nas-cwa-rca-2026-07-11 fix 1).
func TestNetReassertAction_Table(t *testing.T) {
cases := []struct {
fstype string
want string
}{
{"nfs4", NetReassertSkipActive}, // real mount — inherited by fresh namespaces
{"nfs", NetReassertSkipActive}, // real mount
{"cifs", NetReassertSkipActive}, // real mount
{"autofs", NetReassertRearmed}, // idle trigger — NOT inherited, re-arm to propagate
{"", NetReassertSkipNone}, // nothing at the path — removed/orphan, owned elsewhere
{"ext4", NetReassertSkipNone}, // a local fs at the path is not a network state we own
{"tmpfs", NetReassertSkipNone}, //
}
for _, c := range cases {
if got := netReassertAction(c.fstype); got != c.want {
t.Errorf("netReassertAction(%q) = %q, want %q", c.fstype, got, c.want)
}
}
}
// installNetUnitFile writes a rendered network .mount unit straight into unitDir (bypassing the
// runner-mediated install — the reassert only READS unit files).
func installNetUnitFile(t *testing.T, unitDir string, spec NetworkMountSpec) (where, automountUnit string) {
t.Helper()
where = spec.Where()
mountUnit, err := UnitNameForMount(where)
if err != nil {
t.Fatalf("unit name: %v", err)
}
if err := os.WriteFile(filepath.Join(unitDir, mountUnit), []byte(renderNetworkMountUnit(spec)), 0o644); err != nil {
t.Fatalf("write unit: %v", err)
}
return where, strings.TrimSuffix(mountUnit, ".mount") + ".automount"
}
func netReassertOps(t *testing.T, unitDir string, mounts []Mount) (*SudoHostOps, *recordingRunner) {
t.Helper()
rr := &recordingRunner{}
ops := NewSudoHostOps(SudoHostOpsConfig{
Runner: rr, Bins: Binaries{}.withDefaults(), UnitDir: unitDir, StageDir: t.TempDir(),
Host: &fakeHostReader{mounts: mounts}, Logger: quietLogger(),
})
return ops, rr
}
// An idle trigger (autofs at the mountpoint) must be re-armed with EXACTLY the granted verbs:
// `systemctl stop -- <unit>.automount` then `systemctl enable --now -- <unit>.automount`.
func TestReassertNetworkAutomounts_RearmsIdleTrigger(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("systemd-escaped unit filename contains a backslash; exercised on the Linux build server")
}
unitDir := t.TempDir()
spec := NetworkMountSpec{Name: "media", Protocol: ProtocolNFS, Server: "10.0.0.5", Export: "/srv/media", MappedUID: 1000, MappedGID: 1000}
where, autoUnit := installNetUnitFile(t, unitDir, spec)
ops, rr := netReassertOps(t, unitDir, []Mount{{MountPoint: where, FSType: "autofs"}})
results := ops.ReassertNetworkAutomounts(context.Background())
if len(results) != 1 || results[0].Action != NetReassertRearmed || results[0].Err != nil {
t.Fatalf("want one rearmed result, got %+v", results)
}
if results[0].Where != where || results[0].Name != "media" {
t.Fatalf("result identity wrong: %+v", results[0])
}
if len(rr.calls) != 2 {
t.Fatalf("want exactly stop + enable --now, got %d calls: %v", len(rr.calls), rr.calls)
}
stop, enable := strings.Join(rr.calls[0], " "), strings.Join(rr.calls[1], " ")
if !strings.Contains(stop, "systemctl stop -- "+autoUnit) {
t.Errorf("first call must be `systemctl stop -- %s`, got: %s", autoUnit, stop)
}
if !strings.Contains(enable, "systemctl enable --now -- "+autoUnit) {
t.Errorf("second call must be `systemctl enable --now -- %s`, got: %s", autoUnit, enable)
}
}
// An ACTIVE real mount must not be touched — stopping the automount of a live mount would churn it.
// (Red-proof companion: a naive always-rearm implementation fails this with 2 recorded calls.)
func TestReassertNetworkAutomounts_ActiveMountUntouched(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("systemd-escaped unit filename contains a backslash; exercised on the Linux build server")
}
unitDir := t.TempDir()
spec := NetworkMountSpec{Name: "media", Protocol: ProtocolNFS, Server: "10.0.0.5", Export: "/srv/media", MappedUID: 1000, MappedGID: 1000}
where, _ := installNetUnitFile(t, unitDir, spec)
ops, rr := netReassertOps(t, unitDir, []Mount{{MountPoint: where, FSType: "nfs4"}})
results := ops.ReassertNetworkAutomounts(context.Background())
if len(results) != 1 || results[0].Action != NetReassertSkipActive {
t.Fatalf("want one skip-active result, got %+v", results)
}
if len(rr.calls) != 0 {
t.Fatalf("an actively-mounted share must trigger ZERO systemctl calls, got: %v", rr.calls)
}
}
// Neither a mount nor an armed trigger → skip (removed/orphan state, owned by add/remove flows).
func TestReassertNetworkAutomounts_NoTriggerSkips(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("systemd-escaped unit filename contains a backslash; exercised on the Linux build server")
}
unitDir := t.TempDir()
spec := NetworkMountSpec{Name: "media", Protocol: ProtocolNFS, Server: "10.0.0.5", Export: "/srv/media", MappedUID: 1000, MappedGID: 1000}
installNetUnitFile(t, unitDir, spec)
ops, rr := netReassertOps(t, unitDir, nil) // nothing at the mountpoint
results := ops.ReassertNetworkAutomounts(context.Background())
if len(results) != 1 || results[0].Action != NetReassertSkipNone {
t.Fatalf("want one skip-none result, got %+v", results)
}
if len(rr.calls) != 0 {
t.Fatalf("a unit with no trigger must not be acted on, got: %v", rr.calls)
}
}
// Idempotency: two consecutive passes over an idle trigger both succeed with the same action and no
// error (re-arming a fresh trigger is harmless — same end state).
func TestReassertNetworkAutomounts_Idempotent(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("systemd-escaped unit filename contains a backslash; exercised on the Linux build server")
}
unitDir := t.TempDir()
spec := NetworkMountSpec{Name: "media", Protocol: ProtocolNFS, Server: "10.0.0.5", Export: "/srv/media", MappedUID: 1000, MappedGID: 1000}
where, _ := installNetUnitFile(t, unitDir, spec)
ops, rr := netReassertOps(t, unitDir, []Mount{{MountPoint: where, FSType: "autofs"}})
first := ops.ReassertNetworkAutomounts(context.Background())
second := ops.ReassertNetworkAutomounts(context.Background())
if first[0].Action != NetReassertRearmed || second[0].Action != NetReassertRearmed {
t.Fatalf("both passes must re-arm: first=%+v second=%+v", first, second)
}
if first[0].Err != nil || second[0].Err != nil {
t.Fatalf("idempotent passes must not error: first=%v second=%v", first[0].Err, second[0].Err)
}
if len(rr.calls) != 4 {
t.Fatalf("two passes = 2×(stop+enable), got %d: %v", len(rr.calls), rr.calls)
}
}
// Zero configured network units → empty pass, zero commands.
func TestReassertNetworkAutomounts_NoUnitsNoOp(t *testing.T) {
ops, rr := netReassertOps(t, t.TempDir(), nil)
if results := ops.ReassertNetworkAutomounts(context.Background()); len(results) != 0 {
t.Fatalf("no units must yield no results, got %+v", results)
}
if len(rr.calls) != 0 {
t.Fatalf("no units must construct zero commands, got: %v", rr.calls)
}
}