R-25 (agent half): the format answer carries the NEW filesystem's UUID, bound to the durable id

After mkfs the agent re-resolves the bound durable id, requires it to name the
device it just formatted, reads the superblock back (blkid -p, requested fstype)
and returns fs_uuid in POST /disks/format and GET /disks/format/status. Anything
unverified returns "" — never a path-resolved guess. The controller half
(mount fs_uuid instead of re-resolving the UUID from the /dev path) is owed
in felhom-controller.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 21:15:01 +02:00
parent 64f704d0f7
commit 769c4c3cf2
7 changed files with 172 additions and 9 deletions
+39 -3
View File
@@ -22,6 +22,10 @@ type formatJob struct {
Blank bool `json:"blank,omitempty"` // audit D3: blank (benign) format — recovery re-checks STILL-blank, not data-bearing
Phase string `json:"phase"` // running | done | failed
Error string `json:"error,omitempty"`
// FSUUID is the filesystem UUID of the NEW filesystem, read by the agent right after mkfs on the
// device the bound durable id still resolves to (R-25). "" = not verified (the controller must not
// read that as a UUID). Set only on phase done.
FSUUID string `json:"fs_uuid,omitempty"`
StartedAt string `json:"started_at"`
UpdatedAt string `json:"updated_at"`
}
@@ -96,7 +100,10 @@ func (s *FormatJobStore) save(j *formatJob) error {
// runs to completion and records the outcome. device is the ALREADY anti-retarget-resolved device; the
// record carries durableID so a restart can re-resolve + re-run. blank marks a benign (blank-device)
// format, so restart recovery re-checks STILL-blank rather than data-bearing (audit D3).
func (s *Server) startFormatDetached(device, durableID, fstype string, blank bool) <-chan error {
//
// The returned job may be read (FSUUID) only AFTER a value arrives on done — the goroutine writes it
// before the send, which is the happens-before edge.
func (s *Server) startFormatDetached(device, durableID, fstype string, blank bool) (*formatJob, <-chan error) {
base := s.baseCtx
if base == nil {
base = context.Background()
@@ -116,10 +123,39 @@ func (s *Server) startFormatDetached(device, durableID, fstype string, blank boo
ctx, cancel := context.WithTimeout(base, 60*time.Minute)
defer cancel()
err := s.disks.Format(ctx, device, fstype)
if err == nil {
job.FSUUID = s.formattedFSUUID(ctx, device, durableID, fstype)
}
s.finishFormatJob(job, err)
done <- err
}()
return done
return job, done
}
// formattedFSUUID reads the UUID of the filesystem the agent has JUST made (R-25). The caller used to
// re-resolve the UUID from the mutable /dev path afterwards, over separate requests — a re-enumeration
// in that window could hand it ANOTHER disk's filesystem to mount. Here the bound durable id must still
// resolve to the very device that was formatted (and re-derive to the same id), and the superblock must
// carry the fstype that was asked for; anything else returns "" (not verified), never a guess.
func (s *Server) formattedFSUUID(ctx context.Context, device, durableID, fstype string) string {
if durableID == "" || s.reresolveWipe == nil {
return ""
}
// The device now holds a filesystem, so the data-bearing anti-retarget re-resolve is the right one.
now, err := s.reresolveWipe(ctx, durableID)
if err != nil || now != device {
s.logger.Warn("format: new filesystem UUID NOT reported — bound durable id no longer resolves to the formatted device",
"device", device, "durable_id", durableID, "resolves_to", now, "err", err)
return ""
}
probe, err := s.disks.InspectDevice(ctx, device)
if err != nil || !probe.Probed || probe.FSType != fstype || probe.FSUUID == "" {
s.logger.Warn("format: new filesystem UUID NOT reported — superblock did not read back as the requested filesystem",
"device", device, "want_fstype", fstype, "got_fstype", probe.FSType, "has_uuid", probe.FSUUID != "", "err", err)
return ""
}
s.logger.Info("format: new filesystem bound to its durable id", "device", device, "durable_id", durableID, "fs_uuid", probe.FSUUID)
return probe.FSUUID
}
// finishFormatJob updates the persisted record to done/failed.
@@ -172,7 +208,7 @@ func (s *Server) RecoverFormatJob(ctx context.Context) {
return
}
s.logger.Warn("format-job recover: re-running interrupted format detached", "durable_id", job.DurableID, "device", device, "fstype", job.FSType, "blank", job.Blank)
_ = s.startFormatDetached(device, job.DurableID, job.FSType, job.Blank) // detached; updates the record on completion
_, _ = s.startFormatDetached(device, job.DurableID, job.FSType, job.Blank) // detached; updates the record on completion
}
// nowFn returns the server clock (testable), defaulting to time.Now.