R-25 (agent half): the format answer carries the NEW filesystem's UUID, bound to the durable id

After mkfs the agent re-resolves the bound durable id, requires it to name the
device it just formatted, reads the superblock back (blkid -p, requested fstype)
and returns fs_uuid in POST /disks/format and GET /disks/format/status. Anything
unverified returns "" — never a path-resolved guess. The controller half
(mount fs_uuid instead of re-resolving the UUID from the /dev path) is owed
in felhom-controller.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 21:15:01 +02:00
parent 64f704d0f7
commit 769c4c3cf2
7 changed files with 172 additions and 9 deletions
+1 -1
View File
@@ -66,7 +66,7 @@
|---|---|---|---|---|
| `IntentStore` (`Get/SetEnrolled/SetEjected/SetDecommissioned/OnAbsent`) | internal/storage/intent.go | `OpenIntentStore(path)` | drive intent (4-state self-heal) | Keyed by durable-id only; `OnAbsent` is the ONLY ejected→enrolled path; refuses empty ids |
| `GuestBindStore` (`Record/Remove/Guests`) | internal/localapi/guestbindstore.go | `OpenGuestBindStore(path)` | per-guest enrolled binds (F9 re-assert) | Same tmp+rename 0600 pattern as IntentStore |
| `FormatJobStore` + `startFormatDetached` + `RecoverFormatJob` | internal/localapi/formatjob.go | `startFormatDetached(device, durableID, fstype, blank) <-chan error` | detached, restart-surviving mkfs (F20-BUG3) | Runs off `s.baseCtx` (60-min bound) so a request deadline can't SIGKILL mkfs; recovery re-resolves by durable id; blank jobs re-check STILL-blank |
| `FormatJobStore` + `startFormatDetached` + `RecoverFormatJob` | internal/localapi/formatjob.go | `startFormatDetached(device, durableID, fstype, blank) (*formatJob, <-chan error)` | detached, restart-surviving mkfs (F20-BUG3) | Runs off `s.baseCtx` (60-min bound) so a request deadline can't SIGKILL mkfs; recovery re-resolves by durable id; blank jobs re-check STILL-blank; on success `job.FSUUID` = the new filesystem's UUID, read back only when the durable id still resolves to the formatted device (R-25) — read it only after `done` delivers |
| `TokenStore.Mint` / `Lookup` | internal/localapi/tokenstore.go | `Mint(vmid) (plaintext, error)` | per-guest local-API tokens | Only the SHA-256 hash persists (fsync'd append log); constant-time compare on lookup; plaintext returned exactly once. Lookup stats the file on EVERY call and reloads BEFORE answering when the append-only log grew (R-269; was reload-on-miss only, v0.63.0 B3, which let a token rotated out by another process keep authorizing as a map hit): the one-shot provisioner mints into the same file the daemon indexes — cross-process coherence both ways without a restart; unchanged size = no re-read. Pinned by `TestTokenStore_RotatedOutTokenRejectedFirst` |
| `FileNonceStore.SeenOrRecord` | internal/authz/noncestore.go | `SeenOrRecord(nonce, exp) bool` | durable anti-replay | fsync'd before returning false; prune only after exp |
| `Journal` (`Append/Latest/InFlight/AlreadyApplied`) | internal/reconcile/journal.go | `OpenJournal(path)` | op journal + idempotency + crash recovery | `Recover` consumes `InFlight()`; scratch entries special-cased |
+10 -5
View File
@@ -768,6 +768,11 @@ type FormatResponse struct {
// signature — the customer authorizes the wipe of their own data drive.
NeedsConfirmation bool `json:"needs_confirmation,omitempty"`
DurableID string `json:"durable_id,omitempty"` // the durable id to confirm against (user-data)
// FSUUID (on Formatted) is the UUID of the filesystem the agent just made, verified against the
// bound durable id after mkfs (R-25). The caller mounts THIS — re-resolving a UUID from the /dev
// path later can name another disk if /dev re-enumerated. "" = not verified: the caller must not
// substitute a path-resolved guess silently.
FSUUID string `json:"fs_uuid,omitempty"`
// PendingOp is set on a SYSTEM/BACKUP data-bearing refusal — the exact op the operator must sign.
PendingOp *PendingOp `json:"pending_op,omitempty"`
}
@@ -815,7 +820,7 @@ func (s *Server) handleDiskFormatStatus(w http.ResponseWriter, r *http.Request,
writeOK(w, map[string]any{
"vmid": vmid, "phase": job.Phase, "device": job.Device, "fstype": job.FSType,
"durable_id": job.DurableID, "error": job.Error, "started_at": job.StartedAt, "updated_at": job.UpdatedAt,
"job_id": job.JobID,
"job_id": job.JobID, "fs_uuid": job.FSUUID, // R-25: "" until done + verified
})
}
@@ -878,7 +883,7 @@ func (s *Server) handleDiskFormat(w http.ResponseWriter, r *http.Request, vmid i
"format refused (device may have changed since inspection): "+rerr.Error())
return
}
done := s.startFormatDetached(device, blankDurable, req.FSType, true)
job, done := s.startFormatDetached(device, blankDurable, req.FSType, true)
if err := s.awaitFormat(r.Context(), done, vmid, device); err != nil {
if err == errFormatClientGone {
return // client gone; mkfs continues detached + the job record records the outcome
@@ -887,7 +892,7 @@ func (s *Server) handleDiskFormat(w http.ResponseWriter, r *http.Request, vmid i
writeErr(w, http.StatusBadGateway, "format failed: "+err.Error())
return
}
writeOK(w, FormatResponse{VMID: vmid, Device: device, Formatted: true, DataBearing: false, DurableID: blankDurable, Reason: "blank device formatted " + req.FSType})
writeOK(w, FormatResponse{VMID: vmid, Device: device, Formatted: true, DataBearing: false, DurableID: blankDurable, FSUUID: job.FSUUID, Reason: "blank device formatted " + req.FSType})
return
}
@@ -924,7 +929,7 @@ func (s *Server) handleDiskFormat(w http.ResponseWriter, r *http.Request, vmid i
// F20-BUG3: run the destructive mkfs DETACHED off s.baseCtx (bound durable id recorded for
// restart-recovery), so a request/client deadline can never SIGKILL it mid-write and corrupt the
// disk. We still wait to return the synchronous result (backward-compatible with the controller).
done := s.startFormatDetached(device, deviceDurable, req.FSType, false)
job, done := s.startFormatDetached(device, deviceDurable, req.FSType, false)
if err := s.awaitFormat(r.Context(), done, vmid, device); err != nil {
if err == errFormatClientGone {
return // client gone; the wipe continues detached + survives a restart via the job record
@@ -936,7 +941,7 @@ func (s *Server) handleDiskFormat(w http.ResponseWriter, r *http.Request, vmid i
s.logger.Warn("local-api: USER-DATA data-bearing format — CUSTOMER CONFIRMED (no operator signature)",
"vmid", vmid, "device", device, "durable_id", deviceDurable, "fstype", req.FSType, "why", probe.Reason())
writeOK(w, FormatResponse{VMID: vmid, Device: device, Formatted: true, DataBearing: true,
Role: string(role), DurableID: deviceDurable, Reason: "customer-confirmed wipe (" + probe.Reason() + ")"})
Role: string(role), DurableID: deviceDurable, FSUUID: job.FSUUID, Reason: "customer-confirmed wipe (" + probe.Reason() + ")"})
return
}
+6
View File
@@ -28,6 +28,7 @@ type fakeDiskOps struct {
unmountCalls []string
candidates []storage.CandidateDisk // returned by ListCandidateDisks
candErr error
afterFormat *storage.DeviceProbe // R-25: when set, InspectDevice returns it once a format ran
}
func (f *fakeDiskOps) ListCandidateDisks(_ context.Context) ([]storage.CandidateDisk, error) {
@@ -35,7 +36,12 @@ func (f *fakeDiskOps) ListCandidateDisks(_ context.Context) ([]storage.Candidate
}
func (f *fakeDiskOps) InspectDevice(_ context.Context, device string) (storage.DeviceProbe, error) {
f.mu.Lock()
p := f.probe
if f.afterFormat != nil && len(f.formatCalls) > 0 {
p = *f.afterFormat
}
f.mu.Unlock()
p.Device = device
return p, f.inspectErr
}
+96
View File
@@ -0,0 +1,96 @@
package localapi
import (
"context"
"encoding/json"
"net/http"
"sync"
"testing"
"gitea.dooplex.hu/admin/felhom-agent/internal/storage"
)
// R-25: the format answer carries the UUID of the filesystem the agent JUST made, verified against the
// bound durable id after mkfs, so the controller mounts that filesystem rather than whatever the /dev
// path resolves to a few requests later. The consequence asserted: the UUID on the wire (and in the
// polled job record) is the new superblock's — and is EMPTY whenever the binding cannot be re-proved.
const newFSUUID = "0fc63daf-8483-4772-8e79-3d69d8477de4"
func confirmedFormat(t *testing.T, d *fakeDiskOps, srv *Server, fj *FormatJobStore) (string, *formatJob) {
t.Helper()
w := do(t, srv.Handler(), "POST", "/disks/format", "A", `{"device":"/dev/sdb1","fstype":"ext4","confirmed":true,"durable_id":"byid:wwn-/dev/sdb1"}`)
if w.Code != http.StatusOK {
t.Fatalf("confirmed format: %d (%s)", w.Code, w.Body.String())
}
var resp struct {
Data FormatResponse `json:"data"`
}
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("decode: %v (%s)", err, w.Body.String())
}
if !resp.Data.Formatted {
t.Fatalf("not formatted: %s", w.Body.String())
}
return resp.Data.FSUUID, waitFormatPhase(t, fj, formatPhaseDone)
}
func confirmedGate() *fakeGate {
return &fakeGate{decision: WipeDecision{Allowed: true, Tier: "customer_confirmable", Reason: "customer_confirmed"}}
}
func TestFormat_ReportsNewFilesystemUUID(t *testing.T) {
d := &fakeDiskOps{probe: deviceProbeDataBearing(),
afterFormat: &storage.DeviceProbe{Probed: true, HasFilesystem: true, FSType: "ext4", FSUUID: newFSUUID}}
fj := tempFormatStore(t)
srv := formatServer(t, d, confirmedGate(), fj)
got, job := confirmedFormat(t, d, srv, fj)
if got != newFSUUID {
t.Fatalf("response fs_uuid = %q, want the new filesystem's %q", got, newFSUUID)
}
if job.FSUUID != newFSUUID {
t.Fatalf("job record fs_uuid = %q, want %q (the polled status path)", job.FSUUID, newFSUUID)
}
}
// The node moved between mkfs and the read-back: the bound durable id now resolves elsewhere. The
// UUID must NOT be reported — reading it would name the other disk's filesystem.
func TestFormat_FSUUIDWithheldWhenDurableIDMoved(t *testing.T) {
d := &fakeDiskOps{probe: deviceProbeDataBearing(),
afterFormat: &storage.DeviceProbe{Probed: true, HasFilesystem: true, FSType: "ext4", FSUUID: newFSUUID}}
fj := tempFormatStore(t)
srv := formatServer(t, d, confirmedGate(), fj)
var mu sync.Mutex
calls := 0
srv.reresolveWipe = func(_ context.Context, _ string) (string, error) {
mu.Lock()
defer mu.Unlock()
calls++
if calls == 1 {
return "/dev/sdb", nil // the pre-mkfs anti-retarget re-resolve
}
return "/dev/sdc", nil // after mkfs: the durable id now names another node
}
got, job := confirmedFormat(t, d, srv, fj)
if got != "" || job.FSUUID != "" {
t.Fatalf("fs_uuid reported after the durable id moved (response %q, job %q) — must be empty", got, job.FSUUID)
}
if calls < 2 {
t.Fatalf("the post-mkfs re-resolve never ran (calls=%d)", calls)
}
}
// The superblock did not read back as the requested filesystem → not verified → empty.
func TestFormat_FSUUIDWithheldOnFSTypeMismatch(t *testing.T) {
d := &fakeDiskOps{probe: deviceProbeDataBearing(),
afterFormat: &storage.DeviceProbe{Probed: true, HasFilesystem: true, FSType: "xfs", FSUUID: newFSUUID}}
fj := tempFormatStore(t)
srv := formatServer(t, d, confirmedGate(), fj)
got, job := confirmedFormat(t, d, srv, fj)
if got != "" || job.FSUUID != "" {
t.Fatalf("fs_uuid reported for a superblock of the wrong type (response %q, job %q)", got, job.FSUUID)
}
}
+39 -3
View File
@@ -22,6 +22,10 @@ type formatJob struct {
Blank bool `json:"blank,omitempty"` // audit D3: blank (benign) format — recovery re-checks STILL-blank, not data-bearing
Phase string `json:"phase"` // running | done | failed
Error string `json:"error,omitempty"`
// FSUUID is the filesystem UUID of the NEW filesystem, read by the agent right after mkfs on the
// device the bound durable id still resolves to (R-25). "" = not verified (the controller must not
// read that as a UUID). Set only on phase done.
FSUUID string `json:"fs_uuid,omitempty"`
StartedAt string `json:"started_at"`
UpdatedAt string `json:"updated_at"`
}
@@ -96,7 +100,10 @@ func (s *FormatJobStore) save(j *formatJob) error {
// runs to completion and records the outcome. device is the ALREADY anti-retarget-resolved device; the
// record carries durableID so a restart can re-resolve + re-run. blank marks a benign (blank-device)
// format, so restart recovery re-checks STILL-blank rather than data-bearing (audit D3).
func (s *Server) startFormatDetached(device, durableID, fstype string, blank bool) <-chan error {
//
// The returned job may be read (FSUUID) only AFTER a value arrives on done — the goroutine writes it
// before the send, which is the happens-before edge.
func (s *Server) startFormatDetached(device, durableID, fstype string, blank bool) (*formatJob, <-chan error) {
base := s.baseCtx
if base == nil {
base = context.Background()
@@ -116,10 +123,39 @@ func (s *Server) startFormatDetached(device, durableID, fstype string, blank boo
ctx, cancel := context.WithTimeout(base, 60*time.Minute)
defer cancel()
err := s.disks.Format(ctx, device, fstype)
if err == nil {
job.FSUUID = s.formattedFSUUID(ctx, device, durableID, fstype)
}
s.finishFormatJob(job, err)
done <- err
}()
return done
return job, done
}
// formattedFSUUID reads the UUID of the filesystem the agent has JUST made (R-25). The caller used to
// re-resolve the UUID from the mutable /dev path afterwards, over separate requests — a re-enumeration
// in that window could hand it ANOTHER disk's filesystem to mount. Here the bound durable id must still
// resolve to the very device that was formatted (and re-derive to the same id), and the superblock must
// carry the fstype that was asked for; anything else returns "" (not verified), never a guess.
func (s *Server) formattedFSUUID(ctx context.Context, device, durableID, fstype string) string {
if durableID == "" || s.reresolveWipe == nil {
return ""
}
// The device now holds a filesystem, so the data-bearing anti-retarget re-resolve is the right one.
now, err := s.reresolveWipe(ctx, durableID)
if err != nil || now != device {
s.logger.Warn("format: new filesystem UUID NOT reported — bound durable id no longer resolves to the formatted device",
"device", device, "durable_id", durableID, "resolves_to", now, "err", err)
return ""
}
probe, err := s.disks.InspectDevice(ctx, device)
if err != nil || !probe.Probed || probe.FSType != fstype || probe.FSUUID == "" {
s.logger.Warn("format: new filesystem UUID NOT reported — superblock did not read back as the requested filesystem",
"device", device, "want_fstype", fstype, "got_fstype", probe.FSType, "has_uuid", probe.FSUUID != "", "err", err)
return ""
}
s.logger.Info("format: new filesystem bound to its durable id", "device", device, "durable_id", durableID, "fs_uuid", probe.FSUUID)
return probe.FSUUID
}
// finishFormatJob updates the persisted record to done/failed.
@@ -172,7 +208,7 @@ func (s *Server) RecoverFormatJob(ctx context.Context) {
return
}
s.logger.Warn("format-job recover: re-running interrupted format detached", "durable_id", job.DurableID, "device", device, "fstype", job.FSType, "blank", job.Blank)
_ = s.startFormatDetached(device, job.DurableID, job.FSType, job.Blank) // detached; updates the record on completion
_, _ = s.startFormatDetached(device, job.DurableID, job.FSType, job.Blank) // detached; updates the record on completion
}
// nowFn returns the server clock (testable), defaulting to time.Now.
+6
View File
@@ -57,6 +57,10 @@ type DeviceProbe struct {
HasPartitions bool `json:"has_partitions"` // child partitions present (lsblk)
Mounted bool `json:"mounted"` // currently mounted somewhere
FSType string `json:"fstype,omitempty"`
// FSUUID is the filesystem UUID blkid read from the on-disk superblock (`blkid -p`, no cache),
// "" when there is none. R-25: the format path reports it so the caller mounts the filesystem the
// agent just made, not whatever a /dev path resolves to later.
FSUUID string `json:"fs_uuid,omitempty"`
}
// DataBearing is the conservative verdict: any signature / partition table / partition / mount —
@@ -423,6 +427,8 @@ func (h *SudoHostOps) InspectDevice(ctx context.Context, device string) (DeviceP
probe.FSType = v
case "PTTYPE":
probe.HasPartitionTable = true
case "UUID":
probe.FSUUID = v
case "USAGE":
if v != "" {
probe.HasFilesystem = true // filesystem/raid/crypto member = data-bearing
+14
View File
@@ -208,3 +208,17 @@ func TestFormat_RejectsBadArgs(t *testing.T) {
t.Fatalf("mkfs ran despite invalid input: %v", r.calls)
}
}
// R-25: the probe carries the superblock's filesystem UUID so the format path can report the new one.
func TestInspect_ReadsFilesystemUUID(t *testing.T) {
r := &scriptedRunner{
outputs: map[string][]byte{
"blkid": []byte("DEVNAME=/dev/sdb\nUUID=0fc63daf-8483-4772-8e79-3d69d8477de4\nTYPE=ext4\nUSAGE=filesystem\n"),
"lsblk": []byte(`{"blockdevices":[{"name":"sdb","fstype":"ext4","pttype":null,"mountpoint":null}]}`),
},
}
p, _ := newSudo(r).InspectDevice(context.Background(), "/dev/sdb")
if p.FSUUID != "0fc63daf-8483-4772-8e79-3d69d8477de4" {
t.Fatalf("FSUUID = %q, want the blkid UUID", p.FSUUID)
}
}