slice 8C Phase A: agent disk endpoints + data-bearing classifier gate + mkfs (v0.12.0)

internal/storage: mkfs executor (Format, device-pinned, narrow FELHOM_FORMAT
sudoers) + data-bearing device inspection (InspectDevice/DeviceProbe via
blkid+lsblk; conservative — ambiguous=data-bearing). internal/localapi: /disks
(+ data-bearing flag), /disks/assign (EnsureMount), /disks/eject (Unmount +
dependent guests), /disks/format. SECURITY CENTERPIECE: the agent inspects the
device itself; data-bearing format -> ClassStorageWipe gate -> pending_signature
refused; the caller's claim is never trusted. Additive (no controller change yet).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-10 12:52:22 +02:00
parent 4d9e76e66a
commit c17cfde236
10 changed files with 1074 additions and 10 deletions
+197
View File
@@ -2,6 +2,7 @@ package storage
import (
"context"
"encoding/json"
"fmt"
"log/slog"
"os"
@@ -34,6 +35,55 @@ type HostOps interface {
// ThinPoolMetadata returns the lvmthin pool's metadata-used fraction (0..1) via lvs.
// ok=false when it cannot be read (the field stays null in the report).
ThinPoolMetadata(ctx context.Context, vg, pool string) (fraction float64, ok bool)
// InspectDevice probes a block device for data-bearing evidence (filesystem signature,
// partition table, partitions, mounted) — the AGENT-INTERNAL evidence the 8C classifier
// uses, NEVER the caller's claim. Conservative: a failed/ambiguous probe → DataBearing()
// true (fail-safe). This is the read that decides whether a format is benign or destructive.
InspectDevice(ctx context.Context, device string) (DeviceProbe, error)
// Format runs mkfs.<fstype> on a (validated) device. DESTRUCTIVE to whatever is on the
// device — the caller MUST have classified it non-data-bearing AND/OR routed it through the
// gate first; HostOps only performs an already-authorized format.
Format(ctx context.Context, device, fstype string) error
}
// DeviceProbe is the result of inspecting a block device for data-bearing evidence (8C). The
// agent decides data-bearing-ness from THIS (its own device read), never from the caller's claim.
type DeviceProbe struct {
Device string `json:"device"`
Probed bool `json:"probed"` // false = the probe failed/was ambiguous → treat as data-bearing
HasFilesystem bool `json:"has_filesystem"` // a filesystem signature (blkid TYPE / USAGE)
HasPartitionTable bool `json:"has_partition_table"` // a partition table (blkid PTTYPE)
HasPartitions bool `json:"has_partitions"` // child partitions present (lsblk)
Mounted bool `json:"mounted"` // currently mounted somewhere
FSType string `json:"fstype,omitempty"`
}
// DataBearing is the conservative verdict: any signature / partition table / partition / mount —
// OR a probe that did not complete cleanly — makes the device data-bearing. Only a device that
// probed cleanly AND shows none of those is considered blank (benign to format).
func (p DeviceProbe) DataBearing() bool {
if !p.Probed {
return true // fail-safe: never call an unprobed device blank
}
return p.HasFilesystem || p.HasPartitionTable || p.HasPartitions || p.Mounted
}
// Reason returns a short human string for why the device is data-bearing (for the UI/audit).
func (p DeviceProbe) Reason() string {
switch {
case !p.Probed:
return "device could not be reliably inspected"
case p.Mounted:
return "device is mounted"
case p.HasFilesystem:
return "device has a " + p.FSType + " filesystem"
case p.HasPartitionTable:
return "device has a partition table"
case p.HasPartitions:
return "device has partitions"
default:
return "device is blank"
}
}
// MountSpec describes a persistent by-UUID mount.
@@ -52,6 +102,10 @@ type Binaries struct {
Install string
Smartctl string
Lvs string
Blkid string // device signature probe (8C data-bearing detection)
Lsblk string // partition/mount topology (8C)
MkfsExt4 string // 8C format executor (ext4)
MkfsXfs string // 8C format executor (xfs)
}
func (b Binaries) withDefaults() Binaries {
@@ -67,6 +121,18 @@ func (b Binaries) withDefaults() Binaries {
if b.Lvs == "" {
b.Lvs = "/usr/sbin/lvs"
}
if b.Blkid == "" {
b.Blkid = "/usr/sbin/blkid"
}
if b.Lsblk == "" {
b.Lsblk = "/usr/bin/lsblk"
}
if b.MkfsExt4 == "" {
b.MkfsExt4 = "/usr/sbin/mkfs.ext4"
}
if b.MkfsXfs == "" {
b.MkfsXfs = "/usr/sbin/mkfs.xfs"
}
return b
}
@@ -210,6 +276,87 @@ func (h *SudoHostOps) ThinPoolMetadata(ctx context.Context, vg, pool string) (fl
return parseThinPoolMetadata(out)
}
// InspectDevice probes a device for data-bearing evidence (8C). It runs `blkid -p -o export`
// (the reliable signature probe) for filesystem/partition-table signatures and `lsblk -J` for
// child partitions + mount state. The verdict defaults to data-bearing on ANY read failure
// (Probed=false), so a compromised caller cannot get a data-bearing device declared blank.
func (h *SudoHostOps) InspectDevice(ctx context.Context, device string) (DeviceProbe, error) {
if err := ValidateBlockDevice(device); err != nil {
return DeviceProbe{Device: device}, err // Probed=false → DataBearing()=true
}
probe := DeviceProbe{Device: device}
// blkid -p -o export is the authoritative on-disk SIGNATURE probe. Its OUTPUT is the signal:
// any TYPE/PTTYPE/USAGE line is positive data-bearing evidence. Its exit code is NOT relied
// on (blkid exits 2 on a blank device) — output presence is what matters. A broken/empty
// blkid simply adds no positive evidence; lsblk (below) is the read-success authority.
bout, _, _ := h.runner.Run(ctx, h.bins.Blkid, "-p", "-o", "export", device)
for k, v := range parseBlkidExport(bout) {
switch k {
case "TYPE":
probe.HasFilesystem = true
probe.FSType = v
case "PTTYPE":
probe.HasPartitionTable = true
case "USAGE":
if v != "" {
probe.HasFilesystem = true // filesystem/raid/crypto member = data-bearing
}
}
}
// lsblk -J is the READ-SUCCESS authority + the partition/mount view. It exits 0 on any valid
// device (blank or not), so a clean parse means the agent reliably read the device. If lsblk
// fails, Probed stays false → DataBearing()=true (fail-safe — never call a device blank on a
// failed read).
lout, _, lerr := h.runner.Run(ctx, h.bins.Lsblk, "-J", "-o", "NAME,FSTYPE,PTTYPE,MOUNTPOINT", device)
if lerr == nil {
probe.Probed = true
hasChildren, mounted, fstype, pttype := parseLsblkDevice(lout)
if hasChildren {
probe.HasPartitions = true
}
if mounted {
probe.Mounted = true
}
if fstype != "" {
probe.HasFilesystem = true
if probe.FSType == "" {
probe.FSType = fstype
}
}
if pttype != "" {
probe.HasPartitionTable = true
}
}
return probe, nil
}
// Format runs mkfs.<fstype> on a validated device. The caller is responsible for authorization
// (8C: only after classifying the device non-data-bearing, or via a slice-10 operator signature).
func (h *SudoHostOps) Format(ctx context.Context, device, fstype string) error {
if err := ValidateBlockDevice(device); err != nil {
return err
}
if err := ValidateFSType(fstype); err != nil {
return err
}
switch fstype {
case "ext4":
if err := h.run(ctx, h.bins.MkfsExt4, "-F", device); err != nil {
return fmt.Errorf("storage: mkfs.ext4 %s: %w", device, err)
}
case "xfs":
if err := h.run(ctx, h.bins.MkfsXfs, "-f", device); err != nil {
return fmt.Errorf("storage: mkfs.xfs %s: %w", device, err)
}
default:
return fmt.Errorf("storage: unsupported fstype %q", fstype) // unreachable after Validate
}
h.logger.Info("storage: formatted device", "device", device, "fstype", fstype)
return nil
}
// run execs an allow-listed command with a fixed arg vector and wraps a nonzero exit.
func (h *SudoHostOps) run(ctx context.Context, name string, args ...string) error {
_, stderr, err := h.runner.Run(ctx, name, args...)
@@ -239,6 +386,48 @@ func trim(b []byte) string {
return s
}
// parseBlkidExport parses `blkid -p -o export` output (KEY=value lines) into a map.
func parseBlkidExport(out []byte) map[string]string {
m := map[string]string{}
for _, line := range strings.Split(string(out), "\n") {
line = strings.TrimSpace(line)
if i := strings.IndexByte(line, '='); i > 0 {
m[line[:i]] = line[i+1:]
}
}
return m
}
// lsblkDevice mirrors the `lsblk -J` device shape (only the fields we read).
type lsblkDevice struct {
Name string `json:"name"`
FSType string `json:"fstype"`
PTType string `json:"pttype"`
MountPoint string `json:"mountpoint"`
Children []lsblkDevice `json:"children"`
}
// parseLsblkDevice parses `lsblk -J -o NAME,FSTYPE,PTTYPE,MOUNTPOINT <device>` for the top device:
// whether it has child partitions, is mounted (itself or any child), and its fstype/pttype.
func parseLsblkDevice(out []byte) (hasChildren, mounted bool, fstype, pttype string) {
var doc struct {
BlockDevices []lsblkDevice `json:"blockdevices"`
}
if json.Unmarshal(out, &doc) != nil || len(doc.BlockDevices) == 0 {
return false, false, "", ""
}
d := doc.BlockDevices[0]
fstype, pttype = d.FSType, d.PTType
hasChildren = len(d.Children) > 0
mounted = d.MountPoint != ""
for _, c := range d.Children {
if c.MountPoint != "" {
mounted = true
}
}
return hasChildren, mounted, fstype, pttype
}
// NoopHostOps is the safe fallback when the privileged surface is unavailable or declined
// (a missing sudoers entry must degrade with a clear warning, not crash — slice notes). It
// reports SMART as UNKNOWN, no thin-pool metadata, and errors on any write (so a benign
@@ -257,3 +446,11 @@ func (n NoopHostOps) SMART(context.Context, string) (hub.SmartSummary, error) {
func (n NoopHostOps) ThinPoolMetadata(context.Context, string, string) (float64, bool) {
return 0, false
}
func (n NoopHostOps) InspectDevice(_ context.Context, device string) (DeviceProbe, error) {
// Probed=false → DataBearing()=true: with no privileged surface we MUST NOT call any device
// blank (fail-safe — a format would then be refused as destructive).
return DeviceProbe{Device: device}, nil
}
func (n NoopHostOps) Format(context.Context, string, string) error {
return fmt.Errorf("storage: privileged HostOps not configured; cannot format")
}
+185
View File
@@ -0,0 +1,185 @@
package storage
import (
"context"
"errors"
"strings"
"testing"
)
// scriptedRunner returns canned stdout/stderr/err per command name (last arg = device).
type scriptedRunner struct {
calls [][]string
outputs map[string][]byte // keyed by binary basename
errs map[string]error
}
func (r *scriptedRunner) Run(_ context.Context, name string, args ...string) ([]byte, []byte, error) {
r.calls = append(r.calls, append([]string{name}, args...))
base := name
if i := strings.LastIndexByte(name, '/'); i >= 0 {
base = name[i+1:]
}
return r.outputs[base], nil, r.errs[base]
}
func (r *scriptedRunner) ran(substr string) bool {
for _, c := range r.calls {
if strings.Contains(strings.Join(c, " "), substr) {
return true
}
}
return false
}
func newSudo(r *scriptedRunner) *SudoHostOps {
return NewSudoHostOps(SudoHostOpsConfig{Runner: r})
}
// ---- validators -------------------------------------------------------------------------
func TestValidateBlockDevice(t *testing.T) {
ok := []string{"/dev/sdb", "/dev/sdb1", "/dev/nvme0n1", "/dev/nvme0n1p2", "/dev/vdb3"}
for _, d := range ok {
if err := ValidateBlockDevice(d); err != nil {
t.Errorf("expected %q valid: %v", d, err)
}
}
bad := []string{"/dev/disk/by-uuid/x", "/dev/../etc/passwd", "/dev/sdb; rm -rf /", "/etc/shadow", "/dev/mapper/x", "sdb", ""}
for _, d := range bad {
if err := ValidateBlockDevice(d); err == nil {
t.Errorf("expected %q rejected", d)
}
}
}
func TestValidateFSType(t *testing.T) {
for _, f := range []string{"ext4", "xfs"} {
if err := ValidateFSType(f); err != nil {
t.Errorf("expected %q valid", f)
}
}
for _, f := range []string{"ntfs", "vfat", "ext4 ", "", "ext4;ls"} {
if err := ValidateFSType(f); err == nil {
t.Errorf("expected %q rejected", f)
}
}
}
// ---- InspectDevice (data-bearing detection) ---------------------------------------------
func TestInspect_Blank(t *testing.T) {
// blkid finds nothing (empty output); lsblk reads cleanly and shows a bare disk.
r := &scriptedRunner{
outputs: map[string][]byte{
"blkid": nil,
"lsblk": []byte(`{"blockdevices":[{"name":"sdb","fstype":null,"pttype":null,"mountpoint":null}]}`),
},
errs: map[string]error{"blkid": errors.New("exit status 2")}, // blkid exits non-zero on blank
}
p, err := newSudo(r).InspectDevice(context.Background(), "/dev/sdb")
if err != nil {
t.Fatal(err)
}
if !p.Probed {
t.Fatal("expected a clean probe (lsblk read cleanly)")
}
if p.DataBearing() {
t.Fatalf("blank device classified data-bearing: %+v", p)
}
}
func TestInspect_HasFilesystem(t *testing.T) {
r := &scriptedRunner{
outputs: map[string][]byte{
"blkid": []byte("DEVNAME=/dev/sdb\nTYPE=ext4\nUSAGE=filesystem\n"),
"lsblk": []byte(`{"blockdevices":[{"name":"sdb","fstype":"ext4","pttype":null,"mountpoint":null}]}`),
},
}
p, _ := newSudo(r).InspectDevice(context.Background(), "/dev/sdb")
if !p.DataBearing() || !p.HasFilesystem || p.FSType != "ext4" {
t.Fatalf("filesystem not detected: %+v", p)
}
}
func TestInspect_HasPartitionTable(t *testing.T) {
r := &scriptedRunner{
outputs: map[string][]byte{
"blkid": []byte("DEVNAME=/dev/sdb\nPTTYPE=gpt\n"),
"lsblk": []byte(`{"blockdevices":[{"name":"sdb","pttype":"gpt","children":[{"name":"sdb1","fstype":"ext4"}]}]}`),
},
}
p, _ := newSudo(r).InspectDevice(context.Background(), "/dev/sdb")
if !p.DataBearing() || !p.HasPartitionTable || !p.HasPartitions {
t.Fatalf("partition table/children not detected: %+v", p)
}
}
func TestInspect_Mounted(t *testing.T) {
r := &scriptedRunner{
outputs: map[string][]byte{
"blkid": []byte("TYPE=xfs\n"),
"lsblk": []byte(`{"blockdevices":[{"name":"sdb","fstype":"xfs","mountpoint":"/mnt/data"}]}`),
},
}
p, _ := newSudo(r).InspectDevice(context.Background(), "/dev/sdb")
if !p.Mounted || !p.DataBearing() {
t.Fatalf("mounted not detected: %+v", p)
}
}
// A probe that fails to read cleanly must be conservative (data-bearing).
func TestInspect_FailedProbe_FailSafe(t *testing.T) {
r := &scriptedRunner{
outputs: map[string][]byte{"blkid": nil, "lsblk": nil},
errs: map[string]error{"blkid": errors.New("blkid broke"), "lsblk": errors.New("lsblk broke")},
}
p, _ := newSudo(r).InspectDevice(context.Background(), "/dev/sdb")
if p.Probed {
t.Fatal("a broken probe must not be 'Probed'")
}
if !p.DataBearing() {
t.Fatal("a broken probe must be treated as data-bearing (fail-safe)")
}
}
func TestInspect_RejectsBadDevice(t *testing.T) {
if _, err := newSudo(&scriptedRunner{}).InspectDevice(context.Background(), "/dev/../etc"); err == nil {
t.Fatal("expected a bad device to be rejected before any exec")
}
}
// ---- Format (mkfs) ----------------------------------------------------------------------
func TestFormat_Ext4(t *testing.T) {
r := &scriptedRunner{}
if err := newSudo(r).Format(context.Background(), "/dev/sdb", "ext4"); err != nil {
t.Fatal(err)
}
if !r.ran("mkfs.ext4 -F /dev/sdb") {
t.Fatalf("mkfs.ext4 not invoked correctly: %v", r.calls)
}
}
func TestFormat_Xfs(t *testing.T) {
r := &scriptedRunner{}
if err := newSudo(r).Format(context.Background(), "/dev/nvme0n1p1", "xfs"); err != nil {
t.Fatal(err)
}
if !r.ran("mkfs.xfs -f /dev/nvme0n1p1") {
t.Fatalf("mkfs.xfs not invoked correctly: %v", r.calls)
}
}
func TestFormat_RejectsBadArgs(t *testing.T) {
r := &scriptedRunner{}
if err := newSudo(r).Format(context.Background(), "/dev/disk/by-uuid/x", "ext4"); err == nil {
t.Fatal("expected bad device rejected")
}
if err := newSudo(r).Format(context.Background(), "/dev/sdb", "ntfs"); err == nil {
t.Fatal("expected bad fstype rejected")
}
if len(r.calls) != 0 {
t.Fatalf("mkfs ran despite invalid input: %v", r.calls)
}
}
+4
View File
@@ -177,6 +177,10 @@ func (f *fakeHostOps) ThinPoolMetadata(_ context.Context, vg, pool string) (floa
v, ok := f.metaByPool[vg+"/"+pool]
return v, ok
}
func (f *fakeHostOps) InspectDevice(_ context.Context, device string) (DeviceProbe, error) {
return DeviceProbe{Device: device, Probed: true}, nil
}
func (f *fakeHostOps) Format(context.Context, string, string) error { return nil }
func TestObserve_EnrichesSMARTAndThinPoolMetadata(t *testing.T) {
api := &fakeStorageAPI{
+29
View File
@@ -25,6 +25,16 @@ var (
// smartctl is run against. Anything else is refused.
reSMARTDevice = regexp.MustCompile(`^/dev/(sd[a-z]+|nvme[0-9]+n[0-9]+|hd[a-z]+|vd[a-z]+)$`)
// Block device for inspection / mkfs: a raw disk OR a partition under /dev. Like the SMART
// whitelist but also allows the trailing partition number (sda1, nvme0n1p2, vdb3). No
// by-* symlinks, no device-mapper, no traversal. This is the mkfs/inspect target — it is
// validated AND the agent device-inspects it before any destructive decision (8C).
reBlockDevice = regexp.MustCompile(`^/dev/(sd[a-z]+[0-9]*|nvme[0-9]+n[0-9]+(p[0-9]+)?|hd[a-z]+[0-9]*|vd[a-z]+[0-9]*)$`)
// Filesystem types the agent will mkfs. Deliberately tiny — the sudoers mkfs entries are
// per-fstype binaries (mkfs.ext4 / mkfs.xfs), so this set MUST match those entries.
reFSType = regexp.MustCompile(`^(ext4|xfs)$`)
// LVM VG / pool names: LVM permits [A-Za-z0-9._+-]; we forbid leading '-' (would look
// like a flag) and cap the length.
reLVMName = regexp.MustCompile(`^[A-Za-z0-9_+.][A-Za-z0-9_+.-]*$`)
@@ -106,6 +116,25 @@ func ValidateSMARTDevice(device string) error {
return nil
}
// ValidateBlockDevice accepts only a raw disk or partition path under /dev (the mkfs / inspect
// target). The same strict-whitelist discipline as ValidateSMARTDevice: no symlinks, no
// device-mapper, no traversal — refused before any command is built.
func ValidateBlockDevice(device string) error {
if !reBlockDevice.MatchString(device) {
return fmt.Errorf("storage: refusing to operate on non-whitelisted block device %q", device)
}
return nil
}
// ValidateFSType accepts only a filesystem type the agent is configured to mkfs (ext4|xfs). The
// set MUST match the per-fstype sudoers entries.
func ValidateFSType(fstype string) error {
if !reFSType.MatchString(fstype) {
return fmt.Errorf("storage: unsupported filesystem type %q (want ext4|xfs)", fstype)
}
return nil
}
// ValidateLVMName accepts an LVM VG or LV (pool) name.
func ValidateLVMName(name string) error {
if name == "" {