ee71abd1d4
Operator ruling 09 §3 decision 139. One exact sudoers rule FELHOM_FSTRIM
(`/usr/sbin/pct ^fstrim [0-9]+$`) + manifest entry guest-fstrim; new
internal/fstrim job: due Wednesday from 10:00 host-local, starts only
10:00-20:59, holds backup.InFlight (busy -> deferred to the next hourly
tick), failed trim retried at most 3x per week, bytes parsed from the
"(N bytes) trimmed" lines, last result per guest persisted in
<state_dir>/guest-disk-trim.json and reported as guest_disk_trim.
Config opt-out: "disk_trim": {"disable": true}.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
347 lines
12 KiB
Go
347 lines
12 KiB
Go
// Package fstrim is the weekly guest disk trim (R-444, operator ruling `09` §3 decision 139).
|
||
//
|
||
// Why: a thin pool only ever grows from blocks a guest has already FREED — `fstrim` inside the unprivileged container
|
||
// is refused (FITRIM: Operation not permitted), and nothing else on the box gives the blocks back. A full thin pool
|
||
// takes every guest on the host read-only, so the pool can reach 100 % from deleted data alone. Measured on demo-hp
|
||
// 2026-10-06 09:14Z: `pct fstrim 9201` rc 0 in 24.4 s, pool 65.53 % -> 33.40 %, 18/18 app probes 200, max 1.1 s
|
||
// (audits/ten-answers-2026-10-06/r444-measure.txt).
|
||
//
|
||
// The rule, each part pinned by a test in fstrim_test.go:
|
||
// - Weekly: a guest is DUE from Wednesday 10:00 local until it has been trimmed once since then (a box that was off
|
||
// on Wednesday catches up at its next eligible hour).
|
||
// - Daytime only: a trim starts only between 10:00 and 20:59 local — never in the night window (01:00–06:59) where
|
||
// the backups and the restore-tests run (TestEligibleHourNeverInTheNight).
|
||
// - Never beside a backup, a restore-test or another heavy operation: the pass holds the host-wide one-heavy-op gate
|
||
// (backup.InFlight) for its whole run; a busy gate DEFERS the pass to the next hourly tick.
|
||
// - A failed trim is retried at the next eligible hour, at most MaxAttemptsPerWeek times in one week.
|
||
// - The last result per guest (time, bytes, ok/fail) is persisted, so a restart neither loses it nor re-trims.
|
||
//
|
||
// The command is the ONE exact sudoers shape `pct fstrim <vmid>` (FELHOM_FSTRIM). Only guests from the pool-verified
|
||
// source (ListLXC ∩ the felhom pool, audit A1) and only RUNNING ones are trimmed.
|
||
package fstrim
|
||
|
||
import (
|
||
"context"
|
||
"encoding/json"
|
||
"fmt"
|
||
"log/slog"
|
||
"os"
|
||
"path/filepath"
|
||
"regexp"
|
||
"sort"
|
||
"strconv"
|
||
"strings"
|
||
"sync"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
||
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
|
||
)
|
||
|
||
// Schedule. The weekday/hours are fixed on purpose (one sentence the operator can read on the System page).
|
||
const (
|
||
Weekday = time.Wednesday
|
||
StartHour = 10 // first eligible local hour (inclusive)
|
||
EndHour = 21 // first NOT-eligible local hour (exclusive): last start is 20:59
|
||
MaxAttemptsPerWeek = 3
|
||
// TickInterval is how often the job looks; a deferred or failed pass is therefore retried the next hour.
|
||
TickInterval = time.Hour
|
||
// FirstTickDelay lets the agent settle after a start before the first look.
|
||
FirstTickDelay = 5 * time.Minute
|
||
// PerGuestTimeout bounds one `pct fstrim` (measured 24.4 s for 84 GiB).
|
||
PerGuestTimeout = 30 * time.Minute
|
||
)
|
||
|
||
// ScheduleText is the human description carried on the host report.
|
||
const ScheduleText = "weekly, due Wednesday from 10:00 host-local time; starts only 10:00-20:59; never beside a backup or restore-test"
|
||
|
||
// Runner runs a host command (proxmox.ExecRunner in production, through `sudo -n`).
|
||
type Runner interface {
|
||
Run(ctx context.Context, name string, args ...string) (stdout, stderr []byte, err error)
|
||
}
|
||
|
||
// GuestSource yields the guests this agent OWNS (the pool-verified source, never a bare ListLXC).
|
||
type GuestSource interface {
|
||
Guests(ctx context.Context) ([]proxmox.Guest, error)
|
||
}
|
||
|
||
// Gate is the host-wide one-heavy-operation gate (*backup.InFlight).
|
||
type Gate interface {
|
||
TryAcquire(what string) (release func(), busy string, ok bool)
|
||
}
|
||
|
||
// GateName is what the gate reports as busy while a trim runs.
|
||
const GateName = "guest-fstrim"
|
||
|
||
// Record is one guest's last trim attempt, as persisted.
|
||
type Record struct {
|
||
LastAttemptAt time.Time `json:"last_attempt_at"`
|
||
OK bool `json:"ok"`
|
||
BytesTrimmed int64 `json:"bytes_trimmed"`
|
||
Mounts int `json:"mounts"`
|
||
DurationSeconds float64 `json:"duration_seconds"`
|
||
LastOKAt time.Time `json:"last_ok_at,omitempty"`
|
||
Error string `json:"error,omitempty"`
|
||
// Attempts counts the attempts since the current week's due time (reset by the first attempt of a new week).
|
||
Attempts int `json:"attempts"`
|
||
}
|
||
|
||
// Trimmer is the weekly job.
|
||
type Trimmer struct {
|
||
runner Runner
|
||
guests GuestSource
|
||
gate Gate
|
||
statePath string
|
||
logger *slog.Logger
|
||
loc *time.Location
|
||
now func() time.Time
|
||
|
||
mu sync.Mutex
|
||
records map[int]Record
|
||
}
|
||
|
||
// New builds the job and loads the persisted state. A missing state file is an empty state; a corrupt one is logged
|
||
// and treated as empty (the cost is one extra trim, never a missed one).
|
||
func New(runner Runner, guests GuestSource, gate Gate, statePath string, logger *slog.Logger) *Trimmer {
|
||
if logger == nil {
|
||
logger = slog.Default()
|
||
}
|
||
t := &Trimmer{runner: runner, guests: guests, gate: gate, statePath: statePath, logger: logger,
|
||
loc: time.Local, now: time.Now, records: map[int]Record{}}
|
||
t.load()
|
||
return t
|
||
}
|
||
|
||
func (t *Trimmer) load() {
|
||
data, err := os.ReadFile(t.statePath)
|
||
if err != nil {
|
||
if !os.IsNotExist(err) {
|
||
t.logger.Warn("fstrim: state read failed — starting empty", "path", t.statePath, "err", err)
|
||
}
|
||
return
|
||
}
|
||
var raw map[string]Record
|
||
if err := json.Unmarshal(data, &raw); err != nil {
|
||
t.logger.Warn("fstrim: state file corrupt — starting empty", "path", t.statePath, "err", err)
|
||
return
|
||
}
|
||
for k, r := range raw {
|
||
if id, err := strconv.Atoi(k); err == nil && id > 0 {
|
||
t.records[id] = r
|
||
}
|
||
}
|
||
}
|
||
|
||
func (t *Trimmer) saveLocked() error {
|
||
raw := make(map[string]Record, len(t.records))
|
||
for id, r := range t.records {
|
||
raw[strconv.Itoa(id)] = r
|
||
}
|
||
data, err := json.MarshalIndent(raw, "", " ")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := os.MkdirAll(filepath.Dir(t.statePath), 0o755); err != nil {
|
||
return err
|
||
}
|
||
tmp := t.statePath + ".tmp"
|
||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||
os.Remove(tmp)
|
||
return err
|
||
}
|
||
return os.Rename(tmp, t.statePath)
|
||
}
|
||
|
||
// EligibleHour reports whether a trim may START at local time lt.
|
||
func EligibleHour(lt time.Time) bool {
|
||
h := lt.Hour()
|
||
return h >= StartHour && h < EndHour
|
||
}
|
||
|
||
// weekAnchor is the most recent Wednesday StartHour:00 at or before lt (same location as lt).
|
||
func weekAnchor(lt time.Time) time.Time {
|
||
daysBack := (int(lt.Weekday()) - int(Weekday) + 7) % 7
|
||
d := lt.AddDate(0, 0, -daysBack)
|
||
a := time.Date(d.Year(), d.Month(), d.Day(), StartHour, 0, 0, 0, lt.Location())
|
||
if a.After(lt) {
|
||
d = d.AddDate(0, 0, -7)
|
||
a = time.Date(d.Year(), d.Month(), d.Day(), StartHour, 0, 0, 0, lt.Location())
|
||
}
|
||
return a
|
||
}
|
||
|
||
// due reports whether a guest with record r (ok=false: none) is due at local time lt.
|
||
func due(r Record, has bool, lt time.Time) bool {
|
||
if !has {
|
||
return true
|
||
}
|
||
anchor := weekAnchor(lt)
|
||
if r.LastAttemptAt.Before(anchor) {
|
||
return true // not tried this week
|
||
}
|
||
return !r.OK && r.Attempts < MaxAttemptsPerWeek
|
||
}
|
||
|
||
// Run looks every TickInterval until ctx ends. It never returns an error: a failed trim is a reported fact.
|
||
func (t *Trimmer) Run(ctx context.Context) {
|
||
t.logger.Info("fstrim: weekly guest disk trim starting", "schedule", ScheduleText)
|
||
timer := time.NewTimer(FirstTickDelay)
|
||
defer timer.Stop()
|
||
for {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-timer.C:
|
||
t.Pass(ctx)
|
||
timer.Reset(TickInterval)
|
||
}
|
||
}
|
||
}
|
||
|
||
// Pass is one look: outside the daytime window it does nothing; otherwise it trims every due, running, owned guest
|
||
// while holding the heavy-op gate.
|
||
func (t *Trimmer) Pass(ctx context.Context) {
|
||
lt := t.now().In(t.loc)
|
||
if !EligibleHour(lt) {
|
||
t.logger.Debug("fstrim: outside the daytime window — not looking", "local", lt.Format("Mon 15:04"))
|
||
return
|
||
}
|
||
guests, err := t.guests.Guests(ctx)
|
||
if err != nil {
|
||
t.logger.Warn("fstrim: owned-guest list unavailable — skipping this pass", "err", err)
|
||
return
|
||
}
|
||
owned := make(map[int]bool, len(guests))
|
||
var todo []int
|
||
t.mu.Lock()
|
||
for _, g := range guests {
|
||
owned[g.VMID] = true
|
||
r, has := t.records[g.VMID]
|
||
if !due(r, has, lt) {
|
||
continue
|
||
}
|
||
if g.Status != "running" {
|
||
t.logger.Info("fstrim: guest not running — trimmed when it runs", "vmid", g.VMID, "status", g.Status)
|
||
continue
|
||
}
|
||
todo = append(todo, g.VMID)
|
||
}
|
||
// A guest the agent no longer owns has no result to report.
|
||
pruned := false
|
||
for id := range t.records {
|
||
if !owned[id] {
|
||
delete(t.records, id)
|
||
pruned = true
|
||
}
|
||
}
|
||
if pruned {
|
||
if err := t.saveLocked(); err != nil {
|
||
t.logger.Warn("fstrim: state save failed", "err", err)
|
||
}
|
||
}
|
||
t.mu.Unlock()
|
||
if len(todo) == 0 {
|
||
return
|
||
}
|
||
sort.Ints(todo)
|
||
release, busy, ok := t.gate.TryAcquire(GateName)
|
||
if !ok {
|
||
t.logger.Info("fstrim: deferred — a heavy operation is in flight; retrying next hour", "busy", busy, "due_guests", len(todo))
|
||
return
|
||
}
|
||
defer release()
|
||
for _, vmid := range todo {
|
||
if ctx.Err() != nil {
|
||
return
|
||
}
|
||
t.trimOne(ctx, vmid, lt)
|
||
}
|
||
}
|
||
|
||
var trimmedLine = regexp.MustCompile(`\((\d+) bytes\) trimmed`)
|
||
|
||
// ParseTrimmed sums the "(N bytes) trimmed" lines of `pct fstrim` output and counts them (one per mount point), e.g.
|
||
// `/var/lib/lxc/9201/rootfs/: 30.1 GiB (32277680128 bytes) trimmed`.
|
||
func ParseTrimmed(out string) (bytes int64, mounts int) {
|
||
for _, m := range trimmedLine.FindAllStringSubmatch(out, -1) {
|
||
n, err := strconv.ParseInt(m[1], 10, 64)
|
||
if err != nil {
|
||
continue
|
||
}
|
||
bytes += n
|
||
mounts++
|
||
}
|
||
return bytes, mounts
|
||
}
|
||
|
||
// GiB renders bytes as "30.1 GiB".
|
||
func GiB(b int64) string { return fmt.Sprintf("%.1f GiB", float64(b)/(1<<30)) }
|
||
|
||
func (t *Trimmer) trimOne(ctx context.Context, vmid int, lt time.Time) {
|
||
start := t.now()
|
||
cctx, cancel := context.WithTimeout(ctx, PerGuestTimeout)
|
||
stdout, stderr, err := t.runner.Run(cctx, "pct", "fstrim", strconv.Itoa(vmid))
|
||
cancel()
|
||
dur := t.now().Sub(start)
|
||
bytes, mounts := ParseTrimmed(string(stdout) + "\n" + string(stderr))
|
||
|
||
t.mu.Lock()
|
||
prev, has := t.records[vmid]
|
||
r := Record{LastAttemptAt: start.UTC(), OK: err == nil, BytesTrimmed: bytes, Mounts: mounts,
|
||
DurationSeconds: float64(dur.Round(100*time.Millisecond)) / float64(time.Second), LastOKAt: prev.LastOKAt}
|
||
if has && !prev.LastAttemptAt.Before(weekAnchor(lt)) {
|
||
r.Attempts = prev.Attempts + 1
|
||
} else {
|
||
r.Attempts = 1
|
||
}
|
||
if err == nil {
|
||
r.LastOKAt = start.UTC()
|
||
} else {
|
||
msg := strings.TrimSpace(err.Error() + ": " + strings.TrimSpace(string(stderr)))
|
||
if len(msg) > 300 {
|
||
msg = msg[:300]
|
||
}
|
||
r.Error = msg
|
||
}
|
||
t.records[vmid] = r
|
||
saveErr := t.saveLocked()
|
||
t.mu.Unlock()
|
||
|
||
if err == nil {
|
||
t.logger.Info(fmt.Sprintf("fstrim: guest %d trimmed %s in %.1fs", vmid, GiB(bytes), r.DurationSeconds),
|
||
"vmid", vmid, "bytes_trimmed", bytes, "mounts", mounts, "duration_s", r.DurationSeconds)
|
||
if mounts == 0 {
|
||
t.logger.Warn("fstrim: pct fstrim succeeded but reported no trimmed mount — output not understood",
|
||
"vmid", vmid, "stdout", strings.TrimSpace(string(stdout)))
|
||
}
|
||
} else {
|
||
t.logger.Warn(fmt.Sprintf("fstrim: guest %d trim FAILED after %.1fs", vmid, r.DurationSeconds),
|
||
"vmid", vmid, "attempt", r.Attempts, "max_attempts_per_week", MaxAttemptsPerWeek, "err", r.Error)
|
||
}
|
||
if saveErr != nil {
|
||
t.logger.Warn("fstrim: state save failed — the result will not survive a restart", "path", t.statePath, "err", saveErr)
|
||
}
|
||
}
|
||
|
||
// GuestDiskTrimStatus implements hub.GuestDiskTrimReporter: a pure read of the persisted results (never runs pct).
|
||
func (t *Trimmer) GuestDiskTrimStatus(context.Context) *hub.GuestDiskTrimStatus {
|
||
t.mu.Lock()
|
||
defer t.mu.Unlock()
|
||
out := &hub.GuestDiskTrimStatus{Schedule: ScheduleText}
|
||
ids := make([]int, 0, len(t.records))
|
||
for id := range t.records {
|
||
ids = append(ids, id)
|
||
}
|
||
sort.Ints(ids)
|
||
for _, id := range ids {
|
||
r := t.records[id]
|
||
g := hub.GuestDiskTrim{VMID: id, LastAttemptAt: r.LastAttemptAt.UTC().Format(time.RFC3339), OK: r.OK,
|
||
BytesTrimmed: r.BytesTrimmed, Mounts: r.Mounts, DurationSeconds: r.DurationSeconds, Error: r.Error}
|
||
if !r.LastOKAt.IsZero() {
|
||
g.LastOKAt = r.LastOKAt.UTC().Format(time.RFC3339)
|
||
}
|
||
out.Guests = append(out.Guests, g)
|
||
}
|
||
return out
|
||
}
|