R-444: weekly guest disk trim (pct fstrim) outside the night, under the heavy-op gate

Operator ruling 09 §3 decision 139. One exact sudoers rule FELHOM_FSTRIM
(`/usr/sbin/pct ^fstrim [0-9]+$`) + manifest entry guest-fstrim; new
internal/fstrim job: due Wednesday from 10:00 host-local, starts only
10:00-20:59, holds backup.InFlight (busy -> deferred to the next hourly
tick), failed trim retried at most 3x per week, bytes parsed from the
"(N bytes) trimmed" lines, last result per guest persisted in
<state_dir>/guest-disk-trim.json and reported as guest_disk_trim.
Config opt-out: "disk_trim": {"disable": true}.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-06 11:22:45 +02:00
parent 37e98f452b
commit ee71abd1d4
12 changed files with 838 additions and 1 deletions
+346
View File
@@ -0,0 +1,346 @@
// Package fstrim is the weekly guest disk trim (R-444, operator ruling `09` §3 decision 139).
//
// Why: a thin pool only ever grows from blocks a guest has already FREED — `fstrim` inside the unprivileged container
// is refused (FITRIM: Operation not permitted), and nothing else on the box gives the blocks back. A full thin pool
// takes every guest on the host read-only, so the pool can reach 100 % from deleted data alone. Measured on demo-hp
// 2026-10-06 09:14Z: `pct fstrim 9201` rc 0 in 24.4 s, pool 65.53 % -> 33.40 %, 18/18 app probes 200, max 1.1 s
// (audits/ten-answers-2026-10-06/r444-measure.txt).
//
// The rule, each part pinned by a test in fstrim_test.go:
// - Weekly: a guest is DUE from Wednesday 10:00 local until it has been trimmed once since then (a box that was off
// on Wednesday catches up at its next eligible hour).
// - Daytime only: a trim starts only between 10:00 and 20:59 local — never in the night window (01:00–06:59) where
// the backups and the restore-tests run (TestEligibleHourNeverInTheNight).
// - Never beside a backup, a restore-test or another heavy operation: the pass holds the host-wide one-heavy-op gate
// (backup.InFlight) for its whole run; a busy gate DEFERS the pass to the next hourly tick.
// - A failed trim is retried at the next eligible hour, at most MaxAttemptsPerWeek times in one week.
// - The last result per guest (time, bytes, ok/fail) is persisted, so a restart neither loses it nor re-trims.
//
// The command is the ONE exact sudoers shape `pct fstrim <vmid>` (FELHOM_FSTRIM). Only guests from the pool-verified
// source (ListLXC ∩ the felhom pool, audit A1) and only RUNNING ones are trimmed.
package fstrim
import (
"context"
"encoding/json"
"fmt"
"log/slog"
"os"
"path/filepath"
"regexp"
"sort"
"strconv"
"strings"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
)
// Schedule. The weekday/hours are fixed on purpose (one sentence the operator can read on the System page).
const (
Weekday = time.Wednesday
StartHour = 10 // first eligible local hour (inclusive)
EndHour = 21 // first NOT-eligible local hour (exclusive): last start is 20:59
MaxAttemptsPerWeek = 3
// TickInterval is how often the job looks; a deferred or failed pass is therefore retried the next hour.
TickInterval = time.Hour
// FirstTickDelay lets the agent settle after a start before the first look.
FirstTickDelay = 5 * time.Minute
// PerGuestTimeout bounds one `pct fstrim` (measured 24.4 s for 84 GiB).
PerGuestTimeout = 30 * time.Minute
)
// ScheduleText is the human description carried on the host report.
const ScheduleText = "weekly, due Wednesday from 10:00 host-local time; starts only 10:00-20:59; never beside a backup or restore-test"
// Runner runs a host command (proxmox.ExecRunner in production, through `sudo -n`).
type Runner interface {
Run(ctx context.Context, name string, args ...string) (stdout, stderr []byte, err error)
}
// GuestSource yields the guests this agent OWNS (the pool-verified source, never a bare ListLXC).
type GuestSource interface {
Guests(ctx context.Context) ([]proxmox.Guest, error)
}
// Gate is the host-wide one-heavy-operation gate (*backup.InFlight).
type Gate interface {
TryAcquire(what string) (release func(), busy string, ok bool)
}
// GateName is what the gate reports as busy while a trim runs.
const GateName = "guest-fstrim"
// Record is one guest's last trim attempt, as persisted.
type Record struct {
LastAttemptAt time.Time `json:"last_attempt_at"`
OK bool `json:"ok"`
BytesTrimmed int64 `json:"bytes_trimmed"`
Mounts int `json:"mounts"`
DurationSeconds float64 `json:"duration_seconds"`
LastOKAt time.Time `json:"last_ok_at,omitempty"`
Error string `json:"error,omitempty"`
// Attempts counts the attempts since the current week's due time (reset by the first attempt of a new week).
Attempts int `json:"attempts"`
}
// Trimmer is the weekly job.
type Trimmer struct {
runner Runner
guests GuestSource
gate Gate
statePath string
logger *slog.Logger
loc *time.Location
now func() time.Time
mu sync.Mutex
records map[int]Record
}
// New builds the job and loads the persisted state. A missing state file is an empty state; a corrupt one is logged
// and treated as empty (the cost is one extra trim, never a missed one).
func New(runner Runner, guests GuestSource, gate Gate, statePath string, logger *slog.Logger) *Trimmer {
if logger == nil {
logger = slog.Default()
}
t := &Trimmer{runner: runner, guests: guests, gate: gate, statePath: statePath, logger: logger,
loc: time.Local, now: time.Now, records: map[int]Record{}}
t.load()
return t
}
func (t *Trimmer) load() {
data, err := os.ReadFile(t.statePath)
if err != nil {
if !os.IsNotExist(err) {
t.logger.Warn("fstrim: state read failed — starting empty", "path", t.statePath, "err", err)
}
return
}
var raw map[string]Record
if err := json.Unmarshal(data, &raw); err != nil {
t.logger.Warn("fstrim: state file corrupt — starting empty", "path", t.statePath, "err", err)
return
}
for k, r := range raw {
if id, err := strconv.Atoi(k); err == nil && id > 0 {
t.records[id] = r
}
}
}
func (t *Trimmer) saveLocked() error {
raw := make(map[string]Record, len(t.records))
for id, r := range t.records {
raw[strconv.Itoa(id)] = r
}
data, err := json.MarshalIndent(raw, "", " ")
if err != nil {
return err
}
if err := os.MkdirAll(filepath.Dir(t.statePath), 0o755); err != nil {
return err
}
tmp := t.statePath + ".tmp"
if err := os.WriteFile(tmp, data, 0o600); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, t.statePath)
}
// EligibleHour reports whether a trim may START at local time lt.
func EligibleHour(lt time.Time) bool {
h := lt.Hour()
return h >= StartHour && h < EndHour
}
// weekAnchor is the most recent Wednesday StartHour:00 at or before lt (same location as lt).
func weekAnchor(lt time.Time) time.Time {
daysBack := (int(lt.Weekday()) - int(Weekday) + 7) % 7
d := lt.AddDate(0, 0, -daysBack)
a := time.Date(d.Year(), d.Month(), d.Day(), StartHour, 0, 0, 0, lt.Location())
if a.After(lt) {
d = d.AddDate(0, 0, -7)
a = time.Date(d.Year(), d.Month(), d.Day(), StartHour, 0, 0, 0, lt.Location())
}
return a
}
// due reports whether a guest with record r (ok=false: none) is due at local time lt.
func due(r Record, has bool, lt time.Time) bool {
if !has {
return true
}
anchor := weekAnchor(lt)
if r.LastAttemptAt.Before(anchor) {
return true // not tried this week
}
return !r.OK && r.Attempts < MaxAttemptsPerWeek
}
// Run looks every TickInterval until ctx ends. It never returns an error: a failed trim is a reported fact.
func (t *Trimmer) Run(ctx context.Context) {
t.logger.Info("fstrim: weekly guest disk trim starting", "schedule", ScheduleText)
timer := time.NewTimer(FirstTickDelay)
defer timer.Stop()
for {
select {
case <-ctx.Done():
return
case <-timer.C:
t.Pass(ctx)
timer.Reset(TickInterval)
}
}
}
// Pass is one look: outside the daytime window it does nothing; otherwise it trims every due, running, owned guest
// while holding the heavy-op gate.
func (t *Trimmer) Pass(ctx context.Context) {
lt := t.now().In(t.loc)
if !EligibleHour(lt) {
t.logger.Debug("fstrim: outside the daytime window — not looking", "local", lt.Format("Mon 15:04"))
return
}
guests, err := t.guests.Guests(ctx)
if err != nil {
t.logger.Warn("fstrim: owned-guest list unavailable — skipping this pass", "err", err)
return
}
owned := make(map[int]bool, len(guests))
var todo []int
t.mu.Lock()
for _, g := range guests {
owned[g.VMID] = true
r, has := t.records[g.VMID]
if !due(r, has, lt) {
continue
}
if g.Status != "running" {
t.logger.Info("fstrim: guest not running — trimmed when it runs", "vmid", g.VMID, "status", g.Status)
continue
}
todo = append(todo, g.VMID)
}
// A guest the agent no longer owns has no result to report.
pruned := false
for id := range t.records {
if !owned[id] {
delete(t.records, id)
pruned = true
}
}
if pruned {
if err := t.saveLocked(); err != nil {
t.logger.Warn("fstrim: state save failed", "err", err)
}
}
t.mu.Unlock()
if len(todo) == 0 {
return
}
sort.Ints(todo)
release, busy, ok := t.gate.TryAcquire(GateName)
if !ok {
t.logger.Info("fstrim: deferred — a heavy operation is in flight; retrying next hour", "busy", busy, "due_guests", len(todo))
return
}
defer release()
for _, vmid := range todo {
if ctx.Err() != nil {
return
}
t.trimOne(ctx, vmid, lt)
}
}
var trimmedLine = regexp.MustCompile(`\((\d+) bytes\) trimmed`)
// ParseTrimmed sums the "(N bytes) trimmed" lines of `pct fstrim` output and counts them (one per mount point), e.g.
// `/var/lib/lxc/9201/rootfs/: 30.1 GiB (32277680128 bytes) trimmed`.
func ParseTrimmed(out string) (bytes int64, mounts int) {
for _, m := range trimmedLine.FindAllStringSubmatch(out, -1) {
n, err := strconv.ParseInt(m[1], 10, 64)
if err != nil {
continue
}
bytes += n
mounts++
}
return bytes, mounts
}
// GiB renders bytes as "30.1 GiB".
func GiB(b int64) string { return fmt.Sprintf("%.1f GiB", float64(b)/(1<<30)) }
func (t *Trimmer) trimOne(ctx context.Context, vmid int, lt time.Time) {
start := t.now()
cctx, cancel := context.WithTimeout(ctx, PerGuestTimeout)
stdout, stderr, err := t.runner.Run(cctx, "pct", "fstrim", strconv.Itoa(vmid))
cancel()
dur := t.now().Sub(start)
bytes, mounts := ParseTrimmed(string(stdout) + "\n" + string(stderr))
t.mu.Lock()
prev, has := t.records[vmid]
r := Record{LastAttemptAt: start.UTC(), OK: err == nil, BytesTrimmed: bytes, Mounts: mounts,
DurationSeconds: float64(dur.Round(100*time.Millisecond)) / float64(time.Second), LastOKAt: prev.LastOKAt}
if has && !prev.LastAttemptAt.Before(weekAnchor(lt)) {
r.Attempts = prev.Attempts + 1
} else {
r.Attempts = 1
}
if err == nil {
r.LastOKAt = start.UTC()
} else {
msg := strings.TrimSpace(err.Error() + ": " + strings.TrimSpace(string(stderr)))
if len(msg) > 300 {
msg = msg[:300]
}
r.Error = msg
}
t.records[vmid] = r
saveErr := t.saveLocked()
t.mu.Unlock()
if err == nil {
t.logger.Info(fmt.Sprintf("fstrim: guest %d trimmed %s in %.1fs", vmid, GiB(bytes), r.DurationSeconds),
"vmid", vmid, "bytes_trimmed", bytes, "mounts", mounts, "duration_s", r.DurationSeconds)
if mounts == 0 {
t.logger.Warn("fstrim: pct fstrim succeeded but reported no trimmed mount — output not understood",
"vmid", vmid, "stdout", strings.TrimSpace(string(stdout)))
}
} else {
t.logger.Warn(fmt.Sprintf("fstrim: guest %d trim FAILED after %.1fs", vmid, r.DurationSeconds),
"vmid", vmid, "attempt", r.Attempts, "max_attempts_per_week", MaxAttemptsPerWeek, "err", r.Error)
}
if saveErr != nil {
t.logger.Warn("fstrim: state save failed — the result will not survive a restart", "path", t.statePath, "err", saveErr)
}
}
// GuestDiskTrimStatus implements hub.GuestDiskTrimReporter: a pure read of the persisted results (never runs pct).
func (t *Trimmer) GuestDiskTrimStatus(context.Context) *hub.GuestDiskTrimStatus {
t.mu.Lock()
defer t.mu.Unlock()
out := &hub.GuestDiskTrimStatus{Schedule: ScheduleText}
ids := make([]int, 0, len(t.records))
for id := range t.records {
ids = append(ids, id)
}
sort.Ints(ids)
for _, id := range ids {
r := t.records[id]
g := hub.GuestDiskTrim{VMID: id, LastAttemptAt: r.LastAttemptAt.UTC().Format(time.RFC3339), OK: r.OK,
BytesTrimmed: r.BytesTrimmed, Mounts: r.Mounts, DurationSeconds: r.DurationSeconds, Error: r.Error}
if !r.LastOKAt.IsZero() {
g.LastOKAt = r.LastOKAt.UTC().Format(time.RFC3339)
}
out.Guests = append(out.Guests, g)
}
return out
}