// Package fstrim is the weekly guest disk trim (R-444, operator ruling `09` §3 decision 139). // // Why: a thin pool only ever grows from blocks a guest has already FREED — `fstrim` inside the unprivileged container // is refused (FITRIM: Operation not permitted), and nothing else on the box gives the blocks back. A full thin pool // takes every guest on the host read-only, so the pool can reach 100 % from deleted data alone. Measured on demo-hp // 2026-10-06 09:14Z: `pct fstrim 9201` rc 0 in 24.4 s, pool 65.53 % -> 33.40 %, 18/18 app probes 200, max 1.1 s // (audits/ten-answers-2026-10-06/r444-measure.txt). // // The rule, each part pinned by a test in fstrim_test.go: // - Weekly: a guest is DUE from Wednesday 10:00 local until it has been trimmed once since then (a box that was off // on Wednesday catches up at its next eligible hour). // - Daytime only: a trim starts only between 10:00 and 20:59 local — never in the night window (01:00–06:59) where // the backups and the restore-tests run (TestEligibleHourNeverInTheNight). // - Never beside a backup, a restore-test or another heavy operation: the pass holds the host-wide one-heavy-op gate // (backup.InFlight) for its whole run; a busy gate DEFERS the pass to the next hourly tick. // - A failed trim is retried at the next eligible hour, at most MaxAttemptsPerWeek times in one week. // - The last result per guest (time, bytes, ok/fail) is persisted, so a restart neither loses it nor re-trims. // // The command is the ONE exact sudoers shape `pct fstrim ` (FELHOM_FSTRIM). Only guests from the pool-verified // source (ListLXC ∩ the felhom pool, audit A1) and only RUNNING ones are trimmed. package fstrim import ( "context" "encoding/json" "fmt" "log/slog" "os" "path/filepath" "regexp" "sort" "strconv" "strings" "sync" "time" "gitea.dooplex.hu/admin/felhom-agent/internal/hub" "gitea.dooplex.hu/admin/felhom-agent/internal/proxmox" ) // Schedule. The weekday/hours are fixed on purpose (one sentence the operator can read on the System page). const ( Weekday = time.Wednesday StartHour = 10 // first eligible local hour (inclusive) EndHour = 21 // first NOT-eligible local hour (exclusive): last start is 20:59 MaxAttemptsPerWeek = 3 // TickInterval is how often the job looks; a deferred or failed pass is therefore retried the next hour. TickInterval = time.Hour // FirstTickDelay lets the agent settle after a start before the first look. FirstTickDelay = 5 * time.Minute // PerGuestTimeout bounds one `pct fstrim` (measured 24.4 s for 84 GiB). PerGuestTimeout = 30 * time.Minute ) // ScheduleText is the human description carried on the host report. const ScheduleText = "weekly, due Wednesday from 10:00 host-local time; starts only 10:00-20:59; never beside a backup or restore-test" // Runner runs a host command (proxmox.ExecRunner in production, through `sudo -n`). type Runner interface { Run(ctx context.Context, name string, args ...string) (stdout, stderr []byte, err error) } // GuestSource yields the guests this agent OWNS (the pool-verified source, never a bare ListLXC). type GuestSource interface { Guests(ctx context.Context) ([]proxmox.Guest, error) } // Gate is the host-wide one-heavy-operation gate (*backup.InFlight). type Gate interface { TryAcquire(what string) (release func(), busy string, ok bool) } // GateName is what the gate reports as busy while a trim runs. const GateName = "guest-fstrim" // Record is one guest's last trim attempt, as persisted. type Record struct { LastAttemptAt time.Time `json:"last_attempt_at"` OK bool `json:"ok"` BytesTrimmed int64 `json:"bytes_trimmed"` Mounts int `json:"mounts"` DurationSeconds float64 `json:"duration_seconds"` LastOKAt time.Time `json:"last_ok_at,omitempty"` Error string `json:"error,omitempty"` // Attempts counts the attempts since the current week's due time (reset by the first attempt of a new week). Attempts int `json:"attempts"` } // Trimmer is the weekly job. type Trimmer struct { runner Runner guests GuestSource gate Gate statePath string logger *slog.Logger loc *time.Location now func() time.Time mu sync.Mutex records map[int]Record } // New builds the job and loads the persisted state. A missing state file is an empty state; a corrupt one is logged // and treated as empty (the cost is one extra trim, never a missed one). func New(runner Runner, guests GuestSource, gate Gate, statePath string, logger *slog.Logger) *Trimmer { if logger == nil { logger = slog.Default() } t := &Trimmer{runner: runner, guests: guests, gate: gate, statePath: statePath, logger: logger, loc: time.Local, now: time.Now, records: map[int]Record{}} t.load() return t } func (t *Trimmer) load() { data, err := os.ReadFile(t.statePath) if err != nil { if !os.IsNotExist(err) { t.logger.Warn("fstrim: state read failed — starting empty", "path", t.statePath, "err", err) } return } var raw map[string]Record if err := json.Unmarshal(data, &raw); err != nil { t.logger.Warn("fstrim: state file corrupt — starting empty", "path", t.statePath, "err", err) return } for k, r := range raw { if id, err := strconv.Atoi(k); err == nil && id > 0 { t.records[id] = r } } } func (t *Trimmer) saveLocked() error { raw := make(map[string]Record, len(t.records)) for id, r := range t.records { raw[strconv.Itoa(id)] = r } data, err := json.MarshalIndent(raw, "", " ") if err != nil { return err } if err := os.MkdirAll(filepath.Dir(t.statePath), 0o755); err != nil { return err } tmp := t.statePath + ".tmp" if err := os.WriteFile(tmp, data, 0o600); err != nil { os.Remove(tmp) return err } return os.Rename(tmp, t.statePath) } // EligibleHour reports whether a trim may START at local time lt. func EligibleHour(lt time.Time) bool { h := lt.Hour() return h >= StartHour && h < EndHour } // weekAnchor is the most recent Wednesday StartHour:00 at or before lt (same location as lt). func weekAnchor(lt time.Time) time.Time { daysBack := (int(lt.Weekday()) - int(Weekday) + 7) % 7 d := lt.AddDate(0, 0, -daysBack) a := time.Date(d.Year(), d.Month(), d.Day(), StartHour, 0, 0, 0, lt.Location()) if a.After(lt) { d = d.AddDate(0, 0, -7) a = time.Date(d.Year(), d.Month(), d.Day(), StartHour, 0, 0, 0, lt.Location()) } return a } // due reports whether a guest with record r (ok=false: none) is due at local time lt. func due(r Record, has bool, lt time.Time) bool { if !has { return true } anchor := weekAnchor(lt) if r.LastAttemptAt.Before(anchor) { return true // not tried this week } return !r.OK && r.Attempts < MaxAttemptsPerWeek } // Run looks every TickInterval until ctx ends. It never returns an error: a failed trim is a reported fact. func (t *Trimmer) Run(ctx context.Context) { t.logger.Info("fstrim: weekly guest disk trim starting", "schedule", ScheduleText) timer := time.NewTimer(FirstTickDelay) defer timer.Stop() for { select { case <-ctx.Done(): return case <-timer.C: t.Pass(ctx) timer.Reset(TickInterval) } } } // Pass is one look: outside the daytime window it does nothing; otherwise it trims every due, running, owned guest // while holding the heavy-op gate. func (t *Trimmer) Pass(ctx context.Context) { lt := t.now().In(t.loc) if !EligibleHour(lt) { t.logger.Debug("fstrim: outside the daytime window — not looking", "local", lt.Format("Mon 15:04")) return } guests, err := t.guests.Guests(ctx) if err != nil { t.logger.Warn("fstrim: owned-guest list unavailable — skipping this pass", "err", err) return } owned := make(map[int]bool, len(guests)) var todo []int t.mu.Lock() for _, g := range guests { owned[g.VMID] = true r, has := t.records[g.VMID] if !due(r, has, lt) { continue } if g.Status != "running" { t.logger.Info("fstrim: guest not running — trimmed when it runs", "vmid", g.VMID, "status", g.Status) continue } todo = append(todo, g.VMID) } // A guest the agent no longer owns has no result to report. pruned := false for id := range t.records { if !owned[id] { delete(t.records, id) pruned = true } } if pruned { if err := t.saveLocked(); err != nil { t.logger.Warn("fstrim: state save failed", "err", err) } } t.mu.Unlock() if len(todo) == 0 { return } sort.Ints(todo) release, busy, ok := t.gate.TryAcquire(GateName) if !ok { t.logger.Info("fstrim: deferred — a heavy operation is in flight; retrying next hour", "busy", busy, "due_guests", len(todo)) return } defer release() for _, vmid := range todo { if ctx.Err() != nil { return } t.trimOne(ctx, vmid, lt) } } var trimmedLine = regexp.MustCompile(`\((\d+) bytes\) trimmed`) // ParseTrimmed sums the "(N bytes) trimmed" lines of `pct fstrim` output and counts them (one per mount point), e.g. // `/var/lib/lxc/9201/rootfs/: 30.1 GiB (32277680128 bytes) trimmed`. func ParseTrimmed(out string) (bytes int64, mounts int) { for _, m := range trimmedLine.FindAllStringSubmatch(out, -1) { n, err := strconv.ParseInt(m[1], 10, 64) if err != nil { continue } bytes += n mounts++ } return bytes, mounts } // GiB renders bytes as "30.1 GiB". func GiB(b int64) string { return fmt.Sprintf("%.1f GiB", float64(b)/(1<<30)) } func (t *Trimmer) trimOne(ctx context.Context, vmid int, lt time.Time) { start := t.now() cctx, cancel := context.WithTimeout(ctx, PerGuestTimeout) stdout, stderr, err := t.runner.Run(cctx, "pct", "fstrim", strconv.Itoa(vmid)) cancel() dur := t.now().Sub(start) bytes, mounts := ParseTrimmed(string(stdout) + "\n" + string(stderr)) t.mu.Lock() prev, has := t.records[vmid] r := Record{LastAttemptAt: start.UTC(), OK: err == nil, BytesTrimmed: bytes, Mounts: mounts, DurationSeconds: float64(dur.Round(100*time.Millisecond)) / float64(time.Second), LastOKAt: prev.LastOKAt} if has && !prev.LastAttemptAt.Before(weekAnchor(lt)) { r.Attempts = prev.Attempts + 1 } else { r.Attempts = 1 } if err == nil { r.LastOKAt = start.UTC() } else { msg := strings.TrimSpace(err.Error() + ": " + strings.TrimSpace(string(stderr))) if len(msg) > 300 { msg = msg[:300] } r.Error = msg } t.records[vmid] = r saveErr := t.saveLocked() t.mu.Unlock() if err == nil { t.logger.Info(fmt.Sprintf("fstrim: guest %d trimmed %s in %.1fs", vmid, GiB(bytes), r.DurationSeconds), "vmid", vmid, "bytes_trimmed", bytes, "mounts", mounts, "duration_s", r.DurationSeconds) if mounts == 0 { t.logger.Warn("fstrim: pct fstrim succeeded but reported no trimmed mount — output not understood", "vmid", vmid, "stdout", strings.TrimSpace(string(stdout))) } } else { t.logger.Warn(fmt.Sprintf("fstrim: guest %d trim FAILED after %.1fs", vmid, r.DurationSeconds), "vmid", vmid, "attempt", r.Attempts, "max_attempts_per_week", MaxAttemptsPerWeek, "err", r.Error) } if saveErr != nil { t.logger.Warn("fstrim: state save failed — the result will not survive a restart", "path", t.statePath, "err", saveErr) } } // GuestDiskTrimStatus implements hub.GuestDiskTrimReporter: a pure read of the persisted results (never runs pct). func (t *Trimmer) GuestDiskTrimStatus(context.Context) *hub.GuestDiskTrimStatus { t.mu.Lock() defer t.mu.Unlock() out := &hub.GuestDiskTrimStatus{Schedule: ScheduleText} ids := make([]int, 0, len(t.records)) for id := range t.records { ids = append(ids, id) } sort.Ints(ids) for _, id := range ids { r := t.records[id] g := hub.GuestDiskTrim{VMID: id, LastAttemptAt: r.LastAttemptAt.UTC().Format(time.RFC3339), OK: r.OK, BytesTrimmed: r.BytesTrimmed, Mounts: r.Mounts, DurationSeconds: r.DurationSeconds, Error: r.Error} if !r.LastOKAt.IsZero() { g.LastOKAt = r.LastOKAt.UTC().Format(time.RFC3339) } out.Guests = append(out.Guests, g) } return out }