Files
felhom-agent/cmd/felhom-agent/janitor.go
T
admin 9bdb4dae8f
gates / gates (push) Successful in 13s
v0.133.0: a restore-test can never fill a box's disk; leftovers retried on a timer (R-672, R-673)
Space preflight before anything is created (uncompressed size from the vzdump log / PBS
snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage
is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and
the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin
pool crossing 90% requests an immediate host report. Six red-proofs.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-24 16:23:15 +02:00

82 lines
3.0 KiB
Go

package main
import (
"context"
"fmt"
"log/slog"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/backup"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
"gitea.dooplex.hu/admin/felhom-agent/internal/reconcile"
)
// janitorInterval is how often the leftovers of an interrupted restore-test or backup are retried
// (R-672 rule 3, R-673). Both used to be resolved ONLY at agent start: on 2026-09-24 a failed scratch
// teardown kept a full thin pool full for 2.5 h, and a stale `snapshot-delete` lock blocked 9201's
// whole-box backups for five hours — each cleared within a minute of an agent restart.
const janitorInterval = 10 * time.Minute
// janitorDeps are the janitor's seams (tests drive one pass with fakes).
type janitorDeps struct {
retryScratch func(ctx context.Context) reconcile.ScratchRetryResult
staleLocks func(ctx context.Context) // localapi Server.RecoverStaleLockedGuests; nil when the local API is off
heavy *backup.InFlight
record func(hub.RestoreTest)
now func() time.Time
logger *slog.Logger
}
// janitorPass is one pass. The stale-lock sweep runs only while holding the one-heavy-operation gate, so
// no agent backup can START between its "no vzdump is running" check and its unlock (at start-up the
// sweep ran before the backup loop existed; on a timer that ordering must be made, not assumed). A busy
// gate skips the sweep this pass — the next pass retries.
func janitorPass(ctx context.Context, d janitorDeps) {
if d.retryScratch != nil {
r := d.retryScratch(ctx)
if r.Examined > 0 {
d.logger.Info("janitor: restore-test scratch retry pass", "examined", r.Examined,
"destroyed", r.Destroyed, "already_gone", r.Clean, "failed", r.Failed)
}
for _, vmid := range r.GaveUp {
// The operator is told through the existing restore-test failure path: the hub raises
// restore_test_failed (operator) once per distinct archive — this record's archive names
// the stuck scratch guest.
if d.record != nil {
d.record(hub.RestoreTest{
SourceArchive: fmt.Sprintf("scratch-teardown:%d", vmid),
ScratchVMID: vmid,
Pass: false,
Error: fmt.Sprintf("restore-test scratch guest %d could not be torn down after %d retries — it holds its disks; remove it by hand (pct destroy %d) after checking what keeps it busy",
vmid, reconcile.MaxTeardownTries, vmid),
TestedAt: d.now().UTC().Format(time.RFC3339),
})
}
}
}
if d.staleLocks != nil {
release, busy, ok := d.heavy.TryAcquire("stale-lock-sweep")
if !ok {
d.logger.Info("janitor: stale-lock sweep deferred — a heavy operation is in flight", "busy", busy)
return
}
defer release()
d.staleLocks(ctx)
}
}
// runJanitor runs janitorPass every janitorInterval until ctx ends.
func runJanitor(ctx context.Context, d janitorDeps) {
d.logger.Info("janitor: starting (restore-test scratch retry + stale-lock sweep)", "interval", janitorInterval)
t := time.NewTicker(janitorInterval)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
janitorPass(ctx, d)
}
}
}