9bdb4dae8f
gates / gates (push) Successful in 13s
Space preflight before anything is created (uncompressed size from the vzdump log / PBS snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin pool crossing 90% requests an immediate host report. Six red-proofs. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
82 lines
3.0 KiB
Go
82 lines
3.0 KiB
Go
package main
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"log/slog"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/backup"
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/reconcile"
|
|
)
|
|
|
|
// janitorInterval is how often the leftovers of an interrupted restore-test or backup are retried
|
|
// (R-672 rule 3, R-673). Both used to be resolved ONLY at agent start: on 2026-09-24 a failed scratch
|
|
// teardown kept a full thin pool full for 2.5 h, and a stale `snapshot-delete` lock blocked 9201's
|
|
// whole-box backups for five hours — each cleared within a minute of an agent restart.
|
|
const janitorInterval = 10 * time.Minute
|
|
|
|
// janitorDeps are the janitor's seams (tests drive one pass with fakes).
|
|
type janitorDeps struct {
|
|
retryScratch func(ctx context.Context) reconcile.ScratchRetryResult
|
|
staleLocks func(ctx context.Context) // localapi Server.RecoverStaleLockedGuests; nil when the local API is off
|
|
heavy *backup.InFlight
|
|
record func(hub.RestoreTest)
|
|
now func() time.Time
|
|
logger *slog.Logger
|
|
}
|
|
|
|
// janitorPass is one pass. The stale-lock sweep runs only while holding the one-heavy-operation gate, so
|
|
// no agent backup can START between its "no vzdump is running" check and its unlock (at start-up the
|
|
// sweep ran before the backup loop existed; on a timer that ordering must be made, not assumed). A busy
|
|
// gate skips the sweep this pass — the next pass retries.
|
|
func janitorPass(ctx context.Context, d janitorDeps) {
|
|
if d.retryScratch != nil {
|
|
r := d.retryScratch(ctx)
|
|
if r.Examined > 0 {
|
|
d.logger.Info("janitor: restore-test scratch retry pass", "examined", r.Examined,
|
|
"destroyed", r.Destroyed, "already_gone", r.Clean, "failed", r.Failed)
|
|
}
|
|
for _, vmid := range r.GaveUp {
|
|
// The operator is told through the existing restore-test failure path: the hub raises
|
|
// restore_test_failed (operator) once per distinct archive — this record's archive names
|
|
// the stuck scratch guest.
|
|
if d.record != nil {
|
|
d.record(hub.RestoreTest{
|
|
SourceArchive: fmt.Sprintf("scratch-teardown:%d", vmid),
|
|
ScratchVMID: vmid,
|
|
Pass: false,
|
|
Error: fmt.Sprintf("restore-test scratch guest %d could not be torn down after %d retries — it holds its disks; remove it by hand (pct destroy %d) after checking what keeps it busy",
|
|
vmid, reconcile.MaxTeardownTries, vmid),
|
|
TestedAt: d.now().UTC().Format(time.RFC3339),
|
|
})
|
|
}
|
|
}
|
|
}
|
|
if d.staleLocks != nil {
|
|
release, busy, ok := d.heavy.TryAcquire("stale-lock-sweep")
|
|
if !ok {
|
|
d.logger.Info("janitor: stale-lock sweep deferred — a heavy operation is in flight", "busy", busy)
|
|
return
|
|
}
|
|
defer release()
|
|
d.staleLocks(ctx)
|
|
}
|
|
}
|
|
|
|
// runJanitor runs janitorPass every janitorInterval until ctx ends.
|
|
func runJanitor(ctx context.Context, d janitorDeps) {
|
|
d.logger.Info("janitor: starting (restore-test scratch retry + stale-lock sweep)", "interval", janitorInterval)
|
|
t := time.NewTicker(janitorInterval)
|
|
defer t.Stop()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-t.C:
|
|
janitorPass(ctx, d)
|
|
}
|
|
}
|
|
}
|