v0.133.0: a restore-test can never fill a box's disk; leftovers retried on a timer (R-672, R-673)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
Space preflight before anything is created (uncompressed size from the vzdump log / PBS snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin pool crossing 90% requests an immediate host report. Six red-proofs. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,81 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/backup"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/reconcile"
|
||||
)
|
||||
|
||||
// janitorInterval is how often the leftovers of an interrupted restore-test or backup are retried
|
||||
// (R-672 rule 3, R-673). Both used to be resolved ONLY at agent start: on 2026-09-24 a failed scratch
|
||||
// teardown kept a full thin pool full for 2.5 h, and a stale `snapshot-delete` lock blocked 9201's
|
||||
// whole-box backups for five hours — each cleared within a minute of an agent restart.
|
||||
const janitorInterval = 10 * time.Minute
|
||||
|
||||
// janitorDeps are the janitor's seams (tests drive one pass with fakes).
|
||||
type janitorDeps struct {
|
||||
retryScratch func(ctx context.Context) reconcile.ScratchRetryResult
|
||||
staleLocks func(ctx context.Context) // localapi Server.RecoverStaleLockedGuests; nil when the local API is off
|
||||
heavy *backup.InFlight
|
||||
record func(hub.RestoreTest)
|
||||
now func() time.Time
|
||||
logger *slog.Logger
|
||||
}
|
||||
|
||||
// janitorPass is one pass. The stale-lock sweep runs only while holding the one-heavy-operation gate, so
|
||||
// no agent backup can START between its "no vzdump is running" check and its unlock (at start-up the
|
||||
// sweep ran before the backup loop existed; on a timer that ordering must be made, not assumed). A busy
|
||||
// gate skips the sweep this pass — the next pass retries.
|
||||
func janitorPass(ctx context.Context, d janitorDeps) {
|
||||
if d.retryScratch != nil {
|
||||
r := d.retryScratch(ctx)
|
||||
if r.Examined > 0 {
|
||||
d.logger.Info("janitor: restore-test scratch retry pass", "examined", r.Examined,
|
||||
"destroyed", r.Destroyed, "already_gone", r.Clean, "failed", r.Failed)
|
||||
}
|
||||
for _, vmid := range r.GaveUp {
|
||||
// The operator is told through the existing restore-test failure path: the hub raises
|
||||
// restore_test_failed (operator) once per distinct archive — this record's archive names
|
||||
// the stuck scratch guest.
|
||||
if d.record != nil {
|
||||
d.record(hub.RestoreTest{
|
||||
SourceArchive: fmt.Sprintf("scratch-teardown:%d", vmid),
|
||||
ScratchVMID: vmid,
|
||||
Pass: false,
|
||||
Error: fmt.Sprintf("restore-test scratch guest %d could not be torn down after %d retries — it holds its disks; remove it by hand (pct destroy %d) after checking what keeps it busy",
|
||||
vmid, reconcile.MaxTeardownTries, vmid),
|
||||
TestedAt: d.now().UTC().Format(time.RFC3339),
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
if d.staleLocks != nil {
|
||||
release, busy, ok := d.heavy.TryAcquire("stale-lock-sweep")
|
||||
if !ok {
|
||||
d.logger.Info("janitor: stale-lock sweep deferred — a heavy operation is in flight", "busy", busy)
|
||||
return
|
||||
}
|
||||
defer release()
|
||||
d.staleLocks(ctx)
|
||||
}
|
||||
}
|
||||
|
||||
// runJanitor runs janitorPass every janitorInterval until ctx ends.
|
||||
func runJanitor(ctx context.Context, d janitorDeps) {
|
||||
d.logger.Info("janitor: starting (restore-test scratch retry + stale-lock sweep)", "interval", janitorInterval)
|
||||
t := time.NewTicker(janitorInterval)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
janitorPass(ctx, d)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/backup"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/reconcile"
|
||||
)
|
||||
|
||||
// R-672 / R-673 (v0.133.0): one janitor pass, driven with fakes.
|
||||
|
||||
func quiet() *slog.Logger { return slog.New(slog.NewTextHandler(io.Discard, nil)) }
|
||||
|
||||
// A scratch the engine gave up on reaches the hub as a failed restore-test record naming it — the
|
||||
// existing operator path (restore_test_failed). Never a pass.
|
||||
func TestJanitor_GaveUpIsReportedAsAFailure(t *testing.T) {
|
||||
var got []hub.RestoreTest
|
||||
janitorPass(context.Background(), janitorDeps{
|
||||
retryScratch: func(context.Context) reconcile.ScratchRetryResult {
|
||||
return reconcile.ScratchRetryResult{Examined: 1, Failed: 1, GaveUp: []int{990000}}
|
||||
},
|
||||
heavy: &backup.InFlight{}, record: func(r hub.RestoreTest) { got = append(got, r) },
|
||||
now: time.Now, logger: quiet(),
|
||||
})
|
||||
if len(got) != 1 || got[0].Pass || got[0].ScratchVMID != 990000 || !strings.Contains(got[0].Error, "990000") {
|
||||
t.Fatalf("records = %+v — want one FAILED record naming scratch 990000", got)
|
||||
}
|
||||
}
|
||||
|
||||
// R-673: the stale-lock sweep runs only while holding the one-heavy-operation gate, so no agent backup can
|
||||
// start between its "no vzdump running" check and its unlock.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT): drop the TryAcquire → "the sweep ran while a backup held the gate".
|
||||
func TestJanitor_StaleLockSweepWaitsForTheHeavyGate(t *testing.T) {
|
||||
heavy := &backup.InFlight{}
|
||||
swept := 0
|
||||
d := janitorDeps{staleLocks: func(context.Context) { swept++ }, heavy: heavy, now: time.Now, logger: quiet()}
|
||||
release, _, ok := heavy.TryAcquire("backup")
|
||||
if !ok {
|
||||
t.Fatal("setup")
|
||||
}
|
||||
janitorPass(context.Background(), d)
|
||||
if swept != 0 {
|
||||
t.Fatal("the sweep ran while a backup held the gate")
|
||||
}
|
||||
release()
|
||||
janitorPass(context.Background(), d)
|
||||
if swept != 1 {
|
||||
t.Fatalf("swept %d times with the gate free — want 1", swept)
|
||||
}
|
||||
if _, _, ok := heavy.TryAcquire("after"); !ok {
|
||||
t.Fatal("the sweep did not release the gate")
|
||||
}
|
||||
}
|
||||
@@ -50,6 +50,7 @@ import (
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/provision"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/reconcile"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/restorespace"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/selfheal"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/selfupdate"
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/signedjobs"
|
||||
@@ -897,6 +898,13 @@ func runDaemon(cfg config.Config, logger *slog.Logger, logRing *applog.Ring) int
|
||||
// it finds nothing to flag. The re-mount dispatch is off the poll path (a goroutine).
|
||||
storageTrigger := make(chan struct{}, 1)
|
||||
loop.SetTrigger(storageTrigger)
|
||||
// R-672: a thin pool crossing 90 % requests a report at once (the hub's storage-fill alarm).
|
||||
observer.SetThinHighTrigger(func() {
|
||||
select {
|
||||
case storageTrigger <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
})
|
||||
remounter := &gateRemounter{gate: gate, ops: hostOps, hostID: cfg.Hub.HostID, logger: logger}
|
||||
// Drive intent store (slice 10 P3 self-heal): persisted, durable-id-keyed enroll/eject/decommission
|
||||
// state. Gates the watchdog's self-heal re-mount to ENROLLED drives, and the local API records
|
||||
@@ -963,6 +971,7 @@ func runDaemon(cfg config.Config, logger *slog.Logger, logRing *applog.Ring) int
|
||||
Logger: logger,
|
||||
})
|
||||
|
||||
rtSpace, rtPolicy := restoreSpaceFor(cfg, px, hostOps)
|
||||
engine := reconcile.NewEngine(reconcile.EngineOptions{
|
||||
API: px,
|
||||
Queue: queue,
|
||||
@@ -971,6 +980,9 @@ func runDaemon(cfg config.Config, logger *slog.Logger, logRing *applog.Ring) int
|
||||
Gate: gate,
|
||||
HostID: cfg.Hub.HostID,
|
||||
Logger: logger,
|
||||
// R-672: the restore-test's space preflight (nil would refuse every test — fail-closed).
|
||||
RestoreSpace: rtSpace,
|
||||
SpacePolicy: rtPolicy,
|
||||
})
|
||||
|
||||
// Crash recovery (doc 03 §10): resolve any op that was in flight when the agent
|
||||
@@ -1407,6 +1419,16 @@ func runDaemon(cfg config.Config, logger *slog.Logger, logRing *applog.Ring) int
|
||||
go localSrv.WatchControllers(ctx)
|
||||
go func() { errc <- localSrv.Run(ctx) }()
|
||||
}
|
||||
// R-672 / R-673: retry a failed restore-test teardown and sweep stale backup locks on a timer, not
|
||||
// only at start-up (janitor.go).
|
||||
{
|
||||
jd := janitorDeps{retryScratch: engine.RetryScratchTeardown, heavy: heavyOps,
|
||||
record: backupStore.RecordRestoreTest, now: time.Now, logger: logger}
|
||||
if localSrv != nil {
|
||||
jd.staleLocks = localSrv.RecoverStaleLockedGuests
|
||||
}
|
||||
go runJanitor(ctx, jd)
|
||||
}
|
||||
if lanLoop != nil {
|
||||
lanServers = 1
|
||||
go func() { errc <- lanLoop.Run(ctx) }()
|
||||
@@ -2209,6 +2231,17 @@ func formatOrDash(t time.Time) string {
|
||||
return t.UTC().Format(time.RFC3339)
|
||||
}
|
||||
|
||||
// restoreSpaceFor builds the restore-test's space preflight (R-672): the production provider over the
|
||||
// Proxmox API + the privileged `lvs` metadata read, and the configured margin.
|
||||
func restoreSpaceFor(cfg config.Config, px *proxmox.Client, ops *storage.SudoHostOps) (reconcile.RestoreSpace, reconcile.SpacePolicy) {
|
||||
factor, reserve := cfg.Backup.RestoreTestSpace()
|
||||
p := &restorespace.Provider{API: px}
|
||||
if ops != nil {
|
||||
p.ThinMeta = ops.ThinPoolMetadata
|
||||
}
|
||||
return p, reconcile.SpacePolicy{Factor: factor, ReserveBytes: reserve}
|
||||
}
|
||||
|
||||
func runSelftestRestoreTest(ctx context.Context, cfg config.Config, logger *slog.Logger, archive string) int {
|
||||
if err := cfg.Validate(); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "selftest: proxmox not configured:", err)
|
||||
@@ -2239,8 +2272,10 @@ func runSelftestRestoreTest(ctx context.Context, cfg config.Config, logger *slog
|
||||
}
|
||||
}
|
||||
gate := reconcile.NewGate(nil, cfg.Hub.HostID, reconcile.SlogAudit{Logger: logger}, logger)
|
||||
rtSpace, rtPolicy := restoreSpaceFor(cfg, px, newHostOps(cfg, logger))
|
||||
engine := reconcile.NewEngine(reconcile.EngineOptions{
|
||||
API: px, Queue: queue, Journal: journal, Gate: gate, HostID: cfg.Hub.HostID, Logger: logger,
|
||||
RestoreSpace: rtSpace, SpacePolicy: rtPolicy,
|
||||
})
|
||||
|
||||
fmt.Printf("=== felhom-agent %s selftest=restore-test ===\n", version)
|
||||
@@ -2270,10 +2305,16 @@ func runSelftestRestoreTest(ctx context.Context, cfg config.Config, logger *slog
|
||||
RestoreTaskTimeout: restoreTaskTimeout(cfg, rtTier),
|
||||
})
|
||||
printJSON("restore-test record", backup.ToHubRestoreTest(res, time.Now().UTC()))
|
||||
if res.Skipped && res.SkipReason != "" {
|
||||
fmt.Printf(" space preflight: storage=%s required=%d avail=%d\n", res.TargetStorage, res.RequiredBytes, res.AvailBytes)
|
||||
fmt.Printf("=== selftest=restore-test SKIPPED — %s ===\n", res.SkipReason)
|
||||
return 4
|
||||
}
|
||||
if res.Skipped {
|
||||
fmt.Println("=== selftest=restore-test SKIPPED (no free scratch VMID in band) ===")
|
||||
return 0
|
||||
}
|
||||
fmt.Printf(" space preflight passed: storage=%s required=%d avail=%d\n", res.TargetStorage, res.RequiredBytes, res.AvailBytes)
|
||||
if res.Err != nil || !res.Pass {
|
||||
fmt.Fprintf(os.Stderr, " [FAIL] restore-test (scratch %d): %v\n", res.ScratchVMID, res.Err)
|
||||
return 1
|
||||
|
||||
Reference in New Issue
Block a user