Files
felhom-agent/internal/reconcile/restoretest_retry_test.go
T
admin 9bdb4dae8f
gates / gates (push) Successful in 13s
v0.133.0: a restore-test can never fill a box's disk; leftovers retried on a timer (R-672, R-673)
Space preflight before anything is created (uncompressed size from the vzdump log / PBS
snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage
is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and
the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin
pool crossing 90% requests an immediate host report. Six red-proofs.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-24 16:23:15 +02:00

74 lines
3.0 KiB
Go

package reconcile
import (
"context"
"errors"
"testing"
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
)
// R-672 rule 3 (v0.133.0): a failed scratch teardown is retried on a TIMER. The consequence asserted:
// the leaked scratch guest is destroyed by a timer pass (not only by a restart's Recover), the operator
// is told exactly once after MaxTeardownTries failures, and a vmid a running test owns is never touched.
// leakScratch runs a restore-test whose teardown fails, leaving scratch 990000 in-flight — the
// 2026-09-24 shape ("lvremove … contains a filesystem in use").
func leakScratch(t *testing.T) (*Engine, *fakeAPI, *Journal) {
t.Helper()
api := &fakeAPI{cfg: map[int]proxmox.GuestConfig{990000: scratchCfg()}, restoreUPID: "UPID:r", destroyErr: errors.New("lvremove: contains a filesystem in use")}
e, j := spaceEngine(t, api, roomySpace{})
e.RunRestoreTest(context.Background(), RestoreTestSpec{Archive: "local:backup/x.tar.zst", RestoreStorage: "local-lvm", ScratchMin: 990000, ScratchMax: 990009})
if len(j.InFlight()) != 1 {
t.Fatalf("setup: want the scratch left in-flight after a failed teardown, got %+v", j.InFlight())
}
api.lxc = []proxmox.Guest{{VMID: 990000}}
return e, api, j
}
// COMPANION RED-PROOF (REPORT): make RetryScratchTeardown return without touching the journal (the
// v0.132.0 shape — only Recover at start resolved a leak) → "the leaked scratch was not destroyed by
// the timer".
func TestRetry_TheTimerDestroysALeakedScratch(t *testing.T) {
e, api, j := leakScratch(t)
api.destroyErr = nil // the transient hold is gone
before := len(api.destroys)
r := e.RetryScratchTeardown(context.Background())
if r.Destroyed != 1 || len(api.destroys) != before+1 || api.destroys[len(api.destroys)-1] != 990000 {
t.Fatalf("the leaked scratch was not destroyed by the timer: result=%+v destroys=%v", r, api.destroys)
}
if len(j.InFlight()) != 0 {
t.Fatalf("the entry is still in flight after a successful retry: %+v", j.InFlight())
}
if r2 := e.RetryScratchTeardown(context.Background()); r2.Examined != 0 {
t.Fatalf("a resolved entry was examined again: %+v", r2)
}
}
func TestRetry_OperatorToldOnceAfterThreeFailures(t *testing.T) {
e, _, _ := leakScratch(t)
var gave [][]int
for i := 0; i < MaxTeardownTries+2; i++ {
gave = append(gave, e.RetryScratchTeardown(context.Background()).GaveUp)
}
for i, g := range gave {
want := 0
if i == MaxTeardownTries-1 {
want = 1
}
if len(g) != want {
t.Fatalf("pass %d gave up on %v — want the operator told exactly once, on pass %d", i+1, g, MaxTeardownTries)
}
}
}
func TestRetry_NeverTouchesARunningTest(t *testing.T) {
e, api, _ := leakScratch(t)
api.destroyErr = nil
e.markScratch(990000, true) // a restore-test is (again) working on this vmid
before := len(api.destroys)
if r := e.RetryScratchTeardown(context.Background()); r.Examined != 0 || len(api.destroys) != before {
t.Fatalf("the timer touched a scratch a running test owns: %+v destroys=%v", r, api.destroys)
}
}