9bdb4dae8f
gates / gates (push) Successful in 13s
Space preflight before anything is created (uncompressed size from the vzdump log / PBS snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin pool crossing 90% requests an immediate host report. Six red-proofs. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
74 lines
3.0 KiB
Go
74 lines
3.0 KiB
Go
package reconcile
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"testing"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
|
|
)
|
|
|
|
// R-672 rule 3 (v0.133.0): a failed scratch teardown is retried on a TIMER. The consequence asserted:
|
|
// the leaked scratch guest is destroyed by a timer pass (not only by a restart's Recover), the operator
|
|
// is told exactly once after MaxTeardownTries failures, and a vmid a running test owns is never touched.
|
|
|
|
// leakScratch runs a restore-test whose teardown fails, leaving scratch 990000 in-flight — the
|
|
// 2026-09-24 shape ("lvremove … contains a filesystem in use").
|
|
func leakScratch(t *testing.T) (*Engine, *fakeAPI, *Journal) {
|
|
t.Helper()
|
|
api := &fakeAPI{cfg: map[int]proxmox.GuestConfig{990000: scratchCfg()}, restoreUPID: "UPID:r", destroyErr: errors.New("lvremove: contains a filesystem in use")}
|
|
e, j := spaceEngine(t, api, roomySpace{})
|
|
e.RunRestoreTest(context.Background(), RestoreTestSpec{Archive: "local:backup/x.tar.zst", RestoreStorage: "local-lvm", ScratchMin: 990000, ScratchMax: 990009})
|
|
if len(j.InFlight()) != 1 {
|
|
t.Fatalf("setup: want the scratch left in-flight after a failed teardown, got %+v", j.InFlight())
|
|
}
|
|
api.lxc = []proxmox.Guest{{VMID: 990000}}
|
|
return e, api, j
|
|
}
|
|
|
|
// COMPANION RED-PROOF (REPORT): make RetryScratchTeardown return without touching the journal (the
|
|
// v0.132.0 shape — only Recover at start resolved a leak) → "the leaked scratch was not destroyed by
|
|
// the timer".
|
|
func TestRetry_TheTimerDestroysALeakedScratch(t *testing.T) {
|
|
e, api, j := leakScratch(t)
|
|
api.destroyErr = nil // the transient hold is gone
|
|
before := len(api.destroys)
|
|
r := e.RetryScratchTeardown(context.Background())
|
|
if r.Destroyed != 1 || len(api.destroys) != before+1 || api.destroys[len(api.destroys)-1] != 990000 {
|
|
t.Fatalf("the leaked scratch was not destroyed by the timer: result=%+v destroys=%v", r, api.destroys)
|
|
}
|
|
if len(j.InFlight()) != 0 {
|
|
t.Fatalf("the entry is still in flight after a successful retry: %+v", j.InFlight())
|
|
}
|
|
if r2 := e.RetryScratchTeardown(context.Background()); r2.Examined != 0 {
|
|
t.Fatalf("a resolved entry was examined again: %+v", r2)
|
|
}
|
|
}
|
|
|
|
func TestRetry_OperatorToldOnceAfterThreeFailures(t *testing.T) {
|
|
e, _, _ := leakScratch(t)
|
|
var gave [][]int
|
|
for i := 0; i < MaxTeardownTries+2; i++ {
|
|
gave = append(gave, e.RetryScratchTeardown(context.Background()).GaveUp)
|
|
}
|
|
for i, g := range gave {
|
|
want := 0
|
|
if i == MaxTeardownTries-1 {
|
|
want = 1
|
|
}
|
|
if len(g) != want {
|
|
t.Fatalf("pass %d gave up on %v — want the operator told exactly once, on pass %d", i+1, g, MaxTeardownTries)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestRetry_NeverTouchesARunningTest(t *testing.T) {
|
|
e, api, _ := leakScratch(t)
|
|
api.destroyErr = nil
|
|
e.markScratch(990000, true) // a restore-test is (again) working on this vmid
|
|
before := len(api.destroys)
|
|
if r := e.RetryScratchTeardown(context.Background()); r.Examined != 0 || len(api.destroys) != before {
|
|
t.Fatalf("the timer touched a scratch a running test owns: %+v destroys=%v", r, api.destroys)
|
|
}
|
|
}
|