v0.133.0: a restore-test can never fill a box's disk; leftovers retried on a timer (R-672, R-673)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
Space preflight before anything is created (uncompressed size from the vzdump log / PBS snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin pool crossing 90% requests an immediate host report. Six red-proofs. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,94 @@
|
||||
package reconcile
|
||||
|
||||
import "context"
|
||||
|
||||
// ── A failed scratch teardown is retried on a TIMER, not only at agent start (R-672 rule 3) ────────
|
||||
//
|
||||
// MEASURED 2026-09-24 on demo-hp: the scheduled restore-test's teardown failed (`lvremove … contains a
|
||||
// filesystem in use`, a transient hold) and logged "left for Recover" — and Recover runs ONLY at agent
|
||||
// start, so the 22 GiB scratch guest sat in the full pool for 2.5 hours until an agent restart. The
|
||||
// timer calls RetryScratchTeardown every 10 minutes: the SAME resolution as Recover (recoverScratch —
|
||||
// the gate's benign scratch destroy, idempotent when the guest is already gone), restricted to Scratch
|
||||
// entries that carry a launch-proof UPID and that no running restore-test owns. After
|
||||
// MaxTeardownTries failed attempts for one entry the operator is told (the caller reports it); the
|
||||
// timer keeps trying.
|
||||
|
||||
// MaxTeardownTries is how many failed timer retries of one scratch entry happen before the operator
|
||||
// is told.
|
||||
const MaxTeardownTries = 3
|
||||
|
||||
// ScratchRetryResult summarizes one timer pass.
|
||||
type ScratchRetryResult struct {
|
||||
Examined int
|
||||
Destroyed int
|
||||
Clean int // already gone
|
||||
Failed int
|
||||
// GaveUp lists the scratch vmids whose failed tries reached MaxTeardownTries IN THIS PASS — each
|
||||
// is reported exactly once (the caller tells the operator).
|
||||
GaveUp []int
|
||||
}
|
||||
|
||||
func (e *Engine) markScratch(vmid int, active bool) {
|
||||
e.scratchMu.Lock()
|
||||
defer e.scratchMu.Unlock()
|
||||
if active {
|
||||
e.activeScratch[vmid] = true
|
||||
} else {
|
||||
delete(e.activeScratch, vmid)
|
||||
}
|
||||
}
|
||||
|
||||
func (e *Engine) scratchActive(vmid int) bool {
|
||||
e.scratchMu.Lock()
|
||||
defer e.scratchMu.Unlock()
|
||||
return e.activeScratch[vmid]
|
||||
}
|
||||
|
||||
// RetryScratchTeardown is the timer's pass. It never touches a non-Scratch entry (unlike Recover,
|
||||
// which also resolves generic in-flight operations and must therefore run only at start), never an
|
||||
// entry without a launch-proof UPID (nothing was created), and never a vmid a running test owns.
|
||||
func (e *Engine) RetryScratchTeardown(ctx context.Context) ScratchRetryResult {
|
||||
var out ScratchRetryResult
|
||||
if e.journal == nil {
|
||||
return out
|
||||
}
|
||||
for _, entry := range e.journal.InFlight() {
|
||||
if !entry.Scratch || entry.UPID == "" || e.scratchActive(entry.VMID) {
|
||||
continue
|
||||
}
|
||||
out.Examined++
|
||||
var r RecoverResult
|
||||
e.recoverScratch(ctx, entry, &r)
|
||||
switch {
|
||||
case r.ScratchDestroyed > 0:
|
||||
out.Destroyed++
|
||||
e.forgetTries(entry.OpID)
|
||||
case r.ScratchClean > 0:
|
||||
out.Clean++
|
||||
e.forgetTries(entry.OpID)
|
||||
default:
|
||||
out.Failed++
|
||||
n := e.addTry(entry.OpID)
|
||||
e.logger.Warn("restore-test: scratch teardown retry failed (timer)", "vmid", entry.VMID, "op_id", entry.OpID, "try", n)
|
||||
if n == MaxTeardownTries {
|
||||
out.GaveUp = append(out.GaveUp, entry.VMID)
|
||||
e.logger.Error("restore-test: scratch guest still NOT torn down after repeated retries — telling the operator",
|
||||
"vmid", entry.VMID, "tries", n)
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (e *Engine) addTry(op string) int {
|
||||
e.scratchMu.Lock()
|
||||
defer e.scratchMu.Unlock()
|
||||
e.teardownTries[op]++
|
||||
return e.teardownTries[op]
|
||||
}
|
||||
|
||||
func (e *Engine) forgetTries(op string) {
|
||||
e.scratchMu.Lock()
|
||||
defer e.scratchMu.Unlock()
|
||||
delete(e.teardownTries, op)
|
||||
}
|
||||
Reference in New Issue
Block a user