9bdb4dae8f
gates / gates (push) Successful in 13s
Space preflight before anything is created (uncompressed size from the vzdump log / PBS snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin pool crossing 90% requests an immediate host report. Six red-proofs. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
95 lines
3.1 KiB
Go
95 lines
3.1 KiB
Go
package reconcile
|
|
|
|
import "context"
|
|
|
|
// ── A failed scratch teardown is retried on a TIMER, not only at agent start (R-672 rule 3) ────────
|
|
//
|
|
// MEASURED 2026-09-24 on demo-hp: the scheduled restore-test's teardown failed (`lvremove … contains a
|
|
// filesystem in use`, a transient hold) and logged "left for Recover" — and Recover runs ONLY at agent
|
|
// start, so the 22 GiB scratch guest sat in the full pool for 2.5 hours until an agent restart. The
|
|
// timer calls RetryScratchTeardown every 10 minutes: the SAME resolution as Recover (recoverScratch —
|
|
// the gate's benign scratch destroy, idempotent when the guest is already gone), restricted to Scratch
|
|
// entries that carry a launch-proof UPID and that no running restore-test owns. After
|
|
// MaxTeardownTries failed attempts for one entry the operator is told (the caller reports it); the
|
|
// timer keeps trying.
|
|
|
|
// MaxTeardownTries is how many failed timer retries of one scratch entry happen before the operator
|
|
// is told.
|
|
const MaxTeardownTries = 3
|
|
|
|
// ScratchRetryResult summarizes one timer pass.
|
|
type ScratchRetryResult struct {
|
|
Examined int
|
|
Destroyed int
|
|
Clean int // already gone
|
|
Failed int
|
|
// GaveUp lists the scratch vmids whose failed tries reached MaxTeardownTries IN THIS PASS — each
|
|
// is reported exactly once (the caller tells the operator).
|
|
GaveUp []int
|
|
}
|
|
|
|
func (e *Engine) markScratch(vmid int, active bool) {
|
|
e.scratchMu.Lock()
|
|
defer e.scratchMu.Unlock()
|
|
if active {
|
|
e.activeScratch[vmid] = true
|
|
} else {
|
|
delete(e.activeScratch, vmid)
|
|
}
|
|
}
|
|
|
|
func (e *Engine) scratchActive(vmid int) bool {
|
|
e.scratchMu.Lock()
|
|
defer e.scratchMu.Unlock()
|
|
return e.activeScratch[vmid]
|
|
}
|
|
|
|
// RetryScratchTeardown is the timer's pass. It never touches a non-Scratch entry (unlike Recover,
|
|
// which also resolves generic in-flight operations and must therefore run only at start), never an
|
|
// entry without a launch-proof UPID (nothing was created), and never a vmid a running test owns.
|
|
func (e *Engine) RetryScratchTeardown(ctx context.Context) ScratchRetryResult {
|
|
var out ScratchRetryResult
|
|
if e.journal == nil {
|
|
return out
|
|
}
|
|
for _, entry := range e.journal.InFlight() {
|
|
if !entry.Scratch || entry.UPID == "" || e.scratchActive(entry.VMID) {
|
|
continue
|
|
}
|
|
out.Examined++
|
|
var r RecoverResult
|
|
e.recoverScratch(ctx, entry, &r)
|
|
switch {
|
|
case r.ScratchDestroyed > 0:
|
|
out.Destroyed++
|
|
e.forgetTries(entry.OpID)
|
|
case r.ScratchClean > 0:
|
|
out.Clean++
|
|
e.forgetTries(entry.OpID)
|
|
default:
|
|
out.Failed++
|
|
n := e.addTry(entry.OpID)
|
|
e.logger.Warn("restore-test: scratch teardown retry failed (timer)", "vmid", entry.VMID, "op_id", entry.OpID, "try", n)
|
|
if n == MaxTeardownTries {
|
|
out.GaveUp = append(out.GaveUp, entry.VMID)
|
|
e.logger.Error("restore-test: scratch guest still NOT torn down after repeated retries — telling the operator",
|
|
"vmid", entry.VMID, "tries", n)
|
|
}
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
func (e *Engine) addTry(op string) int {
|
|
e.scratchMu.Lock()
|
|
defer e.scratchMu.Unlock()
|
|
e.teardownTries[op]++
|
|
return e.teardownTries[op]
|
|
}
|
|
|
|
func (e *Engine) forgetTries(op string) {
|
|
e.scratchMu.Lock()
|
|
defer e.scratchMu.Unlock()
|
|
delete(e.teardownTries, op)
|
|
}
|