Files
felhom-agent/internal/reconcile/restoretest_retry.go
T
admin 9bdb4dae8f
gates / gates (push) Successful in 13s
v0.133.0: a restore-test can never fill a box's disk; leftovers retried on a timer (R-672, R-673)
Space preflight before anything is created (uncompressed size from the vzdump log / PBS
snapshot, x1.2 + 5 GiB, thin metadata, off the tested guest's pool when another storage
is eligible, unknown refuses, reported as a non-pass result). Failed scratch teardown and
the stale-lock sweep retried every 10 min (the sweep under the heavy-op gate). A thin
pool crossing 90% requests an immediate host report. Six red-proofs.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-24 16:23:15 +02:00

95 lines
3.1 KiB
Go

package reconcile
import "context"
// ── A failed scratch teardown is retried on a TIMER, not only at agent start (R-672 rule 3) ────────
//
// MEASURED 2026-09-24 on demo-hp: the scheduled restore-test's teardown failed (`lvremove … contains a
// filesystem in use`, a transient hold) and logged "left for Recover" — and Recover runs ONLY at agent
// start, so the 22 GiB scratch guest sat in the full pool for 2.5 hours until an agent restart. The
// timer calls RetryScratchTeardown every 10 minutes: the SAME resolution as Recover (recoverScratch —
// the gate's benign scratch destroy, idempotent when the guest is already gone), restricted to Scratch
// entries that carry a launch-proof UPID and that no running restore-test owns. After
// MaxTeardownTries failed attempts for one entry the operator is told (the caller reports it); the
// timer keeps trying.
// MaxTeardownTries is how many failed timer retries of one scratch entry happen before the operator
// is told.
const MaxTeardownTries = 3
// ScratchRetryResult summarizes one timer pass.
type ScratchRetryResult struct {
Examined int
Destroyed int
Clean int // already gone
Failed int
// GaveUp lists the scratch vmids whose failed tries reached MaxTeardownTries IN THIS PASS — each
// is reported exactly once (the caller tells the operator).
GaveUp []int
}
func (e *Engine) markScratch(vmid int, active bool) {
e.scratchMu.Lock()
defer e.scratchMu.Unlock()
if active {
e.activeScratch[vmid] = true
} else {
delete(e.activeScratch, vmid)
}
}
func (e *Engine) scratchActive(vmid int) bool {
e.scratchMu.Lock()
defer e.scratchMu.Unlock()
return e.activeScratch[vmid]
}
// RetryScratchTeardown is the timer's pass. It never touches a non-Scratch entry (unlike Recover,
// which also resolves generic in-flight operations and must therefore run only at start), never an
// entry without a launch-proof UPID (nothing was created), and never a vmid a running test owns.
func (e *Engine) RetryScratchTeardown(ctx context.Context) ScratchRetryResult {
var out ScratchRetryResult
if e.journal == nil {
return out
}
for _, entry := range e.journal.InFlight() {
if !entry.Scratch || entry.UPID == "" || e.scratchActive(entry.VMID) {
continue
}
out.Examined++
var r RecoverResult
e.recoverScratch(ctx, entry, &r)
switch {
case r.ScratchDestroyed > 0:
out.Destroyed++
e.forgetTries(entry.OpID)
case r.ScratchClean > 0:
out.Clean++
e.forgetTries(entry.OpID)
default:
out.Failed++
n := e.addTry(entry.OpID)
e.logger.Warn("restore-test: scratch teardown retry failed (timer)", "vmid", entry.VMID, "op_id", entry.OpID, "try", n)
if n == MaxTeardownTries {
out.GaveUp = append(out.GaveUp, entry.VMID)
e.logger.Error("restore-test: scratch guest still NOT torn down after repeated retries — telling the operator",
"vmid", entry.VMID, "tries", n)
}
}
}
return out
}
func (e *Engine) addTry(op string) int {
e.scratchMu.Lock()
defer e.scratchMu.Unlock()
e.teardownTries[op]++
return e.teardownTries[op]
}
func (e *Engine) forgetTries(op string) {
e.scratchMu.Lock()
defer e.scratchMu.Unlock()
delete(e.teardownTries, op)
}