package reconcile import "context" // ── A failed scratch teardown is retried on a TIMER, not only at agent start (R-672 rule 3) ──────── // // MEASURED 2026-09-24 on demo-hp: the scheduled restore-test's teardown failed (`lvremove … contains a // filesystem in use`, a transient hold) and logged "left for Recover" — and Recover runs ONLY at agent // start, so the 22 GiB scratch guest sat in the full pool for 2.5 hours until an agent restart. The // timer calls RetryScratchTeardown every 10 minutes: the SAME resolution as Recover (recoverScratch — // the gate's benign scratch destroy, idempotent when the guest is already gone), restricted to Scratch // entries that carry a launch-proof UPID and that no running restore-test owns. After // MaxTeardownTries failed attempts for one entry the operator is told (the caller reports it); the // timer keeps trying. // MaxTeardownTries is how many failed timer retries of one scratch entry happen before the operator // is told. const MaxTeardownTries = 3 // ScratchRetryResult summarizes one timer pass. type ScratchRetryResult struct { Examined int Destroyed int Clean int // already gone Failed int // GaveUp lists the scratch vmids whose failed tries reached MaxTeardownTries IN THIS PASS — each // is reported exactly once (the caller tells the operator). GaveUp []int } func (e *Engine) markScratch(vmid int, active bool) { e.scratchMu.Lock() defer e.scratchMu.Unlock() if active { e.activeScratch[vmid] = true } else { delete(e.activeScratch, vmid) } } func (e *Engine) scratchActive(vmid int) bool { e.scratchMu.Lock() defer e.scratchMu.Unlock() return e.activeScratch[vmid] } // RetryScratchTeardown is the timer's pass. It never touches a non-Scratch entry (unlike Recover, // which also resolves generic in-flight operations and must therefore run only at start), never an // entry without a launch-proof UPID (nothing was created), and never a vmid a running test owns. func (e *Engine) RetryScratchTeardown(ctx context.Context) ScratchRetryResult { var out ScratchRetryResult if e.journal == nil { return out } for _, entry := range e.journal.InFlight() { if !entry.Scratch || entry.UPID == "" || e.scratchActive(entry.VMID) { continue } out.Examined++ var r RecoverResult e.recoverScratch(ctx, entry, &r) switch { case r.ScratchDestroyed > 0: out.Destroyed++ e.forgetTries(entry.OpID) case r.ScratchClean > 0: out.Clean++ e.forgetTries(entry.OpID) default: out.Failed++ n := e.addTry(entry.OpID) e.logger.Warn("restore-test: scratch teardown retry failed (timer)", "vmid", entry.VMID, "op_id", entry.OpID, "try", n) if n == MaxTeardownTries { out.GaveUp = append(out.GaveUp, entry.VMID) e.logger.Error("restore-test: scratch guest still NOT torn down after repeated retries — telling the operator", "vmid", entry.VMID, "tries", n) } } } return out } func (e *Engine) addTry(op string) int { e.scratchMu.Lock() defer e.scratchMu.Unlock() e.teardownTries[op]++ return e.teardownTries[op] } func (e *Engine) forgetTries(op string) { e.scratchMu.Lock() defer e.scratchMu.Unlock() delete(e.teardownTries, op) }