F-REBOOT + F-LEAK: the agent's authority over guest lifecycle (v0.107.0)

F-REBOOT — a guest rebooted mid-backup never came back (fault 11: 9m47s of total
appliance outage, no lock, nothing retrying). The existing stale-lock recovery is
correct but missed it two ways: its predicate needs a stale vzdump lock and that
guest was unlocked, and it runs only at agent startup. New periodic guest-power
watchdog acts on 'should be running, is not, is not locked'.

onboot is the should-be-running signal, not invented here: stalelock.go already
uses it for this same decision, it is 0 on scratch/golden, and pve-guests uses it
at host boot. Guards: onboot:0 never touched (Scenario B), a locked guest is left
to the stale-lock path, a guest with a vzdump in flight is left stopped,
unprovable ownership acts on nothing, unconfirmable backup state fails safe.
Bounded retry 3x at 1/2/4m then ERROR (Scenario C) — a healthy start takes ~25s.

F-LEAK — a failed restore-test could not destroy its scratch (403 VM.Allocate).
It is pool membership, not privsep: VM.Allocate is granted at /pool/felhom only,
and a failed restore never completes the --pool association. Fix needs NO new
grant — Pool.Allocate is already held, so the teardown adopts the stranded
scratch into the pool and retries the destroy. Guarded by scratchAdoptAllowed:
scratch provenance AND the numeric band, both required (Scenario E).

Six red-proofs across both fixes, all observed failing.
This commit is contained in:
2026-07-28 10:27:07 +02:00
parent a18b18e5de
commit 367a503a0f
7 changed files with 691 additions and 5 deletions
+59 -4
View File
@@ -198,7 +198,7 @@ func (e *Engine) runScratchTest(ctx context.Context, vmid int, spec RestoreTestS
launched := false
defer func() {
if launched {
e.teardownScratch(ctx, base)
e.teardownScratch(ctx, base, spec.ScratchMin, spec.ScratchMax)
return
}
e.append(withState(base, OpFailed))
@@ -444,9 +444,44 @@ func sizeToGB(s string) int {
return gb
}
// scratchAdoptAllowed decides whether a stranded guest may be adopted into the felhom pool so the
// pool-scoped token can destroy it (F-LEAK). PURE, so the refusal is unit-testable without PVE.
//
// TWO INDEPENDENT GUARDS, both required. This function is the only thing standing between "clean up
// my own scratch" and "co-opt an arbitrary guest into the pool and delete it", so it does not rely
// on either check alone:
// - PROVENANCE: the journal entry must be one the agent itself created as a restore-test scratch.
// - NUMERIC BAND: the VMID must be inside the configured scratch band (scratch_vmid_min..max).
//
// A guest failing either is refused, loudly. Adopting a customer guest into the pool would hand the
// token destroy rights over it, which is a far worse outcome than a leaked scratch.
func scratchAdoptAllowed(vmid, min, max int, scratch bool) (bool, string) {
if !scratch {
return false, "journal entry is not agent-created scratch provenance"
}
if min <= 0 || max < min {
return false, fmt.Sprintf("scratch band [%d,%d] is not configured", min, max)
}
if vmid < min || vmid > max {
return false, fmt.Sprintf("vmid %d is outside the scratch band [%d,%d]", vmid, min, max)
}
return true, ""
}
// teardownScratch destroys the scratch guest (benign, gated) and records the entry terminal.
// On any teardown failure it leaves the entry in-flight so Recover reaps the guest later.
func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry) {
//
// F-LEAK (Campaign 8): a restore-test whose RESTORE FAILED left a scratch guest the agent could not
// destroy — `DELETE /nodes/x/lxc/990000` returned 403 "missing privilege VM.Allocate". The cause is
// pool membership, not privsep: VM.Allocate is granted at /pool/felhom ONLY (never at /), and a
// failed restore never completes the `--pool felhom` association, so the guest's own path resolves
// to / where the token holds nothing. Verified live: /vms/<non-member> grants only
// Datastore.Audit+SDN.Use+Sys.Audit, while /pool/felhom grants VM.Allocate AND Pool.Allocate.
//
// So the recovery needs NO new privilege: Pool.Allocate is already held, so we adopt the stranded
// scratch into the pool and retry the destroy, which then authorizes via /pool/felhom. Guarded by
// scratchAdoptAllowed — see there for why two guards rather than one.
func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry, scratchMin, scratchMax int) {
// Cancel-immune + bounded, so a shutdown mid-test still tears down.
tctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 2*time.Minute)
defer cancel()
@@ -459,8 +494,28 @@ func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry) {
}
upid, err := e.api.DestroyLXC(tctx, base.VMID)
if err != nil {
e.logger.Error("restore-test: scratch teardown failed; left for Recover", "vmid", base.VMID, "err", err)
return
// F-LEAK: the destroy may have 403'd because a FAILED restore never completed pool
// membership, leaving the guest outside /pool/felhom where the token's VM.Allocate lives.
// Adopt it into the pool (Pool.Allocate, already granted) and retry ONCE. Any other error
// falls through to the original behaviour.
if ok, why := scratchAdoptAllowed(base.VMID, scratchMin, scratchMax, base.Scratch); !ok {
e.logger.Error("restore-test: scratch teardown failed and adoption REFUSED; left for Recover",
"vmid", base.VMID, "refused_because", why, "err", err)
return
}
e.logger.Warn("restore-test: scratch teardown failed — adopting the stranded scratch into the pool and retrying once",
"vmid", base.VMID, "pool", DefaultPool, "err", err)
if perr := e.api.PoolAddVMID(tctx, DefaultPool, base.VMID); perr != nil {
e.logger.Error("restore-test: pool adoption failed; left for Recover", "vmid", base.VMID, "err", perr)
return
}
upid, err = e.api.DestroyLXC(tctx, base.VMID)
if err != nil {
e.logger.Error("restore-test: scratch teardown failed even after pool adoption; left for Recover",
"vmid", base.VMID, "err", err)
return
}
e.logger.Info("restore-test: stranded scratch adopted into the pool and destroyed", "vmid", base.VMID)
}
if upid != "" {
if _, err := e.api.WaitTask(tctx, upid, proxmox.WaitOptions{}); err != nil {
+71
View File
@@ -0,0 +1,71 @@
package reconcile
import "testing"
// F-LEAK (Campaign 8): a restore-test whose RESTORE FAILED could not destroy its own scratch guest —
// `DELETE /nodes/x/lxc/990000` → 403 "missing privilege VM.Allocate". The cause is pool membership,
// not privsep: VM.Allocate is granted at /pool/felhom only, and a failed restore never completes the
// `--pool felhom` association, so the guest's own path resolves to / where the token holds nothing.
//
// The fix adopts the stranded scratch into the pool (Pool.Allocate — already granted) and retries the
// destroy. scratchAdoptAllowed is the guard that keeps that from becoming "co-opt any guest into the
// pool and delete it", and this file is its mirror test.
//
// Scenario D (the scratch is destroyed) is the live replay; Scenario E — the agent still cannot reach
// a non-scratch guest this way — is HERE, because it must hold as a pure property and not depend on
// what PVE happens to refuse.
// Scenario E — the adoption path REFUSES anything outside the scratch band.
//
// RED-PROOF: replace the band check with `return true, ""` → every case below reports allowed and the
// test fails with "adoption ALLOWED for vmid 9201 — that would let the agent co-opt a customer guest
// into the pool and destroy it".
func TestScratchAdoptAllowed_RefusesNonScratch(t *testing.T) {
const min, max = 990000, 990009
for _, tc := range []struct {
name string
vmid int
scratch bool
}{
{"a live customer guest", 9201, true}, // scratch-flagged but OUTSIDE the band
{"the golden image", 9100, true}, // ditto
{"just below the band", min - 1, true}, // off-by-one
{"just above the band", max + 1, true}, // off-by-one
{"in-band but NOT scratch provenance", min, false},
{"neither", 100, false},
} {
t.Run(tc.name, func(t *testing.T) {
ok, why := scratchAdoptAllowed(tc.vmid, min, max, tc.scratch)
if ok {
t.Errorf("adoption ALLOWED for vmid %d — that would let the agent co-opt a non-scratch guest into the pool and destroy it", tc.vmid)
}
if why == "" {
t.Error("refusal carried no reason — a silent refusal is unreviewable")
}
})
}
}
// The in-band, scratch-provenanced case IS allowed — otherwise the fix does nothing and the leak stays.
//
// RED-PROOF: make scratchAdoptAllowed always return false → this fails with "adoption refused for a
// genuine in-band scratch guest", i.e. F-LEAK is not fixed at all.
func TestScratchAdoptAllowed_AllowsGenuineScratch(t *testing.T) {
const min, max = 990000, 990009
for _, vmid := range []int{min, min + 5, max} {
ok, why := scratchAdoptAllowed(vmid, min, max, true)
if !ok {
t.Errorf("adoption refused for a genuine in-band scratch guest %d: %s", vmid, why)
}
}
}
// An unconfigured band must refuse everything rather than defaulting to something permissive — a
// zero-valued band is a wiring bug, and the safe reading of a wiring bug is "do nothing".
func TestScratchAdoptAllowed_UnconfiguredBandRefuses(t *testing.T) {
for _, tc := range []struct{ min, max int }{{0, 0}, {0, 990009}, {990009, 990000}, {-1, 5}} {
if ok, _ := scratchAdoptAllowed(990000, tc.min, tc.max, true); ok {
t.Errorf("adoption allowed with an unconfigured band [%d,%d]", tc.min, tc.max)
}
}
}