F-LEAK: remove the pool-adoption fix — refuted live; the fix is a path-scoped ACL (v0.108.0)

PUT /pools/{pool} ALSO requires VM.Allocate on the VM being added, so Pool.Allocate
cannot bootstrap its own membership. Proven live on demo-hp 2026-07-28. The real fix is
felhom-host-install v1.21.0 granting FelhomAgentGuest at /vms/990000..990009.
This commit is contained in:
2026-07-28 11:05:37 +02:00
parent 367a503a0f
commit 8db92947cd
3 changed files with 40 additions and 130 deletions
+11 -59
View File
@@ -198,7 +198,7 @@ func (e *Engine) runScratchTest(ctx context.Context, vmid int, spec RestoreTestS
launched := false
defer func() {
if launched {
e.teardownScratch(ctx, base, spec.ScratchMin, spec.ScratchMax)
e.teardownScratch(ctx, base)
return
}
e.append(withState(base, OpFailed))
@@ -444,44 +444,9 @@ func sizeToGB(s string) int {
return gb
}
// scratchAdoptAllowed decides whether a stranded guest may be adopted into the felhom pool so the
// pool-scoped token can destroy it (F-LEAK). PURE, so the refusal is unit-testable without PVE.
//
// TWO INDEPENDENT GUARDS, both required. This function is the only thing standing between "clean up
// my own scratch" and "co-opt an arbitrary guest into the pool and delete it", so it does not rely
// on either check alone:
// - PROVENANCE: the journal entry must be one the agent itself created as a restore-test scratch.
// - NUMERIC BAND: the VMID must be inside the configured scratch band (scratch_vmid_min..max).
//
// A guest failing either is refused, loudly. Adopting a customer guest into the pool would hand the
// token destroy rights over it, which is a far worse outcome than a leaked scratch.
func scratchAdoptAllowed(vmid, min, max int, scratch bool) (bool, string) {
if !scratch {
return false, "journal entry is not agent-created scratch provenance"
}
if min <= 0 || max < min {
return false, fmt.Sprintf("scratch band [%d,%d] is not configured", min, max)
}
if vmid < min || vmid > max {
return false, fmt.Sprintf("vmid %d is outside the scratch band [%d,%d]", vmid, min, max)
}
return true, ""
}
// teardownScratch destroys the scratch guest (benign, gated) and records the entry terminal.
// On any teardown failure it leaves the entry in-flight so Recover reaps the guest later.
//
// F-LEAK (Campaign 8): a restore-test whose RESTORE FAILED left a scratch guest the agent could not
// destroy — `DELETE /nodes/x/lxc/990000` returned 403 "missing privilege VM.Allocate". The cause is
// pool membership, not privsep: VM.Allocate is granted at /pool/felhom ONLY (never at /), and a
// failed restore never completes the `--pool felhom` association, so the guest's own path resolves
// to / where the token holds nothing. Verified live: /vms/<non-member> grants only
// Datastore.Audit+SDN.Use+Sys.Audit, while /pool/felhom grants VM.Allocate AND Pool.Allocate.
//
// So the recovery needs NO new privilege: Pool.Allocate is already held, so we adopt the stranded
// scratch into the pool and retry the destroy, which then authorizes via /pool/felhom. Guarded by
// scratchAdoptAllowed — see there for why two guards rather than one.
func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry, scratchMin, scratchMax int) {
func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry) {
// Cancel-immune + bounded, so a shutdown mid-test still tears down.
tctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 2*time.Minute)
defer cancel()
@@ -494,28 +459,15 @@ func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry, scratch
}
upid, err := e.api.DestroyLXC(tctx, base.VMID)
if err != nil {
// F-LEAK: the destroy may have 403'd because a FAILED restore never completed pool
// membership, leaving the guest outside /pool/felhom where the token's VM.Allocate lives.
// Adopt it into the pool (Pool.Allocate, already granted) and retry ONCE. Any other error
// falls through to the original behaviour.
if ok, why := scratchAdoptAllowed(base.VMID, scratchMin, scratchMax, base.Scratch); !ok {
e.logger.Error("restore-test: scratch teardown failed and adoption REFUSED; left for Recover",
"vmid", base.VMID, "refused_because", why, "err", err)
return
}
e.logger.Warn("restore-test: scratch teardown failed — adopting the stranded scratch into the pool and retrying once",
"vmid", base.VMID, "pool", DefaultPool, "err", err)
if perr := e.api.PoolAddVMID(tctx, DefaultPool, base.VMID); perr != nil {
e.logger.Error("restore-test: pool adoption failed; left for Recover", "vmid", base.VMID, "err", perr)
return
}
upid, err = e.api.DestroyLXC(tctx, base.VMID)
if err != nil {
e.logger.Error("restore-test: scratch teardown failed even after pool adoption; left for Recover",
"vmid", base.VMID, "err", err)
return
}
e.logger.Info("restore-test: stranded scratch adopted into the pool and destroyed", "vmid", base.VMID)
// F-LEAK (Campaign 8): a 403 "missing privilege VM.Allocate" here means this scratch is not a
// felhom-pool member — a FAILED restore never completes the `--pool` association, and the
// token's VM.Allocate is granted at /pool/felhom, never at /. Fixed by GRANTING VM.Allocate on
// the scratch VMID band itself (host-install v1.21.0), path-scoped so the agent still cannot
// reach a non-scratch guest. NOT fixable from here: adopting the guest into the pool was tried
// and refused — `PUT /pools/{pool}` ALSO requires VM.Allocate on the VM being added, so
// Pool.Allocate alone cannot bootstrap membership (proven live 2026-07-28).
e.logger.Error("restore-test: scratch teardown failed; left for Recover", "vmid", base.VMID, "err", err)
return
}
if upid != "" {
if _, err := e.api.WaitTask(tctx, upid, proxmox.WaitOptions{}); err != nil {