b9356d60ab
Campaign pool-effects F1 (HIGH): the bring-up compensating rollback and the restore-test teardown destroyed the target vmid even when RestoreLXC failed synchronously without creating anything — destroying a guest the transaction never made (only the pool ACL 403 contained it). A RestoreLXC UPID is now the sole destroy authorization in all three destroy paths (in-process bring-up defer, in-process restore-test teardown, Recover). F2: the restore-test advances past an 'already exists' band vmid (invisible squatter) instead of failing + false-alerting; a fully-occupied band Skips. Red-proof verified: with the gates reverted, the four new tests fail with the innocent-guest destroy. go build/vet/test clean. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
516 lines
21 KiB
Go
516 lines
21 KiB
Go
package reconcile
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"math"
|
|
"regexp"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
|
|
)
|
|
|
|
// The self-restore-test (doc 03 §8) — the piece that closes "a backup you haven't restored
|
|
// isn't a backup". It is a JOURNALED reconcile job so it inherits the slice-4 journal,
|
|
// per-guest serialization, and crash-safe recovery: a mid-test crash can't leak a scratch
|
|
// guest (engine.Recover tears it down). Every step here is BENIGN — restore-to-new
|
|
// (ClassCreate), a benign net-link-down SetConfig, and a scratch teardown that is benign by
|
|
// agent-tagged-scratch provenance (no new destructive class, no new crypto).
|
|
|
|
// scratchKind is the journal Kind for a restore-test scratch-guest-owning entry. Recover
|
|
// keys off JournalEntry.Scratch (not this string), but the Kind aids audit/debug.
|
|
const scratchKind = "scratch_restore_test"
|
|
|
|
// DefaultBootTimeout bounds how long the restore-test waits for the scratch guest to reach
|
|
// running before declaring the verify failed.
|
|
const DefaultBootTimeout = 2 * time.Minute
|
|
|
|
// RestoreTestSpec parameterizes one restore-test.
|
|
type RestoreTestSpec struct {
|
|
Archive string // source archive volid to restore (resolved by the caller)
|
|
SourceTier string // "local" this slice (pbs = Phase B) — for the report
|
|
RestoreStorage string // target storage for the restored rootfs (e.g. "local-lvm")
|
|
ScratchMin int // inclusive scratch VMID band (must be > 0)
|
|
ScratchMax int // inclusive
|
|
BootTimeout time.Duration // 0 → DefaultBootTimeout
|
|
}
|
|
|
|
// RestoreTestResult is the reconcile-local outcome (the backup package maps it to the
|
|
// hub.RestoreTest wire record — reconcile must not import hub for this).
|
|
type RestoreTestResult struct {
|
|
Archive string
|
|
SourceTier string
|
|
ScratchVMID int
|
|
Pass bool
|
|
Verified string // "boot+running" this slice
|
|
Skipped bool // no free scratch VMID in band → test not run
|
|
Err error
|
|
StartedAt time.Time
|
|
Duration time.Duration
|
|
// StartWarnings holds the warning line(s) the guest-start task emitted (e.g. the
|
|
// systemd-nesting advisory). Populated only when the start exited "WARNINGS: N";
|
|
// always surfaced, NEVER used to decide pass/fail (the verdict is liveness — waitRunning).
|
|
StartWarnings []string
|
|
// WarningsRecognized is true iff every StartWarnings line matches the benign anchor.
|
|
// It affects VISIBILITY ONLY (log level / operator attention), never the verdict — so a
|
|
// wrong/stale recognizer can at worst over-notice a benign warning, never false-fail and
|
|
// never hide a real one. Empty StartWarnings ⇒ trivially recognized (N/A).
|
|
WarningsRecognized bool
|
|
}
|
|
|
|
// benignWarningAnchor is a deliberately version-FREE substring of the systemd-nesting start
|
|
// advisory ("Systemd <N> detected. You may need to enable nesting."). It carries no systemd
|
|
// version number, so — unlike an exact-string allowlist on "Systemd 257…" — it cannot rot back
|
|
// into the false-fail bug as guests move to systemd 258+. Matched case-insensitively.
|
|
const benignWarningAnchor = "enable nesting"
|
|
|
|
// extractWarningLines pulls the warning lines out of a task log tail. PVE prefixes task
|
|
// warnings with "WARN" (e.g. "WARN: Systemd 257 detected…"); we keep those, trimmed.
|
|
func extractWarningLines(logTail []string) []string {
|
|
var out []string
|
|
for _, l := range logTail {
|
|
if t := strings.TrimSpace(l); strings.HasPrefix(t, "WARN") {
|
|
out = append(out, t)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// warningsRecognized reports whether EVERY warning line is the benign anchor. Empty ⇒ true
|
|
// (no warnings to worry about). One unrecognized line ⇒ false (operator should look).
|
|
func warningsRecognized(warnings []string) bool {
|
|
for _, w := range warnings {
|
|
if !strings.Contains(strings.ToLower(w), benignWarningAnchor) {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
// IntentForScratchDestroy builds the benign teardown intent for an agent-owned scratch
|
|
// guest: ClassGuestDestroy made benign by AgentTaggedScratch provenance (classify.go). The
|
|
// gate authorizes it unsigned but is genuinely in-path (wrong provenance → pending_signature).
|
|
func IntentForScratchDestroy(hostID string, vmid int) Intent {
|
|
return Intent{
|
|
Class: ClassGuestDestroy,
|
|
HostID: hostID,
|
|
GuestID: strconv.Itoa(vmid),
|
|
VMID: vmid,
|
|
Provenance: Provenance{AgentTaggedScratch: true}, // agent-internal, never hub-sourced
|
|
Source: SourceOneShotJob,
|
|
}
|
|
}
|
|
|
|
// RunRestoreTest runs one restore-test on the per-guest queue lane of a fresh scratch VMID.
|
|
// It journals a Scratch-owned entry BEFORE any mutation, so a crash anywhere after this
|
|
// point is recoverable (Recover destroys a launch-proven scratch guest via its journaled
|
|
// UPID). Teardown runs on every launch-proven path (defer), including a failed verify — but
|
|
// NEVER when the restore failed before creating anything (proof-of-launch, campaign F1b). A
|
|
// band vmid PVE reports "already exists" is advanced past, not failed (F2). The returned Err
|
|
// is the TEST verdict's error (restore or boot failure), independent of teardown success.
|
|
func (e *Engine) RunRestoreTest(ctx context.Context, spec RestoreTestSpec) RestoreTestResult {
|
|
now := time.Now().UTC()
|
|
res := RestoreTestResult{Archive: spec.Archive, SourceTier: spec.SourceTier, StartedAt: now}
|
|
|
|
if spec.Archive == "" || spec.RestoreStorage == "" {
|
|
res.Err = fmt.Errorf("reconcile: restore-test needs an archive and a restore storage")
|
|
return res
|
|
}
|
|
if spec.ScratchMin <= 0 || spec.ScratchMax < spec.ScratchMin {
|
|
res.Err = fmt.Errorf("reconcile: invalid scratch VMID band [%d,%d]", spec.ScratchMin, spec.ScratchMax)
|
|
return res
|
|
}
|
|
|
|
lxc, err := e.api.ListLXC(ctx)
|
|
if err != nil {
|
|
res.Err = fmt.Errorf("reconcile: restore-test list guests: %w", err)
|
|
return res
|
|
}
|
|
// Band-advance loop (campaign pool-effects F2): the band scan below is POOL-BLIND (the
|
|
// scoped token's ListLXC can't see non-pool guests), so a band vmid can look free while a
|
|
// squatter sits on it. PVE tells us at restore time ("already exists"); we then advance to
|
|
// the next band vmid instead of failing — one squatter must not permanently break the
|
|
// restore-test or raise a false "backup unrestorable" alert. Bounded by the band width.
|
|
occupied := make(map[int]bool)
|
|
for {
|
|
vmid, ok := pickScratchVMID(lxc, spec.ScratchMin, spec.ScratchMax, occupied)
|
|
if !ok {
|
|
// Band exhausted (in-use and/or invisible squatters) → skip, never panic, never
|
|
// pick out-of-band, never FAIL. Recover will reap any genuinely leaked ones.
|
|
e.logger.Warn("restore-test skipped: no free scratch VMID in band",
|
|
"min", spec.ScratchMin, "max", spec.ScratchMax, "occupied_invisible", len(occupied))
|
|
res.Skipped = true
|
|
res.ScratchVMID = 0
|
|
res.Err = nil
|
|
return res
|
|
}
|
|
res.ScratchVMID = vmid
|
|
|
|
// Serialize on the scratch VMID's lane (inherits §10), and capture the result.
|
|
var vmidOccupied bool
|
|
ch := e.queue.Submit(vmid, func() error {
|
|
vmidOccupied = e.runScratchTest(ctx, vmid, spec, &res)
|
|
return res.Err
|
|
})
|
|
<-ch
|
|
if !vmidOccupied {
|
|
break
|
|
}
|
|
e.logger.Warn("restore-test: band VMID occupied by a guest invisible to the token; advancing",
|
|
"vmid", vmid)
|
|
occupied[vmid] = true
|
|
}
|
|
res.Duration = time.Since(now)
|
|
return res
|
|
}
|
|
|
|
// runScratchTest is the journaled body (runs on vmid's queue lane). The occupied return is true
|
|
// ONLY when PVE synchronously refused the restore because the vmid already holds a guest (one
|
|
// the pool-blind band scan couldn't see) — the caller then advances to the next band vmid (F2).
|
|
func (e *Engine) runScratchTest(ctx context.Context, vmid int, spec RestoreTestSpec, res *RestoreTestResult) (occupied bool) {
|
|
base := JournalEntry{OpID: e.scratchOpID(vmid), VMID: vmid, Kind: scratchKind, Scratch: true}
|
|
|
|
// OWN the scratch guest's cleanup BEFORE any mutation. From here, a crash is recoverable.
|
|
e.append(withState(base, OpStarted))
|
|
|
|
// Teardown runs on every exit AFTER the restore launched (even on a failed verify), using a
|
|
// cancel-immune context so a daemon shutdown mid-test still tears down; if teardown fails,
|
|
// the entry stays in-flight and Recover reaps the guest on the next start (via the journaled
|
|
// UPID). `launched` is the proof-of-launch gate (campaign pool-effects F1b): a restore that
|
|
// failed synchronously (no UPID) created NOTHING, so teardown must NEVER destroy the vmid —
|
|
// an invisible pre-existing guest may sit there. The entry is then closed terminal-failed
|
|
// (nothing exists to recover).
|
|
launched := false
|
|
defer func() {
|
|
if launched {
|
|
e.teardownScratch(ctx, base)
|
|
return
|
|
}
|
|
e.append(withState(base, OpFailed))
|
|
}()
|
|
|
|
// 1. Restore into the fresh scratch VMID (benign create path). The UPID is for error
|
|
// detection only — it does NOT make the Scratch entry terminal (teardown does).
|
|
// A source guest with a host BIND-mount mountpoint (slice-10 data drive) can't be
|
|
// vzrestore'd by the privsep token ("restoring 'mpN' to bind mount is only possible for
|
|
// root"). The restore-test only needs the guest to BOOT — the data drives are irrelevant
|
|
// (their host paths would also collide). So neutralize each source bind-mount mpN to a
|
|
// throwaway volume on the restore storage (needs no root). Best-effort: if the source
|
|
// config can't be read, restore as-is (a bind-mount guest then fails as before, in the verdict).
|
|
var mountOverrides map[string]string
|
|
if srcVMID, ok := archiveVMID(spec.Archive); ok {
|
|
if srcCfg, cerr := e.api.GuestConfig(ctx, srcVMID); cerr == nil {
|
|
if binds := bindMountOverrides(srcCfg.MountPoints(), spec.RestoreStorage); len(binds) > 0 {
|
|
// PVE refuses a restore that carries mountpoint params unless `rootfs` is also set
|
|
// ("mount points configured, but 'rootfs' not set"). Size the rootfs override from the
|
|
// SOURCE rootfs (restore needs target >= the archive's volume). Without a parseable
|
|
// size we can't safely override, so restore as-is (the bind-mount restore then fails
|
|
// in the verdict rather than risking a wrong rootfs size).
|
|
if sz := rootfsSizeGB(srcCfg.RootFS); sz > 0 {
|
|
binds["rootfs"] = fmt.Sprintf("%s:%d", spec.RestoreStorage, sz)
|
|
mountOverrides = binds
|
|
e.logger.Info("restore-test: neutralizing source bind-mount mountpoints for scratch restore",
|
|
"source_vmid", srcVMID, "scratch", vmid, "bind_mounts", len(binds)-1, "rootfs_gb", sz)
|
|
} else {
|
|
e.logger.Warn("restore-test: source has bind mounts but rootfs size unparseable — restoring as-is",
|
|
"source_vmid", srcVMID, "rootfs", srcCfg.RootFS)
|
|
}
|
|
}
|
|
} else {
|
|
e.logger.Warn("restore-test: could not read source config for mp overrides (restoring as-is)",
|
|
"source_vmid", srcVMID, "err", cerr)
|
|
}
|
|
}
|
|
upid, err := e.api.RestoreLXC(ctx, proxmox.RestoreLXCOptions{
|
|
// Pool=DefaultPool so the scratch guest is created INTO the felhom pool — else a pool-scoped
|
|
// token 403s on the scratch guest's config/start/destroy (SPIKE residual #2).
|
|
VMID: vmid, Archive: spec.Archive, Storage: spec.RestoreStorage, MountOverrides: mountOverrides, Pool: DefaultPool,
|
|
})
|
|
if err != nil {
|
|
if pveAlreadyExists(err) {
|
|
// The band vmid holds a guest the pool-blind scan couldn't see. Nothing was
|
|
// created; NOT a test verdict — the caller advances to the next band vmid (F2).
|
|
return true
|
|
}
|
|
res.Err = fmt.Errorf("reconcile: restore-test restore: %w", err)
|
|
return false
|
|
}
|
|
// Proof-of-launch: the POST was accepted — from here teardown owns the guest. Accepted
|
|
// residual: a crash before the next append leaks a scratch guest Recover won't destroy
|
|
// (no journaled UPID) — cleanable, and preferable to destroying an innocent guest.
|
|
launched = true
|
|
e.append(withUPID(base, upid, OpTaskRunning))
|
|
if upid != "" {
|
|
if _, err := e.api.WaitTask(ctx, upid, proxmox.WaitOptions{}); err != nil {
|
|
res.Err = fmt.Errorf("reconcile: restore-test restore task: %w", err)
|
|
return
|
|
}
|
|
}
|
|
|
|
// 2. Net link-down on every interface BEFORE boot — test-safety so the clone (which
|
|
// keeps the source MAC/hostname; identity-reset is slice 7) can't conflict with a
|
|
// running source on L2/IP. Benign SetConfig.
|
|
cfg, err := e.api.GuestConfig(ctx, vmid)
|
|
if err != nil {
|
|
res.Err = fmt.Errorf("reconcile: restore-test read scratch config: %w", err)
|
|
return
|
|
}
|
|
for key, val := range cfg.Nets() {
|
|
if _, err := e.api.SetConfig(ctx, vmid, map[string]string{key: withLinkDown(val)}); err != nil {
|
|
res.Err = fmt.Errorf("reconcile: restore-test net link-down %s: %w", key, err)
|
|
return
|
|
}
|
|
}
|
|
|
|
// 3. Boot and verify it reaches running (basic liveness; deep app-health is slice 8).
|
|
// The VERDICT is liveness (waitRunning), NEVER the start task's exitstatus. A start
|
|
// that completes with warnings (e.g. the systemd-nesting advisory → exit "WARNINGS: N")
|
|
// and then reaches running is a PASS — deciding pass/fail on an advisory exit code is
|
|
// the crying-wolf bug this guards against. We pass AllowWarnings so WaitTask doesn't
|
|
// hard-fail on it, then fetch + surface the warning text (visibility only).
|
|
startUPID, err := e.api.Start(ctx, vmid)
|
|
if err != nil {
|
|
res.Err = fmt.Errorf("reconcile: restore-test start: %w", err)
|
|
return
|
|
}
|
|
if startUPID != "" {
|
|
st, err := e.api.WaitTask(ctx, startUPID, proxmox.WaitOptions{AllowWarnings: true})
|
|
if err != nil {
|
|
// A real (non-WARNINGS) start-task failure still fails the test.
|
|
res.Err = fmt.Errorf("reconcile: restore-test start task: %w", err)
|
|
return
|
|
}
|
|
if strings.HasPrefix(st.ExitStatus, "WARNINGS") {
|
|
// Surface the warning(s); do NOT fail. Liveness below is the verdict.
|
|
tail, logErr := e.api.TaskLogTail(ctx, startUPID, 50)
|
|
if logErr != nil {
|
|
e.logger.Warn("restore-test: could not read start-task log for warnings",
|
|
"vmid", vmid, "err", logErr)
|
|
}
|
|
res.StartWarnings = extractWarningLines(tail)
|
|
res.WarningsRecognized = warningsRecognized(res.StartWarnings)
|
|
}
|
|
}
|
|
if err := e.waitRunning(ctx, vmid, bootTimeout(spec)); err != nil {
|
|
res.Err = err
|
|
return
|
|
}
|
|
res.Pass = true
|
|
res.Verified = "boot+running"
|
|
return false
|
|
}
|
|
|
|
// archiveVMID extracts the source VMID from a backup archive volid. Handles PBS volids
|
|
// ("<store>:backup/ct/<vmid>/<ts>" and the /vm/<vmid>/ form) and vzdump file volids
|
|
// ("<store>:backup/vzdump-(lxc|qemu)-<vmid>-..."). Returns false when no vmid is found.
|
|
func archiveVMID(volid string) (int, bool) {
|
|
for _, re := range []*regexp.Regexp{pbsArchiveRE, vzdumpArchiveRE} {
|
|
if m := re.FindStringSubmatch(volid); m != nil {
|
|
if n, err := strconv.Atoi(m[1]); err == nil {
|
|
return n, true
|
|
}
|
|
}
|
|
}
|
|
return 0, false
|
|
}
|
|
|
|
var (
|
|
pbsArchiveRE = regexp.MustCompile(`/(?:ct|vm)/(\d+)/`)
|
|
vzdumpArchiveRE = regexp.MustCompile(`vzdump-(?:lxc|qemu)-(\d+)-`)
|
|
)
|
|
|
|
// bindMountOverrides maps each BIND-mount mpN (the volume part is an absolute host PATH, not a
|
|
// "storage:volume") to a throwaway 1G volume override on restoreStorage, preserving the in-guest
|
|
// mount path. Storage-backed mpN are left to restore normally (returns nil when there are none).
|
|
// This is what lets the restore-test boot-verify a slice-10 enrolled guest whose data drive is a
|
|
// host bind mount the privsep token can't otherwise restore.
|
|
func bindMountOverrides(mps map[string]string, restoreStorage string) map[string]string {
|
|
out := map[string]string{}
|
|
for key, val := range mps {
|
|
volPart, rest, _ := strings.Cut(val, ",")
|
|
if !strings.HasPrefix(volPart, "/") {
|
|
continue // "storage:volume" → a real volume, not a host bind mount
|
|
}
|
|
mp := mountPathOf(rest)
|
|
if mp == "" {
|
|
mp = volPart // fall back to the host path if no explicit in-guest mp=
|
|
}
|
|
out[key] = fmt.Sprintf("%s:1,mp=%s,backup=0", restoreStorage, mp)
|
|
}
|
|
if len(out) == 0 {
|
|
return nil
|
|
}
|
|
return out
|
|
}
|
|
|
|
// mountPathOf returns the mp= field from an mpN value's trailing options ("" if absent).
|
|
func mountPathOf(opts string) string {
|
|
for _, kv := range strings.Split(opts, ",") {
|
|
if v, ok := strings.CutPrefix(kv, "mp="); ok {
|
|
return v
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// rootfsSizeGB parses the GB size from a volume spec's "size=<N><unit>" field (e.g.
|
|
// "local-lvm:vm-9201-disk-0,size=8G"), rounding UP to whole GB — a restore needs the target volume
|
|
// >= the archive's. Returns 0 when no size field is present (caller then skips the override).
|
|
func rootfsSizeGB(spec string) int {
|
|
for _, kv := range strings.Split(spec, ",") {
|
|
if v, ok := strings.CutPrefix(kv, "size="); ok {
|
|
return sizeToGB(v)
|
|
}
|
|
}
|
|
return 0
|
|
}
|
|
|
|
// sizeToGB converts a PVE size string ("8G", "512M", "1T", "8192K") to whole GB, rounding up (min 1
|
|
// when positive). Returns 0 on a malformed value.
|
|
func sizeToGB(s string) int {
|
|
if s == "" {
|
|
return 0
|
|
}
|
|
mult := 1.0 // default GB if no recognized unit suffix
|
|
num := s
|
|
switch s[len(s)-1] {
|
|
case 'T', 't':
|
|
mult, num = 1024, s[:len(s)-1]
|
|
case 'G', 'g':
|
|
mult, num = 1, s[:len(s)-1]
|
|
case 'M', 'm':
|
|
mult, num = 1.0/1024, s[:len(s)-1]
|
|
case 'K', 'k':
|
|
mult, num = 1.0/(1024*1024), s[:len(s)-1]
|
|
}
|
|
f, err := strconv.ParseFloat(num, 64)
|
|
if err != nil || f <= 0 {
|
|
return 0
|
|
}
|
|
gb := int(math.Ceil(f * mult))
|
|
if gb < 1 {
|
|
gb = 1
|
|
}
|
|
return gb
|
|
}
|
|
|
|
// teardownScratch destroys the scratch guest (benign, gated) and records the entry terminal.
|
|
// On any teardown failure it leaves the entry in-flight so Recover reaps the guest later.
|
|
func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry) {
|
|
// Cancel-immune + bounded, so a shutdown mid-test still tears down.
|
|
tctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 2*time.Minute)
|
|
defer cancel()
|
|
|
|
dec := e.gate.Authorize(IntentForScratchDestroy(e.hostID, base.VMID), nil)
|
|
if !dec.Allowed {
|
|
e.logger.Error("restore-test: scratch teardown refused by gate (unexpected); left for Recover",
|
|
"vmid", base.VMID, "reason", dec.Reason)
|
|
return
|
|
}
|
|
upid, err := e.api.DestroyLXC(tctx, base.VMID)
|
|
if err != nil {
|
|
e.logger.Error("restore-test: scratch teardown failed; left for Recover", "vmid", base.VMID, "err", err)
|
|
return
|
|
}
|
|
if upid != "" {
|
|
if _, err := e.api.WaitTask(tctx, upid, proxmox.WaitOptions{}); err != nil {
|
|
e.logger.Error("restore-test: scratch teardown task failed; left for Recover", "vmid", base.VMID, "err", err)
|
|
return
|
|
}
|
|
}
|
|
e.append(withState(base, OpSucceeded))
|
|
e.logger.Info("restore-test: scratch guest torn down", "vmid", base.VMID)
|
|
}
|
|
|
|
// waitRunning polls GuestStatus until the guest is running or the timeout elapses. The poll
|
|
// interval is 2s in production, but shrinks for short timeouts so it stays responsive.
|
|
func (e *Engine) waitRunning(ctx context.Context, vmid int, timeout time.Duration) error {
|
|
interval := 2 * time.Second
|
|
if timeout < 4*interval {
|
|
if interval = timeout / 4; interval < 10*time.Millisecond {
|
|
interval = 10 * time.Millisecond
|
|
}
|
|
}
|
|
deadline := time.Now().Add(timeout)
|
|
t := time.NewTicker(interval)
|
|
defer t.Stop()
|
|
for {
|
|
g, err := e.api.GuestStatus(ctx, vmid)
|
|
if err == nil && g.Status == "running" {
|
|
return nil
|
|
}
|
|
if time.Now().After(deadline) {
|
|
if err != nil {
|
|
return fmt.Errorf("reconcile: restore-test verify: guest %d not running within %s (last err: %w)", vmid, timeout, err)
|
|
}
|
|
return fmt.Errorf("reconcile: restore-test verify: guest %d not running within %s", vmid, timeout)
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
return ctx.Err()
|
|
case <-t.C:
|
|
}
|
|
}
|
|
}
|
|
|
|
// pickScratchVMID returns the lowest free VMID in [min,max], excluding the standing 9999
|
|
// scratch, any in-use guest, and the caller's exclude set (band vmids PVE reported occupied by
|
|
// guests the pool-blind list can't see — the F2 band-advance). ok=false when the band is fully
|
|
// occupied (the test is then skipped, never run out-of-band).
|
|
func pickScratchVMID(lxc []proxmox.Guest, min, max int, exclude map[int]bool) (int, bool) {
|
|
used := make(map[int]bool, len(lxc))
|
|
for _, g := range lxc {
|
|
used[g.VMID] = true
|
|
}
|
|
for id := min; id <= max; id++ {
|
|
if id == 9999 || used[id] || exclude[id] {
|
|
continue
|
|
}
|
|
return id, true
|
|
}
|
|
return 0, false
|
|
}
|
|
|
|
// withLinkDown sets link_down=1 on a Proxmox netN config string, REPLACING any existing
|
|
// link_down token (never blind-concatenating, so a re-applied/pre-set value can't produce a
|
|
// malformed netN).
|
|
func withLinkDown(netN string) string {
|
|
parts := strings.Split(netN, ",")
|
|
out := parts[:0]
|
|
for _, p := range parts {
|
|
if p == "" || strings.HasPrefix(p, "link_down=") {
|
|
continue
|
|
}
|
|
out = append(out, p)
|
|
}
|
|
out = append(out, "link_down=1")
|
|
return strings.Join(out, ",")
|
|
}
|
|
|
|
func bootTimeout(spec RestoreTestSpec) time.Duration {
|
|
if spec.BootTimeout > 0 {
|
|
return spec.BootTimeout
|
|
}
|
|
return DefaultBootTimeout
|
|
}
|
|
|
|
func (e *Engine) scratchOpID(vmid int) string {
|
|
return "scratch-restore-" + strconv.Itoa(vmid) + "-" + nextSeq(&e.opSeq)
|
|
}
|
|
|
|
// withState / withUPID build journal records from a base entry, preserving its identity +
|
|
// Scratch flag.
|
|
func withState(base JournalEntry, state OpState) JournalEntry {
|
|
base.State = state
|
|
base.At = time.Now().UTC()
|
|
return base
|
|
}
|
|
func withUPID(base JournalEntry, upid string, state OpState) JournalEntry {
|
|
base.UPID = upid
|
|
base.State = state
|
|
base.At = time.Now().UTC()
|
|
return base
|
|
}
|