package reconcile import ( "context" "fmt" "math" "regexp" "strconv" "strings" "time" "gitea.dooplex.hu/admin/felhom-agent/internal/proxmox" ) // The self-restore-test (doc 03 §8) — the piece that closes "a backup you haven't restored // isn't a backup". It is a JOURNALED reconcile job so it inherits the slice-4 journal, // per-guest serialization, and crash-safe recovery: a mid-test crash can't leak a scratch // guest (engine.Recover tears it down). Every step here is BENIGN — restore-to-new // (ClassCreate), a benign net-link-down SetConfig, and a scratch teardown that is benign by // agent-tagged-scratch provenance (no new destructive class, no new crypto). // scratchKind is the journal Kind for a restore-test scratch-guest-owning entry. Recover // keys off JournalEntry.Scratch (not this string), but the Kind aids audit/debug. const scratchKind = "scratch_restore_test" // DefaultBootTimeout bounds how long the restore-test waits for the scratch guest to reach // running before declaring the verify failed. const DefaultBootTimeout = 2 * time.Minute // RestoreTestSpec parameterizes one restore-test. type RestoreTestSpec struct { Archive string // source archive volid to restore (resolved by the caller) SourceTier string // "local" this slice (pbs = Phase B) — for the report RestoreStorage string // target storage for the restored rootfs (e.g. "local-lvm") ScratchMin int // inclusive scratch VMID band (must be > 0) ScratchMax int // inclusive BootTimeout time.Duration // 0 → DefaultBootTimeout // RestoreTaskTimeout bounds the wait on the restore (vzrestore) task. 0 → WaitOptions' 10m // default (fine for a LOCAL restore). A WAN/pbs restore of a large guest runs long, so the // caller sets this generously for the pbs tier — else the wait expires mid-restore, teardown // fires against a still-restoring (not-yet-pool-associated) guest, and the scratch leaks. RestoreTaskTimeout time.Duration } // RestoreTestResult is the reconcile-local outcome (the backup package maps it to the // hub.RestoreTest wire record — reconcile must not import hub for this). type RestoreTestResult struct { Archive string SourceTier string ScratchVMID int Pass bool Verified string // "boot+running" this slice Skipped bool // no free scratch VMID in band → test not run Err error StartedAt time.Time Duration time.Duration // StartWarnings holds the warning line(s) the guest-start task emitted (e.g. the // systemd-nesting advisory). Populated only when the start exited "WARNINGS: N"; // always surfaced, NEVER used to decide pass/fail (the verdict is liveness — waitRunning). StartWarnings []string // WarningsRecognized is true iff every StartWarnings line matches the benign anchor. // It affects VISIBILITY ONLY (log level / operator attention), never the verdict — so a // wrong/stale recognizer can at worst over-notice a benign warning, never false-fail and // never hide a real one. Empty StartWarnings ⇒ trivially recognized (N/A). WarningsRecognized bool } // benignWarningAnchor is a deliberately version-FREE substring of the systemd-nesting start // advisory ("Systemd detected. You may need to enable nesting."). It carries no systemd // version number, so — unlike an exact-string allowlist on "Systemd 257…" — it cannot rot back // into the false-fail bug as guests move to systemd 258+. Matched case-insensitively. const benignWarningAnchor = "enable nesting" // extractWarningLines pulls the warning lines out of a task log tail. PVE prefixes task // warnings with "WARN" (e.g. "WARN: Systemd 257 detected…"); we keep those, trimmed. func extractWarningLines(logTail []string) []string { var out []string for _, l := range logTail { if t := strings.TrimSpace(l); strings.HasPrefix(t, "WARN") { out = append(out, t) } } return out } // warningsRecognized reports whether EVERY warning line is the benign anchor. Empty ⇒ true // (no warnings to worry about). One unrecognized line ⇒ false (operator should look). func warningsRecognized(warnings []string) bool { for _, w := range warnings { if !strings.Contains(strings.ToLower(w), benignWarningAnchor) { return false } } return true } // IntentForScratchDestroy builds the benign teardown intent for an agent-owned scratch // guest: ClassGuestDestroy made benign by AgentTaggedScratch provenance (classify.go). The // gate authorizes it unsigned but is genuinely in-path (wrong provenance → pending_signature). func IntentForScratchDestroy(hostID string, vmid int) Intent { return Intent{ Class: ClassGuestDestroy, HostID: hostID, GuestID: strconv.Itoa(vmid), VMID: vmid, Provenance: Provenance{AgentTaggedScratch: true}, // agent-internal, never hub-sourced Source: SourceOneShotJob, } } // RunRestoreTest runs one restore-test on the per-guest queue lane of a fresh scratch VMID. // It journals a Scratch-owned entry BEFORE any mutation, so a crash anywhere after this // point is recoverable (Recover destroys a launch-proven scratch guest via its journaled // UPID). Teardown runs on every launch-proven path (defer), including a failed verify — but // NEVER when the restore failed before creating anything (proof-of-launch, campaign F1b). A // band vmid PVE reports "already exists" is advanced past, not failed (F2). The returned Err // is the TEST verdict's error (restore or boot failure), independent of teardown success. func (e *Engine) RunRestoreTest(ctx context.Context, spec RestoreTestSpec) RestoreTestResult { now := time.Now().UTC() res := RestoreTestResult{Archive: spec.Archive, SourceTier: spec.SourceTier, StartedAt: now} if spec.Archive == "" || spec.RestoreStorage == "" { res.Err = fmt.Errorf("reconcile: restore-test needs an archive and a restore storage") return res } if spec.ScratchMin <= 0 || spec.ScratchMax < spec.ScratchMin { res.Err = fmt.Errorf("reconcile: invalid scratch VMID band [%d,%d]", spec.ScratchMin, spec.ScratchMax) return res } lxc, err := e.api.ListLXC(ctx) if err != nil { res.Err = fmt.Errorf("reconcile: restore-test list guests: %w", err) return res } // Band-advance loop (campaign pool-effects F2): the band scan below is POOL-BLIND (the // scoped token's ListLXC can't see non-pool guests), so a band vmid can look free while a // squatter sits on it. PVE tells us at restore time ("already exists"); we then advance to // the next band vmid instead of failing — one squatter must not permanently break the // restore-test or raise a false "backup unrestorable" alert. Bounded by the band width. occupied := make(map[int]bool) for { vmid, ok := pickScratchVMID(lxc, spec.ScratchMin, spec.ScratchMax, occupied) if !ok { // Band exhausted (in-use and/or invisible squatters) → skip, never panic, never // pick out-of-band, never FAIL. Recover will reap any genuinely leaked ones. e.logger.Warn("restore-test skipped: no free scratch VMID in band", "min", spec.ScratchMin, "max", spec.ScratchMax, "occupied_invisible", len(occupied)) res.Skipped = true res.ScratchVMID = 0 res.Err = nil return res } res.ScratchVMID = vmid // Serialize on the scratch VMID's lane (inherits §10), and capture the result. var vmidOccupied bool ch := e.queue.Submit(vmid, func() error { vmidOccupied = e.runScratchTest(ctx, vmid, spec, &res) return res.Err }) <-ch if !vmidOccupied { break } e.logger.Warn("restore-test: band VMID occupied by a guest invisible to the token; advancing", "vmid", vmid) occupied[vmid] = true } res.Duration = time.Since(now) return res } // runScratchTest is the journaled body (runs on vmid's queue lane). The occupied return is true // ONLY when PVE synchronously refused the restore because the vmid already holds a guest (one // the pool-blind band scan couldn't see) — the caller then advances to the next band vmid (F2). func (e *Engine) runScratchTest(ctx context.Context, vmid int, spec RestoreTestSpec, res *RestoreTestResult) (occupied bool) { base := JournalEntry{OpID: e.scratchOpID(vmid), VMID: vmid, Kind: scratchKind, Scratch: true} // OWN the scratch guest's cleanup BEFORE any mutation. From here, a crash is recoverable. e.append(withState(base, OpStarted)) // Teardown runs on every exit AFTER the restore launched (even on a failed verify), using a // cancel-immune context so a daemon shutdown mid-test still tears down; if teardown fails, // the entry stays in-flight and Recover reaps the guest on the next start (via the journaled // UPID). `launched` is the proof-of-launch gate (campaign pool-effects F1b): a restore that // failed synchronously (no UPID) created NOTHING, so teardown must NEVER destroy the vmid — // an invisible pre-existing guest may sit there. The entry is then closed terminal-failed // (nothing exists to recover). launched := false defer func() { if launched { e.teardownScratch(ctx, base) return } e.append(withState(base, OpFailed)) }() // 1. Restore into the fresh scratch VMID (benign create path). The UPID is for error // detection only — it does NOT make the Scratch entry terminal (teardown does). // A source guest with a host BIND-mount mountpoint (slice-10 data drive) can't be // vzrestore'd by the privsep token ("restoring 'mpN' to bind mount is only possible for // root"). The restore-test only needs the guest to BOOT — the data drives are irrelevant // (their host paths would also collide). So neutralize each source bind-mount mpN to a // throwaway volume on the restore storage (needs no root). Best-effort: if the source // config can't be read, restore as-is (a bind-mount guest then fails as before, in the verdict). var mountOverrides map[string]string if srcVMID, ok := archiveVMID(spec.Archive); ok { if srcCfg, cerr := e.api.GuestConfig(ctx, srcVMID); cerr == nil { if binds := bindMountOverrides(srcCfg.MountPoints(), spec.RestoreStorage); len(binds) > 0 { // PVE refuses a restore that carries mountpoint params unless `rootfs` is also set // ("mount points configured, but 'rootfs' not set"). Size the rootfs override from the // SOURCE rootfs (restore needs target >= the archive's volume). Without a parseable // size we can't safely override, so restore as-is (the bind-mount restore then fails // in the verdict rather than risking a wrong rootfs size). if sz := rootfsSizeGB(srcCfg.RootFS); sz > 0 { binds["rootfs"] = fmt.Sprintf("%s:%d", spec.RestoreStorage, sz) mountOverrides = binds e.logger.Info("restore-test: neutralizing source bind-mount mountpoints for scratch restore", "source_vmid", srcVMID, "scratch", vmid, "bind_mounts", len(binds)-1, "rootfs_gb", sz) } else { e.logger.Warn("restore-test: source has bind mounts but rootfs size unparseable — restoring as-is", "source_vmid", srcVMID, "rootfs", srcCfg.RootFS) } } } else { e.logger.Warn("restore-test: could not read source config for mp overrides (restoring as-is)", "source_vmid", srcVMID, "err", cerr) } } upid, err := e.api.RestoreLXC(ctx, proxmox.RestoreLXCOptions{ // Pool=DefaultPool so the scratch guest is created INTO the felhom pool — else a pool-scoped // token 403s on the scratch guest's config/start/destroy (SPIKE residual #2). VMID: vmid, Archive: spec.Archive, Storage: spec.RestoreStorage, MountOverrides: mountOverrides, Pool: DefaultPool, }) if err != nil { if pveAlreadyExists(err) { // The band vmid holds a guest the pool-blind scan couldn't see. Nothing was // created; NOT a test verdict — the caller advances to the next band vmid (F2). return true } res.Err = fmt.Errorf("reconcile: restore-test restore: %w", err) return false } // Proof-of-launch: the POST was accepted — from here teardown owns the guest. Accepted // residual: a crash before the next append leaks a scratch guest Recover won't destroy // (no journaled UPID) — cleanable, and preferable to destroying an innocent guest. launched = true e.append(withUPID(base, upid, OpTaskRunning)) if upid != "" { if _, err := e.api.WaitTask(ctx, upid, proxmox.WaitOptions{Timeout: spec.RestoreTaskTimeout}); err != nil { res.Err = fmt.Errorf("reconcile: restore-test restore task: %w", err) return } } // 2. Net link-down on every interface BEFORE boot — test-safety so the clone (which // keeps the source MAC/hostname; identity-reset is slice 7) can't conflict with a // running source on L2/IP. Benign SetConfig. cfg, err := e.api.GuestConfig(ctx, vmid) if err != nil { res.Err = fmt.Errorf("reconcile: restore-test read scratch config: %w", err) return } for key, val := range cfg.Nets() { if _, err := e.api.SetConfig(ctx, vmid, map[string]string{key: withLinkDown(val)}); err != nil { res.Err = fmt.Errorf("reconcile: restore-test net link-down %s: %w", key, err) return } } // 3. Boot and verify it reaches running (basic liveness; deep app-health is slice 8). // The VERDICT is liveness (waitRunning), NEVER the start task's exitstatus. A start // that completes with warnings (e.g. the systemd-nesting advisory → exit "WARNINGS: N") // and then reaches running is a PASS — deciding pass/fail on an advisory exit code is // the crying-wolf bug this guards against. We pass AllowWarnings so WaitTask doesn't // hard-fail on it, then fetch + surface the warning text (visibility only). startUPID, err := e.api.Start(ctx, vmid) if err != nil { res.Err = fmt.Errorf("reconcile: restore-test start: %w", err) return } if startUPID != "" { st, err := e.api.WaitTask(ctx, startUPID, proxmox.WaitOptions{AllowWarnings: true}) if err != nil { // A real (non-WARNINGS) start-task failure still fails the test. res.Err = fmt.Errorf("reconcile: restore-test start task: %w", err) return } if strings.HasPrefix(st.ExitStatus, "WARNINGS") { // Surface the warning(s); do NOT fail. Liveness below is the verdict. tail, logErr := e.api.TaskLogTail(ctx, startUPID, 50) if logErr != nil { e.logger.Warn("restore-test: could not read start-task log for warnings", "vmid", vmid, "err", logErr) } res.StartWarnings = extractWarningLines(tail) res.WarningsRecognized = warningsRecognized(res.StartWarnings) } } if err := e.waitRunning(ctx, vmid, bootTimeout(spec)); err != nil { res.Err = err return } res.Pass = true res.Verified = "boot+running" return false } // archiveVMID extracts the source VMID from a backup archive volid. Handles PBS volids // (":backup/ct//" and the /vm// form) and vzdump file volids // (":backup/vzdump-(lxc|qemu)--..."). Returns false when no vmid is found. func archiveVMID(volid string) (int, bool) { for _, re := range []*regexp.Regexp{pbsArchiveRE, vzdumpArchiveRE} { if m := re.FindStringSubmatch(volid); m != nil { if n, err := strconv.Atoi(m[1]); err == nil { return n, true } } } return 0, false } var ( pbsArchiveRE = regexp.MustCompile(`/(?:ct|vm)/(\d+)/`) vzdumpArchiveRE = regexp.MustCompile(`vzdump-(?:lxc|qemu)-(\d+)-`) ) // bindMountOverrides maps each BIND-mount mpN (the volume part is an absolute host PATH, not a // "storage:volume") to a throwaway 1G volume override on restoreStorage, preserving the in-guest // mount path. Storage-backed mpN are left to restore normally (returns nil when there are none). // This is what lets the restore-test boot-verify a slice-10 enrolled guest whose data drive is a // host bind mount the privsep token can't otherwise restore. func bindMountOverrides(mps map[string]string, restoreStorage string) map[string]string { out := map[string]string{} for key, val := range mps { volPart, rest, _ := strings.Cut(val, ",") if !strings.HasPrefix(volPart, "/") { continue // "storage:volume" → a real volume, not a host bind mount } mp := mountPathOf(rest) if mp == "" { mp = volPart // fall back to the host path if no explicit in-guest mp= } out[key] = fmt.Sprintf("%s:1,mp=%s,backup=0", restoreStorage, mp) } if len(out) == 0 { return nil } return out } // mountPathOf returns the mp= field from an mpN value's trailing options ("" if absent). func mountPathOf(opts string) string { for _, kv := range strings.Split(opts, ",") { if v, ok := strings.CutPrefix(kv, "mp="); ok { return v } } return "" } // rootfsSizeGB parses the GB size from a volume spec's "size=" field (e.g. // "local-lvm:vm-9201-disk-0,size=8G"), rounding UP to whole GB — a restore needs the target volume // >= the archive's. Returns 0 when no size field is present (caller then skips the override). func rootfsSizeGB(spec string) int { for _, kv := range strings.Split(spec, ",") { if v, ok := strings.CutPrefix(kv, "size="); ok { return sizeToGB(v) } } return 0 } // sizeToGB converts a PVE size string ("8G", "512M", "1T", "8192K") to whole GB, rounding up (min 1 // when positive). Returns 0 on a malformed value. func sizeToGB(s string) int { if s == "" { return 0 } mult := 1.0 // default GB if no recognized unit suffix num := s switch s[len(s)-1] { case 'T', 't': mult, num = 1024, s[:len(s)-1] case 'G', 'g': mult, num = 1, s[:len(s)-1] case 'M', 'm': mult, num = 1.0/1024, s[:len(s)-1] case 'K', 'k': mult, num = 1.0/(1024*1024), s[:len(s)-1] } f, err := strconv.ParseFloat(num, 64) if err != nil || f <= 0 { return 0 } gb := int(math.Ceil(f * mult)) if gb < 1 { gb = 1 } return gb } // teardownScratch destroys the scratch guest (benign, gated) and records the entry terminal. // On any teardown failure it leaves the entry in-flight so Recover reaps the guest later. func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry) { // Cancel-immune + bounded, so a shutdown mid-test still tears down. tctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 2*time.Minute) defer cancel() dec := e.gate.Authorize(IntentForScratchDestroy(e.hostID, base.VMID), nil) if !dec.Allowed { e.logger.Error("restore-test: scratch teardown refused by gate (unexpected); left for Recover", "vmid", base.VMID, "reason", dec.Reason) return } upid, err := e.api.DestroyLXC(tctx, base.VMID) if err != nil { e.logger.Error("restore-test: scratch teardown failed; left for Recover", "vmid", base.VMID, "err", err) return } if upid != "" { if _, err := e.api.WaitTask(tctx, upid, proxmox.WaitOptions{}); err != nil { e.logger.Error("restore-test: scratch teardown task failed; left for Recover", "vmid", base.VMID, "err", err) return } } e.append(withState(base, OpSucceeded)) e.logger.Info("restore-test: scratch guest torn down", "vmid", base.VMID) } // waitRunning polls GuestStatus until the guest is running or the timeout elapses. The poll // interval is 2s in production, but shrinks for short timeouts so it stays responsive. func (e *Engine) waitRunning(ctx context.Context, vmid int, timeout time.Duration) error { interval := 2 * time.Second if timeout < 4*interval { if interval = timeout / 4; interval < 10*time.Millisecond { interval = 10 * time.Millisecond } } deadline := time.Now().Add(timeout) t := time.NewTicker(interval) defer t.Stop() for { g, err := e.api.GuestStatus(ctx, vmid) if err == nil && g.Status == "running" { return nil } if time.Now().After(deadline) { if err != nil { return fmt.Errorf("reconcile: restore-test verify: guest %d not running within %s (last err: %w)", vmid, timeout, err) } return fmt.Errorf("reconcile: restore-test verify: guest %d not running within %s", vmid, timeout) } select { case <-ctx.Done(): return ctx.Err() case <-t.C: } } } // pickScratchVMID returns the lowest free VMID in [min,max], excluding the standing 9999 // scratch, any in-use guest, and the caller's exclude set (band vmids PVE reported occupied by // guests the pool-blind list can't see — the F2 band-advance). ok=false when the band is fully // occupied (the test is then skipped, never run out-of-band). func pickScratchVMID(lxc []proxmox.Guest, min, max int, exclude map[int]bool) (int, bool) { used := make(map[int]bool, len(lxc)) for _, g := range lxc { used[g.VMID] = true } for id := min; id <= max; id++ { if id == 9999 || used[id] || exclude[id] { continue } return id, true } return 0, false } // withLinkDown sets link_down=1 on a Proxmox netN config string, REPLACING any existing // link_down token (never blind-concatenating, so a re-applied/pre-set value can't produce a // malformed netN). func withLinkDown(netN string) string { parts := strings.Split(netN, ",") out := parts[:0] for _, p := range parts { if p == "" || strings.HasPrefix(p, "link_down=") { continue } out = append(out, p) } out = append(out, "link_down=1") return strings.Join(out, ",") } func bootTimeout(spec RestoreTestSpec) time.Duration { if spec.BootTimeout > 0 { return spec.BootTimeout } return DefaultBootTimeout } func (e *Engine) scratchOpID(vmid int) string { return "scratch-restore-" + strconv.Itoa(vmid) + "-" + nextSeq(&e.opSeq) } // withState / withUPID build journal records from a base entry, preserving its identity + // Scratch flag. func withState(base JournalEntry, state OpState) JournalEntry { base.State = state base.At = time.Now().UTC() return base } func withUPID(base JournalEntry, upid string, state OpState) JournalEntry { base.UPID = upid base.State = state base.At = time.Now().UTC() return base }