v0.6.0-rc1: slice 6 Phase A — backup + the self-restore-test (local target)
The guest-level backup layer + the journaled self-restore-test (restore→boot→verify→ teardown) that closes "a backup you haven't restored isn't a backup". All benign (reuses the slice-4 classifier/gate/journal; no new destructive class/crypto). Local target only; PBS = Phase B. Restore to a NEW guest only. Backups crash-consistent. - proxmox: DestroyLXC, VzdumpOptions.Notes (notes-template), LatestBackupVolID. - reconcile: Engine.RunRestoreTest (journal Scratch entry BEFORE mutation; net link-down pre-boot; defer teardown always; benign gated destroy) + Recover extended to reap a leaked scratch guest (Scratch flag, special-cased before the UPID path; idempotent). - internal/backup: runner (vzdump + archive resolve + bulk-gap = backup!=1) + cadence scheduler (4th daemon goroutine, default 24h) + in-memory report store. - hub: Backup/RestoreTest filled; collector seams; cross-repo golden byte-identical + bidirectional key-set tests; hub handler logs a FAILED restore-test prominently. - config BackupConfig (band 990000-990009 default); --selftest=backup / restore-test. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,289 @@
|
||||
package reconcile
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/proxmox"
|
||||
)
|
||||
|
||||
// The self-restore-test (doc 03 §8) — the piece that closes "a backup you haven't restored
|
||||
// isn't a backup". It is a JOURNALED reconcile job so it inherits the slice-4 journal,
|
||||
// per-guest serialization, and crash-safe recovery: a mid-test crash can't leak a scratch
|
||||
// guest (engine.Recover tears it down). Every step here is BENIGN — restore-to-new
|
||||
// (ClassCreate), a benign net-link-down SetConfig, and a scratch teardown that is benign by
|
||||
// agent-tagged-scratch provenance (no new destructive class, no new crypto).
|
||||
|
||||
// scratchKind is the journal Kind for a restore-test scratch-guest-owning entry. Recover
|
||||
// keys off JournalEntry.Scratch (not this string), but the Kind aids audit/debug.
|
||||
const scratchKind = "scratch_restore_test"
|
||||
|
||||
// DefaultBootTimeout bounds how long the restore-test waits for the scratch guest to reach
|
||||
// running before declaring the verify failed.
|
||||
const DefaultBootTimeout = 2 * time.Minute
|
||||
|
||||
// RestoreTestSpec parameterizes one restore-test.
|
||||
type RestoreTestSpec struct {
|
||||
Archive string // source archive volid to restore (resolved by the caller)
|
||||
SourceTier string // "local" this slice (pbs = Phase B) — for the report
|
||||
RestoreStorage string // target storage for the restored rootfs (e.g. "local-lvm")
|
||||
ScratchMin int // inclusive scratch VMID band (must be > 0)
|
||||
ScratchMax int // inclusive
|
||||
BootTimeout time.Duration // 0 → DefaultBootTimeout
|
||||
}
|
||||
|
||||
// RestoreTestResult is the reconcile-local outcome (the backup package maps it to the
|
||||
// hub.RestoreTest wire record — reconcile must not import hub for this).
|
||||
type RestoreTestResult struct {
|
||||
Archive string
|
||||
SourceTier string
|
||||
ScratchVMID int
|
||||
Pass bool
|
||||
Verified string // "boot+running" this slice
|
||||
Skipped bool // no free scratch VMID in band → test not run
|
||||
Err error
|
||||
StartedAt time.Time
|
||||
Duration time.Duration
|
||||
}
|
||||
|
||||
// IntentForScratchDestroy builds the benign teardown intent for an agent-owned scratch
|
||||
// guest: ClassGuestDestroy made benign by AgentTaggedScratch provenance (classify.go). The
|
||||
// gate authorizes it unsigned but is genuinely in-path (wrong provenance → pending_signature).
|
||||
func IntentForScratchDestroy(hostID string, vmid int) Intent {
|
||||
return Intent{
|
||||
Class: ClassGuestDestroy,
|
||||
HostID: hostID,
|
||||
GuestID: strconv.Itoa(vmid),
|
||||
VMID: vmid,
|
||||
Provenance: Provenance{AgentTaggedScratch: true}, // agent-internal, never hub-sourced
|
||||
Source: SourceOneShotJob,
|
||||
}
|
||||
}
|
||||
|
||||
// RunRestoreTest runs one restore-test on the per-guest queue lane of a fresh scratch VMID.
|
||||
// It journals a Scratch-owned entry BEFORE any mutation, so a crash anywhere after this
|
||||
// point is recoverable (Recover destroys the scratch guest). Teardown runs on EVERY path
|
||||
// (defer), including a failed verify. The returned Err is the TEST verdict's error (restore
|
||||
// or boot failure), independent of teardown success.
|
||||
func (e *Engine) RunRestoreTest(ctx context.Context, spec RestoreTestSpec) RestoreTestResult {
|
||||
now := time.Now().UTC()
|
||||
res := RestoreTestResult{Archive: spec.Archive, SourceTier: spec.SourceTier, StartedAt: now}
|
||||
|
||||
if spec.Archive == "" || spec.RestoreStorage == "" {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test needs an archive and a restore storage")
|
||||
return res
|
||||
}
|
||||
if spec.ScratchMin <= 0 || spec.ScratchMax < spec.ScratchMin {
|
||||
res.Err = fmt.Errorf("reconcile: invalid scratch VMID band [%d,%d]", spec.ScratchMin, spec.ScratchMax)
|
||||
return res
|
||||
}
|
||||
|
||||
lxc, err := e.api.ListLXC(ctx)
|
||||
if err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test list guests: %w", err)
|
||||
return res
|
||||
}
|
||||
vmid, ok := pickScratchVMID(lxc, spec.ScratchMin, spec.ScratchMax)
|
||||
if !ok {
|
||||
// Full band (e.g. an accumulation of un-torn-down scratch guests) → skip, never
|
||||
// panic or pick out-of-band. Recover will reap any genuinely leaked ones.
|
||||
e.logger.Warn("restore-test skipped: no free scratch VMID in band",
|
||||
"min", spec.ScratchMin, "max", spec.ScratchMax)
|
||||
res.Skipped = true
|
||||
return res
|
||||
}
|
||||
res.ScratchVMID = vmid
|
||||
|
||||
// Serialize on the scratch VMID's lane (inherits §10), and capture the result.
|
||||
ch := e.queue.Submit(vmid, func() error {
|
||||
e.runScratchTest(ctx, vmid, spec, &res)
|
||||
return res.Err
|
||||
})
|
||||
<-ch
|
||||
res.Duration = time.Since(now)
|
||||
return res
|
||||
}
|
||||
|
||||
// runScratchTest is the journaled body (runs on vmid's queue lane).
|
||||
func (e *Engine) runScratchTest(ctx context.Context, vmid int, spec RestoreTestSpec, res *RestoreTestResult) {
|
||||
base := JournalEntry{OpID: e.scratchOpID(vmid), VMID: vmid, Kind: scratchKind, Scratch: true}
|
||||
|
||||
// OWN the scratch guest's cleanup BEFORE any mutation. From here, a crash is recoverable.
|
||||
e.append(withState(base, OpStarted))
|
||||
|
||||
// Teardown ALWAYS runs (even on a failed verify). Uses a cancel-immune context so a
|
||||
// daemon shutdown mid-test still tears down; if teardown fails, the entry stays
|
||||
// in-flight and Recover reaps the guest on the next start.
|
||||
defer e.teardownScratch(ctx, base)
|
||||
|
||||
// 1. Restore into the fresh scratch VMID (benign create path). The UPID is for error
|
||||
// detection only — it does NOT make the Scratch entry terminal (teardown does).
|
||||
upid, err := e.api.RestoreLXC(ctx, proxmox.RestoreLXCOptions{
|
||||
VMID: vmid, Archive: spec.Archive, Storage: spec.RestoreStorage,
|
||||
})
|
||||
if err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test restore: %w", err)
|
||||
return
|
||||
}
|
||||
e.append(withUPID(base, upid, OpTaskRunning))
|
||||
if upid != "" {
|
||||
if _, err := e.api.WaitTask(ctx, upid, proxmox.WaitOptions{}); err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test restore task: %w", err)
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Net link-down on every interface BEFORE boot — test-safety so the clone (which
|
||||
// keeps the source MAC/hostname; identity-reset is slice 7) can't conflict with a
|
||||
// running source on L2/IP. Benign SetConfig.
|
||||
cfg, err := e.api.GuestConfig(ctx, vmid)
|
||||
if err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test read scratch config: %w", err)
|
||||
return
|
||||
}
|
||||
for key, val := range cfg.Nets() {
|
||||
if _, err := e.api.SetConfig(ctx, vmid, map[string]string{key: withLinkDown(val)}); err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test net link-down %s: %w", key, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Boot and verify it reaches running (basic liveness; deep app-health is slice 8).
|
||||
startUPID, err := e.api.Start(ctx, vmid)
|
||||
if err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test start: %w", err)
|
||||
return
|
||||
}
|
||||
if startUPID != "" {
|
||||
if _, err := e.api.WaitTask(ctx, startUPID, proxmox.WaitOptions{}); err != nil {
|
||||
res.Err = fmt.Errorf("reconcile: restore-test start task: %w", err)
|
||||
return
|
||||
}
|
||||
}
|
||||
if err := e.waitRunning(ctx, vmid, bootTimeout(spec)); err != nil {
|
||||
res.Err = err
|
||||
return
|
||||
}
|
||||
res.Pass = true
|
||||
res.Verified = "boot+running"
|
||||
}
|
||||
|
||||
// teardownScratch destroys the scratch guest (benign, gated) and records the entry terminal.
|
||||
// On any teardown failure it leaves the entry in-flight so Recover reaps the guest later.
|
||||
func (e *Engine) teardownScratch(ctx context.Context, base JournalEntry) {
|
||||
// Cancel-immune + bounded, so a shutdown mid-test still tears down.
|
||||
tctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 2*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
dec := e.gate.Authorize(IntentForScratchDestroy(e.hostID, base.VMID), nil)
|
||||
if !dec.Allowed {
|
||||
e.logger.Error("restore-test: scratch teardown refused by gate (unexpected); left for Recover",
|
||||
"vmid", base.VMID, "reason", dec.Reason)
|
||||
return
|
||||
}
|
||||
upid, err := e.api.DestroyLXC(tctx, base.VMID)
|
||||
if err != nil {
|
||||
e.logger.Error("restore-test: scratch teardown failed; left for Recover", "vmid", base.VMID, "err", err)
|
||||
return
|
||||
}
|
||||
if upid != "" {
|
||||
if _, err := e.api.WaitTask(tctx, upid, proxmox.WaitOptions{}); err != nil {
|
||||
e.logger.Error("restore-test: scratch teardown task failed; left for Recover", "vmid", base.VMID, "err", err)
|
||||
return
|
||||
}
|
||||
}
|
||||
e.append(withState(base, OpSucceeded))
|
||||
e.logger.Info("restore-test: scratch guest torn down", "vmid", base.VMID)
|
||||
}
|
||||
|
||||
// waitRunning polls GuestStatus until the guest is running or the timeout elapses. The poll
|
||||
// interval is 2s in production, but shrinks for short timeouts so it stays responsive.
|
||||
func (e *Engine) waitRunning(ctx context.Context, vmid int, timeout time.Duration) error {
|
||||
interval := 2 * time.Second
|
||||
if timeout < 4*interval {
|
||||
if interval = timeout / 4; interval < 10*time.Millisecond {
|
||||
interval = 10 * time.Millisecond
|
||||
}
|
||||
}
|
||||
deadline := time.Now().Add(timeout)
|
||||
t := time.NewTicker(interval)
|
||||
defer t.Stop()
|
||||
for {
|
||||
g, err := e.api.GuestStatus(ctx, vmid)
|
||||
if err == nil && g.Status == "running" {
|
||||
return nil
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
if err != nil {
|
||||
return fmt.Errorf("reconcile: restore-test verify: guest %d not running within %s (last err: %w)", vmid, timeout, err)
|
||||
}
|
||||
return fmt.Errorf("reconcile: restore-test verify: guest %d not running within %s", vmid, timeout)
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-t.C:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// pickScratchVMID returns the lowest free VMID in [min,max], excluding the standing 9999
|
||||
// scratch and any in-use guest. ok=false when the band is fully occupied (the test is then
|
||||
// skipped, never run out-of-band).
|
||||
func pickScratchVMID(lxc []proxmox.Guest, min, max int) (int, bool) {
|
||||
used := make(map[int]bool, len(lxc))
|
||||
for _, g := range lxc {
|
||||
used[g.VMID] = true
|
||||
}
|
||||
for id := min; id <= max; id++ {
|
||||
if id == 9999 || used[id] {
|
||||
continue
|
||||
}
|
||||
return id, true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// withLinkDown sets link_down=1 on a Proxmox netN config string, REPLACING any existing
|
||||
// link_down token (never blind-concatenating, so a re-applied/pre-set value can't produce a
|
||||
// malformed netN).
|
||||
func withLinkDown(netN string) string {
|
||||
parts := strings.Split(netN, ",")
|
||||
out := parts[:0]
|
||||
for _, p := range parts {
|
||||
if p == "" || strings.HasPrefix(p, "link_down=") {
|
||||
continue
|
||||
}
|
||||
out = append(out, p)
|
||||
}
|
||||
out = append(out, "link_down=1")
|
||||
return strings.Join(out, ",")
|
||||
}
|
||||
|
||||
func bootTimeout(spec RestoreTestSpec) time.Duration {
|
||||
if spec.BootTimeout > 0 {
|
||||
return spec.BootTimeout
|
||||
}
|
||||
return DefaultBootTimeout
|
||||
}
|
||||
|
||||
func (e *Engine) scratchOpID(vmid int) string {
|
||||
return "scratch-restore-" + strconv.Itoa(vmid) + "-" + nextSeq(&e.opSeq)
|
||||
}
|
||||
|
||||
// withState / withUPID build journal records from a base entry, preserving its identity +
|
||||
// Scratch flag.
|
||||
func withState(base JournalEntry, state OpState) JournalEntry {
|
||||
base.State = state
|
||||
base.At = time.Now().UTC()
|
||||
return base
|
||||
}
|
||||
func withUPID(base JournalEntry, upid string, state OpState) JournalEntry {
|
||||
base.UPID = upid
|
||||
base.State = state
|
||||
base.At = time.Now().UTC()
|
||||
return base
|
||||
}
|
||||
Reference in New Issue
Block a user