reconcile: tier-aware restore-task deadline (S4.1 unattended offsite restore-test)
A WAN (pbs-tier) restore of a large guest exceeds the restore-task wait's 10m default → the wait expired mid-restore, teardown fired against a still-restoring (not-yet-pool-associated) scratch guest → leak + a phantom VM.Allocate 403. - RestoreTestSpec.RestoreTaskTimeout (0→10m default); the restore WaitTask passes it. Local tier unchanged (10m). - config RestoreTestPBSRestoreTimeoutSeconds + accessor (default 120m). - main restoreTaskTimeout(cfg,tier): configured PBS timeout only when tier==pbs, else 0. Both scheduler + selftest spec builds. - Tests + WaitOptions red-proof + accessor contract. The "grant scratch-band VM.Allocate" follow-up is diagnosed not blind-applied: the scratch is restored INTO /pool/felhom (ACL already grants VM.Allocate), so the earlier 403 was a consequence of the timeout. No ACL/host-install change. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01PSK5g6qYLknKj8u3QAFEr6
This commit is contained in:
@@ -56,6 +56,49 @@ func TestRunRestoreTest_PassAndTeardown(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunRestoreTest_TierAwareRestoreTimeout (S4.1) pins that the restore-task wait carries the
|
||||
// tier-derived timeout: the configured (generous) value for a pbs/WAN restore, and 0 (→ the 10m
|
||||
// WaitOptions default, UNCHANGED) for a local restore. Red-proof: revert the L246 wait to
|
||||
// WaitOptions{} → the pbs assertion (120m) fails.
|
||||
func TestRunRestoreTest_TierAwareRestoreTimeout(t *testing.T) {
|
||||
const restoreUPID = "UPID:node:1:2:3:4:vzrestore:990000:tok:" // async restore → the wait fires
|
||||
|
||||
restoreWaitTimeout := func(api *fakeAPI) (time.Duration, bool) {
|
||||
for i, u := range api.waits {
|
||||
if u == restoreUPID {
|
||||
return api.waitOpts[i].Timeout, true
|
||||
}
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// pbs tier → the configured generous timeout is passed to WaitTask.
|
||||
pbsAPI := &fakeAPI{cfg: map[int]proxmox.GuestConfig{990000: scratchCfg()}, restoreUPID: restoreUPID}
|
||||
e, _, q := newEngine(t, pbsAPI, EmptyProvider{})
|
||||
defer q.Close()
|
||||
e.RunRestoreTest(context.Background(), RestoreTestSpec{
|
||||
Archive: "felhom-offsite:backup/ct/9201/x", RestoreStorage: "local-lvm",
|
||||
ScratchMin: 990000, ScratchMax: 990009, SourceTier: "pbs",
|
||||
RestoreTaskTimeout: 120 * time.Minute,
|
||||
})
|
||||
if to, ok := restoreWaitTimeout(pbsAPI); !ok || to != 120*time.Minute {
|
||||
t.Errorf("pbs restore wait Timeout = %v (found=%v), want 120m", to, ok)
|
||||
}
|
||||
|
||||
// local tier → 0 (→ WaitOptions' 10m default preserved, UNCHANGED).
|
||||
localAPI := &fakeAPI{cfg: map[int]proxmox.GuestConfig{990000: scratchCfg()}, restoreUPID: restoreUPID}
|
||||
e2, _, q2 := newEngine(t, localAPI, EmptyProvider{})
|
||||
defer q2.Close()
|
||||
e2.RunRestoreTest(context.Background(), RestoreTestSpec{
|
||||
Archive: "local:backup/x.tar.zst", RestoreStorage: "local-lvm",
|
||||
ScratchMin: 990000, ScratchMax: 990009, SourceTier: "local",
|
||||
RestoreTaskTimeout: 0,
|
||||
})
|
||||
if to, ok := restoreWaitTimeout(localAPI); !ok || to != 0 {
|
||||
t.Errorf("local restore wait Timeout = %v (found=%v), want 0 (→10m default)", to, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// startWarnAPI builds a fakeAPI whose guest-start task exits "WARNINGS: 1" and whose start
|
||||
// task log contains the given warning lines. The guest reaches running (status default).
|
||||
func startWarnAPI(startUPID string, logLines []string) *fakeAPI {
|
||||
|
||||
Reference in New Issue
Block a user