R-488: backup tests read ONE health-wait clock; package 444 s -> ~2 s
waitForHealthy's 3 s settle / 5 s poll / 90 s timeout become package vars (production values unchanged, pinned by TestR488_HealthWaitDefaultsAreProduction); TestMain shortens them once. Measured before: 51 of 599 tests held 442 of 444 s, two not-running fixtures 279 s alone. newOffboxManager now installs a failing SSH fake: TestOffbox_ConfirmedReset was spawning a real 'ssh felhom@nas.local' and waiting out its 10 s ConnectTimeout. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,50 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-488 — the package's ONE test clock for the post-restore health wait.
|
||||
//
|
||||
// Before this, every restore fixture slept waitForHealthy's 3 s settle for real, and every fixture
|
||||
// whose fake provider never reports "running" waited out the full 90 s timeout: measured 2026-10-05,
|
||||
// 51 of 599 tests took ≥ 1 s and together held 442 of the package's 444 s, two of them alone 279 s.
|
||||
// The production values are captured here BEFORE they are shortened, so the pin below reads what a
|
||||
// box runs, not what the tests run.
|
||||
var prodHealthSettle, prodHealthInterval, prodHealthTimeout = healthSettle, healthInterval, healthTimeout
|
||||
|
||||
func TestMain(m *testing.M) {
|
||||
healthSettle = 0
|
||||
healthInterval = 2 * time.Millisecond
|
||||
healthTimeout = 50 * time.Millisecond
|
||||
os.Exit(m.Run())
|
||||
}
|
||||
|
||||
// The shortening is a TEST seam only: a box must still give a restored stack 3 s to settle, poll
|
||||
// every 5 s, and wait 90 s before calling the restore unhealthy.
|
||||
func TestR488_HealthWaitDefaultsAreProduction(t *testing.T) {
|
||||
if prodHealthSettle != 3*time.Second || prodHealthInterval != 5*time.Second || prodHealthTimeout != 90*time.Second {
|
||||
t.Fatalf("production health wait changed: settle=%v interval=%v timeout=%v (want 3s/5s/1m30s) — "+
|
||||
"a restored app now gets a different grace on every box", prodHealthSettle, prodHealthInterval, prodHealthTimeout)
|
||||
}
|
||||
}
|
||||
|
||||
// The consequence the seam must not break: a stack that never comes up is still reported as a
|
||||
// failure (only sooner), and one that is up is accepted.
|
||||
func TestR488_WaitForHealthyStillJudges(t *testing.T) {
|
||||
m := &Manager{}
|
||||
m.stackProvider = &fakeRecoveryProvider{running: false}
|
||||
start := time.Now()
|
||||
if err := m.waitForHealthy("app", healthTimeout); err == nil {
|
||||
t.Fatal("a stack that never reaches running must fail the health wait")
|
||||
}
|
||||
if d := time.Since(start); d > 5*time.Second {
|
||||
t.Fatalf("the shortened clock is not in force: the wait took %v", d)
|
||||
}
|
||||
m.stackProvider = &fakeRecoveryProvider{running: true}
|
||||
if err := m.waitForHealthy("app", healthTimeout); err != nil {
|
||||
t.Fatalf("a running stack must pass: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -924,7 +924,7 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
|
||||
// so without this line a successful off-site restore would leave the app refusing its next start.
|
||||
// Pinned by TestR475_OffsiteRestoreClearsAnUpdateHold.
|
||||
m.clearUpdateHoldAfterRestore(stack)
|
||||
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
|
||||
if err := m.waitForHealthy(stack, healthTimeout); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
|
||||
}
|
||||
|
||||
|
||||
@@ -40,6 +40,13 @@ func newOffboxManager(t *testing.T) (*Manager, *settings.Settings) {
|
||||
if err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// R-488: no test reaches a real ssh. Until 2026-10-05 the default runner was left in place, so a
|
||||
// run that reached the set-aside path (TestOffbox_ConfirmedReset's first-night run) spawned a real
|
||||
// `ssh felhom@nas.local` from DooPlex and waited out its 10 s ConnectTimeout. This fake fails the
|
||||
// way an unreachable host does — immediately — and a test that needs ssh to work installs its own.
|
||||
m.SetOffboxSSH(func(context.Context, string, string, int, string, string, string) ([]byte, error) {
|
||||
return []byte("ssh: test harness: no network"), errors.New("exit status 255")
|
||||
})
|
||||
return m, sett
|
||||
}
|
||||
|
||||
|
||||
@@ -87,7 +87,7 @@ func (m *Manager) RestoreApp(stackName, snapshotID string) error {
|
||||
}
|
||||
|
||||
// Verify app started successfully
|
||||
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
|
||||
if err := m.waitForHealthy(stackName, healthTimeout); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] Restore completed but app health check failed: %v", err)
|
||||
}
|
||||
|
||||
@@ -203,13 +203,23 @@ func (m *Manager) restoreDockerVolumesFrom(stackName, dumpDir string) (int, erro
|
||||
return restored, nil
|
||||
}
|
||||
|
||||
// waitForHealthy's clock (R-488). These are the PRODUCTION values; they are variables only so the
|
||||
// package's tests can shorten them in ONE place (main_test.go TestMain) instead of every restore
|
||||
// test sleeping 3 s of settle and every not-running fixture waiting out the full 90 s — that was
|
||||
// ~280 s of the package's ~440 s. TestR488_HealthWaitDefaultsAreProduction pins the values.
|
||||
var (
|
||||
healthSettle = 3 * time.Second // initial settling time before the first poll
|
||||
healthInterval = 5 * time.Second // between polls
|
||||
healthTimeout = 90 * time.Second // how long a restored stack gets to reach running
|
||||
)
|
||||
|
||||
// waitForHealthy waits for a stack to reach running state after restore.
|
||||
// Forces a docker ps refresh on each poll to avoid stale state.
|
||||
func (m *Manager) waitForHealthy(stackName string, timeout time.Duration) error {
|
||||
deadline := time.Now().Add(timeout)
|
||||
interval := 5 * time.Second
|
||||
interval := healthInterval
|
||||
|
||||
time.Sleep(3 * time.Second) // initial settling time
|
||||
time.Sleep(healthSettle) // initial settling time
|
||||
|
||||
for time.Now().Before(deadline) {
|
||||
if m.stackProvider == nil {
|
||||
|
||||
@@ -468,7 +468,7 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
|
||||
if err := m.stackProvider.StartStack(stackName); err != nil {
|
||||
return res, fmt.Errorf("starting %s after restore from unit: %w", stackName, err)
|
||||
}
|
||||
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
|
||||
if err := m.waitForHealthy(stackName, healthTimeout); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err)
|
||||
}
|
||||
|
||||
|
||||
@@ -440,7 +440,7 @@ func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err er
|
||||
if startErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr)
|
||||
}
|
||||
if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil {
|
||||
if healthErr := m.waitForHealthy(stackName, healthTimeout); healthErr != nil {
|
||||
m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr)
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user