R-488: backup tests read ONE health-wait clock; package 444 s -> ~2 s

waitForHealthy's 3 s settle / 5 s poll / 90 s timeout become package vars (production values
unchanged, pinned by TestR488_HealthWaitDefaultsAreProduction); TestMain shortens them once.
Measured before: 51 of 599 tests held 442 of 444 s, two not-running fixtures 279 s alone.
newOffboxManager now installs a failing SSH fake: TestOffbox_ConfirmedReset was spawning a
real 'ssh felhom@nas.local' and waiting out its 10 s ConnectTimeout.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 22:23:40 +02:00
parent 391967bcf1
commit 8ce496f124
7 changed files with 74 additions and 6 deletions
+50
View File
@@ -0,0 +1,50 @@
package backup
import (
"os"
"testing"
"time"
)
// R-488 — the package's ONE test clock for the post-restore health wait.
//
// Before this, every restore fixture slept waitForHealthy's 3 s settle for real, and every fixture
// whose fake provider never reports "running" waited out the full 90 s timeout: measured 2026-10-05,
// 51 of 599 tests took ≥ 1 s and together held 442 of the package's 444 s, two of them alone 279 s.
// The production values are captured here BEFORE they are shortened, so the pin below reads what a
// box runs, not what the tests run.
var prodHealthSettle, prodHealthInterval, prodHealthTimeout = healthSettle, healthInterval, healthTimeout
func TestMain(m *testing.M) {
healthSettle = 0
healthInterval = 2 * time.Millisecond
healthTimeout = 50 * time.Millisecond
os.Exit(m.Run())
}
// The shortening is a TEST seam only: a box must still give a restored stack 3 s to settle, poll
// every 5 s, and wait 90 s before calling the restore unhealthy.
func TestR488_HealthWaitDefaultsAreProduction(t *testing.T) {
if prodHealthSettle != 3*time.Second || prodHealthInterval != 5*time.Second || prodHealthTimeout != 90*time.Second {
t.Fatalf("production health wait changed: settle=%v interval=%v timeout=%v (want 3s/5s/1m30s) — "+
"a restored app now gets a different grace on every box", prodHealthSettle, prodHealthInterval, prodHealthTimeout)
}
}
// The consequence the seam must not break: a stack that never comes up is still reported as a
// failure (only sooner), and one that is up is accepted.
func TestR488_WaitForHealthyStillJudges(t *testing.T) {
m := &Manager{}
m.stackProvider = &fakeRecoveryProvider{running: false}
start := time.Now()
if err := m.waitForHealthy("app", healthTimeout); err == nil {
t.Fatal("a stack that never reaches running must fail the health wait")
}
if d := time.Since(start); d > 5*time.Second {
t.Fatalf("the shortened clock is not in force: the wait took %v", d)
}
m.stackProvider = &fakeRecoveryProvider{running: true}
if err := m.waitForHealthy("app", healthTimeout); err != nil {
t.Fatalf("a running stack must pass: %v", err)
}
}
@@ -924,7 +924,7 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
// so without this line a successful off-site restore would leave the app refusing its next start.
// Pinned by TestR475_OffsiteRestoreClearsAnUpdateHold.
m.clearUpdateHoldAfterRestore(stack)
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
if err := m.waitForHealthy(stack, healthTimeout); err != nil {
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
}
@@ -40,6 +40,13 @@ func newOffboxManager(t *testing.T) (*Manager, *settings.Settings) {
if err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey"); err != nil {
t.Fatal(err)
}
// R-488: no test reaches a real ssh. Until 2026-10-05 the default runner was left in place, so a
// run that reached the set-aside path (TestOffbox_ConfirmedReset's first-night run) spawned a real
// `ssh felhom@nas.local` from DooPlex and waited out its 10 s ConnectTimeout. This fake fails the
// way an unreachable host does — immediately — and a test that needs ssh to work installs its own.
m.SetOffboxSSH(func(context.Context, string, string, int, string, string, string) ([]byte, error) {
return []byte("ssh: test harness: no network"), errors.New("exit status 255")
})
return m, sett
}
+13 -3
View File
@@ -87,7 +87,7 @@ func (m *Manager) RestoreApp(stackName, snapshotID string) error {
}
// Verify app started successfully
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
if err := m.waitForHealthy(stackName, healthTimeout); err != nil {
m.logger.Printf("[WARN] [backup] Restore completed but app health check failed: %v", err)
}
@@ -203,13 +203,23 @@ func (m *Manager) restoreDockerVolumesFrom(stackName, dumpDir string) (int, erro
return restored, nil
}
// waitForHealthy's clock (R-488). These are the PRODUCTION values; they are variables only so the
// package's tests can shorten them in ONE place (main_test.go TestMain) instead of every restore
// test sleeping 3 s of settle and every not-running fixture waiting out the full 90 s — that was
// ~280 s of the package's ~440 s. TestR488_HealthWaitDefaultsAreProduction pins the values.
var (
healthSettle = 3 * time.Second // initial settling time before the first poll
healthInterval = 5 * time.Second // between polls
healthTimeout = 90 * time.Second // how long a restored stack gets to reach running
)
// waitForHealthy waits for a stack to reach running state after restore.
// Forces a docker ps refresh on each poll to avoid stale state.
func (m *Manager) waitForHealthy(stackName string, timeout time.Duration) error {
deadline := time.Now().Add(timeout)
interval := 5 * time.Second
interval := healthInterval
time.Sleep(3 * time.Second) // initial settling time
time.Sleep(healthSettle) // initial settling time
for time.Now().Before(deadline) {
if m.stackProvider == nil {
+1 -1
View File
@@ -468,7 +468,7 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
if err := m.stackProvider.StartStack(stackName); err != nil {
return res, fmt.Errorf("starting %s after restore from unit: %w", stackName, err)
}
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
if err := m.waitForHealthy(stackName, healthTimeout); err != nil {
m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err)
}
+1 -1
View File
@@ -440,7 +440,7 @@ func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err er
if startErr != nil {
m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr)
}
if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil {
if healthErr := m.waitForHealthy(stackName, healthTimeout); healthErr != nil {
m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr)
}