diff --git a/REUSE.md b/REUSE.md index b6be94a..b27b9fe 100644 --- a/REUSE.md +++ b/REUSE.md @@ -322,6 +322,7 @@ |---|---|---|---| | `diskAgent` | controller/internal/web/storage_handlers.go | `*agentapi.Client` | `mockAgent` in controller/internal/web/storage_handlers_test.go | | `netAgent` + `Server.netAgentFn/netProbeFn/netListFn` | controller/internal/web/netstorage_job.go (+ server.go fields) | `*agentapi.Client` / `runNetProbe` (linux re-exec) / `agent.ListNetStorage` | `fakeNetAgent` + fn injections in controller/internal/web/netstorage_job_test.go — the NAS add orchestration never shells/TLS-dials in tests | +| `healthSettle` / `healthInterval` / `healthTimeout` (package vars, R-488) | controller/internal/backup/restore.go | production 3 s / 5 s / 90 s — `waitForHealthy`'s clock, used by every restore path | shortened ONCE in controller/internal/backup/main_test.go `TestMain` (0 / 2 ms / 50 ms; the package went 444 s → ~2 s); `TestR488_HealthWaitDefaultsAreProduction` pins the production values. A new real-clock wait in this package goes behind the same kind of var, not a per-test sleep. Also: `newOffboxManager` installs a failing `SetOffboxSSH` fake, so no backup test spawns a real `ssh` | | `Server.agentLogsFn` (func seam) | controller/internal/web/server.go | nil → `agentClient().DebugLogs` (agent GET /debug/logs) | injected in controller/internal/web/observability_test.go (incl. the pre-0.83 typed-404 notice path) | | `escrowAgent` + `Server.escrowAgentFn/escrowStageFn/escrowStaleFn` | controller/internal/web/escrow_handlers.go (+ server.go fields) | `*agentapi.Client` / `PushOffboxPasswordForEscrow` / `report.EscrowAutoConfirmer.StaleBlob` (SetEscrowStale) | `fakeEscrowAgent` + fn injections in escrow_wizard_test.go — call-ORDER assertions (stage BEFORE trigger) + agent-never-called gates. The claim leg is the ONLY surface R crosses: no-store, never logged, never templated | | `offboxCeremonyWaitState` + `escrowCeremonyGraceWindow` | controller/internal/web/handlers.go | pure pick: (awaiting, timedOut) from `OffboxTarget.{EscrowState,CeremonyCompletedAt}` — the v0.138.0 "megerősítésre vár" card. Stamp SET on claim (escrow_handlers.go), CLEARED on the flip (main.go Flip + offbox_handlers.go manual confirm) | escrow_wait_state_test.go truth table (escrowed/unstamped/unparseable → plain CTA; boundary via `>=`) | diff --git a/controller/internal/backup/main_test.go b/controller/internal/backup/main_test.go new file mode 100644 index 0000000..213eda3 --- /dev/null +++ b/controller/internal/backup/main_test.go @@ -0,0 +1,50 @@ +package backup + +import ( + "os" + "testing" + "time" +) + +// R-488 — the package's ONE test clock for the post-restore health wait. +// +// Before this, every restore fixture slept waitForHealthy's 3 s settle for real, and every fixture +// whose fake provider never reports "running" waited out the full 90 s timeout: measured 2026-10-05, +// 51 of 599 tests took ≥ 1 s and together held 442 of the package's 444 s, two of them alone 279 s. +// The production values are captured here BEFORE they are shortened, so the pin below reads what a +// box runs, not what the tests run. +var prodHealthSettle, prodHealthInterval, prodHealthTimeout = healthSettle, healthInterval, healthTimeout + +func TestMain(m *testing.M) { + healthSettle = 0 + healthInterval = 2 * time.Millisecond + healthTimeout = 50 * time.Millisecond + os.Exit(m.Run()) +} + +// The shortening is a TEST seam only: a box must still give a restored stack 3 s to settle, poll +// every 5 s, and wait 90 s before calling the restore unhealthy. +func TestR488_HealthWaitDefaultsAreProduction(t *testing.T) { + if prodHealthSettle != 3*time.Second || prodHealthInterval != 5*time.Second || prodHealthTimeout != 90*time.Second { + t.Fatalf("production health wait changed: settle=%v interval=%v timeout=%v (want 3s/5s/1m30s) — "+ + "a restored app now gets a different grace on every box", prodHealthSettle, prodHealthInterval, prodHealthTimeout) + } +} + +// The consequence the seam must not break: a stack that never comes up is still reported as a +// failure (only sooner), and one that is up is accepted. +func TestR488_WaitForHealthyStillJudges(t *testing.T) { + m := &Manager{} + m.stackProvider = &fakeRecoveryProvider{running: false} + start := time.Now() + if err := m.waitForHealthy("app", healthTimeout); err == nil { + t.Fatal("a stack that never reaches running must fail the health wait") + } + if d := time.Since(start); d > 5*time.Second { + t.Fatalf("the shortened clock is not in force: the wait took %v", d) + } + m.stackProvider = &fakeRecoveryProvider{running: true} + if err := m.waitForHealthy("app", healthTimeout); err != nil { + t.Fatalf("a running stack must pass: %v", err) + } +} diff --git a/controller/internal/backup/offbox_reconstitute.go b/controller/internal/backup/offbox_reconstitute.go index f2f3c2e..f2de636 100644 --- a/controller/internal/backup/offbox_reconstitute.go +++ b/controller/internal/backup/offbox_reconstitute.go @@ -924,7 +924,7 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack // so without this line a successful off-site restore would leave the app refusing its next start. // Pinned by TestR475_OffsiteRestoreClearsAnUpdateHold. m.clearUpdateHoldAfterRestore(stack) - if err := m.waitForHealthy(stack, 90*time.Second); err != nil { + if err := m.waitForHealthy(stack, healthTimeout); err != nil { m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err) } diff --git a/controller/internal/backup/offbox_test.go b/controller/internal/backup/offbox_test.go index 044b7ea..a785ce8 100644 --- a/controller/internal/backup/offbox_test.go +++ b/controller/internal/backup/offbox_test.go @@ -40,6 +40,13 @@ func newOffboxManager(t *testing.T) (*Manager, *settings.Settings) { if err := m.WriteOffboxSecrets("PRIVATE-KEY-MATERIAL", "nas.local ssh-ed25519 AAAAhostkey"); err != nil { t.Fatal(err) } + // R-488: no test reaches a real ssh. Until 2026-10-05 the default runner was left in place, so a + // run that reached the set-aside path (TestOffbox_ConfirmedReset's first-night run) spawned a real + // `ssh felhom@nas.local` from DooPlex and waited out its 10 s ConnectTimeout. This fake fails the + // way an unreachable host does — immediately — and a test that needs ssh to work installs its own. + m.SetOffboxSSH(func(context.Context, string, string, int, string, string, string) ([]byte, error) { + return []byte("ssh: test harness: no network"), errors.New("exit status 255") + }) return m, sett } diff --git a/controller/internal/backup/restore.go b/controller/internal/backup/restore.go index e81fa75..2be92f5 100644 --- a/controller/internal/backup/restore.go +++ b/controller/internal/backup/restore.go @@ -87,7 +87,7 @@ func (m *Manager) RestoreApp(stackName, snapshotID string) error { } // Verify app started successfully - if err := m.waitForHealthy(stackName, 90*time.Second); err != nil { + if err := m.waitForHealthy(stackName, healthTimeout); err != nil { m.logger.Printf("[WARN] [backup] Restore completed but app health check failed: %v", err) } @@ -203,13 +203,23 @@ func (m *Manager) restoreDockerVolumesFrom(stackName, dumpDir string) (int, erro return restored, nil } +// waitForHealthy's clock (R-488). These are the PRODUCTION values; they are variables only so the +// package's tests can shorten them in ONE place (main_test.go TestMain) instead of every restore +// test sleeping 3 s of settle and every not-running fixture waiting out the full 90 s — that was +// ~280 s of the package's ~440 s. TestR488_HealthWaitDefaultsAreProduction pins the values. +var ( + healthSettle = 3 * time.Second // initial settling time before the first poll + healthInterval = 5 * time.Second // between polls + healthTimeout = 90 * time.Second // how long a restored stack gets to reach running +) + // waitForHealthy waits for a stack to reach running state after restore. // Forces a docker ps refresh on each poll to avoid stale state. func (m *Manager) waitForHealthy(stackName string, timeout time.Duration) error { deadline := time.Now().Add(timeout) - interval := 5 * time.Second + interval := healthInterval - time.Sleep(3 * time.Second) // initial settling time + time.Sleep(healthSettle) // initial settling time for time.Now().Before(deadline) { if m.stackProvider == nil { diff --git a/controller/internal/backup/restore_unit.go b/controller/internal/backup/restore_unit.go index c1f106f..a874cb8 100644 --- a/controller/internal/backup/restore_unit.go +++ b/controller/internal/backup/restore_unit.go @@ -468,7 +468,7 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U if err := m.stackProvider.StartStack(stackName); err != nil { return res, fmt.Errorf("starting %s after restore from unit: %w", stackName, err) } - if err := m.waitForHealthy(stackName, 90*time.Second); err != nil { + if err := m.waitForHealthy(stackName, healthTimeout); err != nil { m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err) } diff --git a/controller/internal/backup/tier2_restore.go b/controller/internal/backup/tier2_restore.go index da56ca7..e032455 100644 --- a/controller/internal/backup/tier2_restore.go +++ b/controller/internal/backup/tier2_restore.go @@ -440,7 +440,7 @@ func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err er if startErr != nil { m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr) } - if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil { + if healthErr := m.waitForHealthy(stackName, healthTimeout); healthErr != nil { m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr) }