R-361 follow-on: the undo-copy prune must run above the already-current check
gates / gates (push) Successful in 11s

Excluding pre-restore-* from db_dumps made that list stable across restores, so
CaptureRecoveryUnit's already-current early return began firing where it never
had - and the prune, which sat after it, stopped running in exactly the case it
exists for. Measured on demo-hp minutes after the change: four undo copies on
disk against a cap of three.

The prune is housekeeping on the dump directory and is independent of whether
the manifest needs rewriting, so it belongs above the check. Pruning cannot
disturb dbDumps, which no longer contains those names.

Pinned by TestR361_UndoCapHoldsWhenTheUnitIsAlreadyCurrent; its red-proof moves
the call back below the return and the cap fails at 5.
This commit is contained in:
2026-08-23 00:02:32 +02:00
parent 968c968559
commit 810b18ab8e
3 changed files with 80 additions and 1 deletions
@@ -234,3 +234,61 @@ func TestR361_FilterOutUndoCopies(t *testing.T) {
t.Error("nil must yield an empty list, not a panic")
}
}
// The undo-copy cap must hold even when the unit is ALREADY CURRENT.
//
// FOUND ON THE LIVE BOX, 2026-08-22, immediately after excluding `pre-restore-*` from `db_dumps`:
// four undo copies on disk against a cap of three. Excluding them made the list stable across
// restores, so `CaptureRecoveryUnit`'s already-current early return began firing — and the prune,
// which sat after it, stopped running in exactly the case it exists for. One change made the other
// unreachable, and only counting files on a real box showed it.
func TestR361_UndoCapHoldsWhenTheUnitIsAlreadyCurrent(t *testing.T) {
tmp := t.TempDir()
stackDir := filepath.Join(tmp, "stack")
drive := filepath.Join(tmp, "drive")
if err := os.MkdirAll(stackDir, 0o755); err != nil {
t.Fatal(err)
}
mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"), "services:\n app:\n image: ex/app:1\n")
mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: Ex\n")
mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: ex\n")
dumps := AppDBDumpPath(drive, "ex")
mustWrite(t, filepath.Join(dumps, "ex-postgres.sql"), "-- own")
m := &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(tmp, "system"),
stackProvider: &fakeRecoveryProvider{info: RecoveryInfo{
StackDir: stackDir, DisplayName: "Ex", ImagePins: []string{"ex/app:1"},
NonSecretEnv: map[string]string{"SUBDOMAIN": "ex", "HDD_PATH": drive},
}, hdd: drive},
version: "vtest",
}
// First capture writes the manifest.
if err := m.CaptureRecoveryUnit("ex"); err != nil {
t.Fatal(err)
}
// Now the unit IS current. Drop five undo copies in, as five restores would.
for _, st := range []string{"20260101T000000Z", "20260201T000000Z", "20260301T000000Z", "20260401T000000Z", "20260501T000000Z"} {
mustWrite(t, filepath.Join(dumps, preRestoreDumpPrefix+st+"-ex-postgres.sql"), "-- undo")
}
// A capture that will take the already-current path — nothing else changed.
if err := m.CaptureRecoveryUnit("ex"); err != nil {
t.Fatal(err)
}
n := 0
ents, _ := os.ReadDir(dumps)
for _, e := range ents {
if strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
n++
}
}
if n != maxUndoCopiesPerApp {
t.Fatalf("the cap must hold on the already-current path too: %d undo copies survived, want %d", n, maxUndoCopiesPerApp)
}
// And the app's own dump is not a prune candidate.
if _, err := os.Stat(filepath.Join(dumps, "ex-postgres.sql")); err != nil {
t.Fatal("the app's own dump was pruned")
}
}