Files
admin 810b18ab8e
gates / gates (push) Successful in 11s
R-361 follow-on: the undo-copy prune must run above the already-current check
Excluding pre-restore-* from db_dumps made that list stable across restores, so
CaptureRecoveryUnit's already-current early return began firing where it never
had - and the prune, which sat after it, stopped running in exactly the case it
exists for. Measured on demo-hp minutes after the change: four undo copies on
disk against a cap of three.

The prune is housekeeping on the dump directory and is independent of whether
the manifest needs rewriting, so it belongs above the check. Pruning cannot
disturb dbDumps, which no longer contains those names.

Pinned by TestR361_UndoCapHoldsWhenTheUnitIsAlreadyCurrent; its red-proof moves
the call back below the return and the cap fails at 5.
2026-08-23 00:02:32 +02:00

295 lines
12 KiB
Go

package backup
import (
"context"
"crypto/sha256"
"encoding/hex"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
)
// R-361. Taking the undo copy DESTROYED the app's own database backup.
//
// `DumpOne` writes `<stack>-<dbtype>.sql` — the app's canonical dump, the name the replay loop
// matches exactly. `writeSafetyDump` called it into the app's OWN unit directory and renamed the
// result to `pre-restore-*` afterwards, so every restore overwrote the app's real backup and then
// moved it away. The app was left with no database backup at all until the next nightly run, and a
// local restore-from-unit in that window tells the customer the app never had a database.
//
// A comment at `offbox_reconstitute.go` asserted the rename meant it "can never overwrite the app's
// real dump". It was false as written. Measured live on demo-hp 2026-08-22: `docmost` and
// `bookstack` each held only `pre-restore-*` files and no canonical dump.
//
// THE DANGEROUS LOOKALIKE, named so nobody writes it by accident: a test asserting "the pre-restore
// file exists" passes just as well when the canonical dump was destroyed. The assertion that
// convicts is THE CANONICAL DUMP'S CONTENT, UNCHANGED.
func sha256File(t *testing.T, p string) string {
t.Helper()
b, err := os.ReadFile(p)
if err != nil {
t.Fatalf("reading %s: %v", p, err)
}
sum := sha256.Sum256(b)
return hex.EncodeToString(sum[:])
}
// The whole of R-361, in one comparison.
func TestR361_SafetyDumpLeavesTheCanonicalDumpByteIdentical(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
db := DiscoveredDB{StackName: "app", DBType: DBTypePostgres, ContainerName: "app-postgres", ContainerID: "cid"}
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{db}, nil }
// The app's OWN dump, already on disk — the thing a local restore reads.
dumpDir := AppDBDumpPath(nsRoot, "app")
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
t.Fatal(err)
}
canonical := filepath.Join(dumpDir, "app-postgres.sql")
const ownContent = "-- THE APP'S OWN NIGHTLY DUMP\nCREATE TABLE nightly();\n"
if err := os.WriteFile(canonical, []byte(ownContent), 0o644); err != nil {
t.Fatal(err)
}
before := sha256File(t, canonical)
// The real dump seam is NOT injected here on purpose for the destination question — we inject a
// writer that behaves like DumpOneTo (writes wherever it is told), so the test measures WHICH PATH
// writeSafetyDump asks for. That is the layer the defect lives in.
var askedFor []string
m.safetyDumpFn = func(_ context.Context, d DiscoveredDB, finalPath string) DumpResult {
askedFor = append(askedFor, finalPath)
if err := os.MkdirAll(filepath.Dir(finalPath), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(finalPath, []byte("-- THE UNDO COPY\n"), 0o644); err != nil {
t.Fatal(err)
}
return DumpResult{DB: d, FilePath: finalPath, Size: 17}
}
set, err := m.writeSafetyDump(context.Background(), "app", nsRoot)
if err != nil {
t.Fatalf("writeSafetyDump: %v", err)
}
// THE ASSERTION THAT CONVICTS.
if _, err := os.Stat(canonical); err != nil {
t.Fatalf("the app's own dump is GONE after taking an undo copy: %v", err)
}
if after := sha256File(t, canonical); after != before {
t.Fatalf("the app's own dump CHANGED across the safety dump.\n before %s\n after %s\n"+
"The undo must never be written to the canonical name.", before, after)
}
// And the destination asked for was never the canonical name — the layer the defect lived at.
for _, p := range askedFor {
if filepath.Base(p) == "app-postgres.sql" {
t.Errorf("writeSafetyDump asked for the CANONICAL name %q — that is the defect itself", p)
}
if !strings.HasPrefix(filepath.Base(p), preRestoreDumpPrefix) {
t.Errorf("the undo copy must be asked for under its own prefix, got %q", filepath.Base(p))
}
}
// The undo exists beside it, under its own name.
if len(set.Files) != 1 {
t.Fatalf("one database → one undo copy, got %d", len(set.Files))
}
if _, err := os.Stat(set.Files[0].Path); err != nil {
t.Fatalf("the undo copy is not on disk: %v", err)
}
}
// Scenario B — two databases: BOTH canonical dumps survive, and the scratch files cannot collide.
func TestR361_TwoDatabases_BothCanonicalDumpsSurvive(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
pg := DiscoveredDB{StackName: "app", DBType: DBTypePostgres, ContainerName: "app-postgres", ContainerID: "a"}
my := DiscoveredDB{StackName: "app", DBType: DBTypeMariaDB, ContainerName: "app-maria", ContainerID: "b"}
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{pg, my}, nil }
dumpDir := AppDBDumpPath(nsRoot, "app")
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
t.Fatal(err)
}
want := map[string]string{}
for _, n := range []string{"app-postgres.sql", "app-mariadb.sql"} {
p := filepath.Join(dumpDir, n)
if err := os.WriteFile(p, []byte("-- own dump of "+n+"\n"), 0o644); err != nil {
t.Fatal(err)
}
want[n] = sha256File(t, p)
}
var asked []string
m.safetyDumpFn = func(_ context.Context, d DiscoveredDB, finalPath string) DumpResult {
asked = append(asked, finalPath)
_ = os.WriteFile(finalPath, []byte("-- undo\n"), 0o644)
return DumpResult{DB: d, FilePath: finalPath, Size: 8}
}
set, err := m.writeSafetyDump(context.Background(), "app", nsRoot)
if err != nil {
t.Fatal(err)
}
for n, w := range want {
got := sha256File(t, filepath.Join(dumpDir, n))
if got != w {
t.Errorf("%s changed across the safety dump (%s → %s) — one database's own backup was destroyed", n, w, got)
}
}
if len(set.Files) != 2 {
t.Fatalf("two databases → two undo copies, got %d", len(set.Files))
}
// The two undo destinations must differ, or one would overwrite the other.
if len(asked) == 2 && asked[0] == asked[1] {
t.Fatalf("both databases were dumped to the SAME path %q", asked[0])
}
}
// The scratch file is derived from the FINAL path, so a nightly dump and a safety dump running into
// the same directory cannot share it. Asserted at the layer that builds it.
func TestR361_ScratchFileCannotCollideWithTheNightlyDump(t *testing.T) {
dir := t.TempDir()
canonical := filepath.Join(dir, "app-postgres.sql")
undo := filepath.Join(dir, preRestoreDumpPrefix+"20260822T211114Z-app-postgres.sql")
// The contract DumpOneTo implements: tmp = final + ".tmp".
if canonical+".tmp" == undo+".tmp" {
t.Fatal("the two scratch paths are equal — a nightly dump and a safety dump would share a file")
}
if filepath.Base(undo+".tmp") == filepath.Base(canonical+".tmp") {
t.Fatalf("scratch basenames collide: %q vs %q", filepath.Base(undo+".tmp"), filepath.Base(canonical+".tmp"))
}
}
// Part 1.3 — the manifest lists the app's OWN dumps and NOT the undo copies.
//
// BEHAVIOURAL, through the real CaptureRecoveryUnit, because the decision is about what ships
// off-site and a source-level check would not prove the manifest.
func TestR361_ManifestExcludesUndoCopies(t *testing.T) {
tmp := t.TempDir()
stackDir := filepath.Join(tmp, "stack")
drive := filepath.Join(tmp, "drive")
if err := os.MkdirAll(stackDir, 0o755); err != nil {
t.Fatal(err)
}
mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"), "services:\n app:\n image: ex/app:1\n")
mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: Ex\n")
mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: ex\n")
// The app's own dump, plus three undo copies of the shape a restore leaves behind.
dumps := AppDBDumpPath(drive, "ex")
mustWrite(t, filepath.Join(dumps, "ex-postgres.sql"), "-- the app's own dump")
for _, st := range []string{"20260822T160544Z", "20260822T162347Z", "20260822T162708Z"} {
mustWrite(t, filepath.Join(dumps, preRestoreDumpPrefix+st+"-ex-postgres.sql"), "-- undo")
}
m := &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(tmp, "system"),
stackProvider: &fakeRecoveryProvider{info: RecoveryInfo{
StackDir: stackDir, DisplayName: "Ex", ImagePins: []string{"ex/app:1"},
NonSecretEnv: map[string]string{"SUBDOMAIN": "ex", "HDD_PATH": drive},
}, hdd: drive},
version: "vtest",
}
if err := m.CaptureRecoveryUnit("ex"); err != nil {
t.Fatalf("capture: %v", err)
}
man := readManifest(RecoveryUnitManifestPath(drive, "ex"))
if man == nil {
t.Fatal("no manifest written")
}
if len(man.DBDumps) != 1 || man.DBDumps[0] != "ex-postgres.sql" {
t.Fatalf("db_dumps must list ONLY the app's own dump, got %v", man.DBDumps)
}
// POSITIVE CONTROL for the absence: the undo copies really are on disk, so "not listed" is the
// manifest excluding them rather than the fixture never creating them.
n := 0
ents, _ := os.ReadDir(dumps)
for _, e := range ents {
if strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
n++
}
}
if n != 3 {
t.Fatalf("fixture: expected 3 undo copies on disk, found %d — the exclusion above would prove nothing", n)
}
}
// filterOutUndoCopies, exactly.
func TestR361_FilterOutUndoCopies(t *testing.T) {
in := []string{"app-postgres.sql", preRestoreDumpPrefix + "20260822T160544Z-app-postgres.sql", "app-mariadb.sql"}
got := filterOutUndoCopies(in)
if len(got) != 2 || got[0] != "app-postgres.sql" || got[1] != "app-mariadb.sql" {
t.Fatalf("got %v, want the two canonical dumps only", got)
}
if len(filterOutUndoCopies(nil)) != 0 {
t.Error("nil must yield an empty list, not a panic")
}
}
// The undo-copy cap must hold even when the unit is ALREADY CURRENT.
//
// FOUND ON THE LIVE BOX, 2026-08-22, immediately after excluding `pre-restore-*` from `db_dumps`:
// four undo copies on disk against a cap of three. Excluding them made the list stable across
// restores, so `CaptureRecoveryUnit`'s already-current early return began firing — and the prune,
// which sat after it, stopped running in exactly the case it exists for. One change made the other
// unreachable, and only counting files on a real box showed it.
func TestR361_UndoCapHoldsWhenTheUnitIsAlreadyCurrent(t *testing.T) {
tmp := t.TempDir()
stackDir := filepath.Join(tmp, "stack")
drive := filepath.Join(tmp, "drive")
if err := os.MkdirAll(stackDir, 0o755); err != nil {
t.Fatal(err)
}
mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"), "services:\n app:\n image: ex/app:1\n")
mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: Ex\n")
mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: ex\n")
dumps := AppDBDumpPath(drive, "ex")
mustWrite(t, filepath.Join(dumps, "ex-postgres.sql"), "-- own")
m := &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(tmp, "system"),
stackProvider: &fakeRecoveryProvider{info: RecoveryInfo{
StackDir: stackDir, DisplayName: "Ex", ImagePins: []string{"ex/app:1"},
NonSecretEnv: map[string]string{"SUBDOMAIN": "ex", "HDD_PATH": drive},
}, hdd: drive},
version: "vtest",
}
// First capture writes the manifest.
if err := m.CaptureRecoveryUnit("ex"); err != nil {
t.Fatal(err)
}
// Now the unit IS current. Drop five undo copies in, as five restores would.
for _, st := range []string{"20260101T000000Z", "20260201T000000Z", "20260301T000000Z", "20260401T000000Z", "20260501T000000Z"} {
mustWrite(t, filepath.Join(dumps, preRestoreDumpPrefix+st+"-ex-postgres.sql"), "-- undo")
}
// A capture that will take the already-current path — nothing else changed.
if err := m.CaptureRecoveryUnit("ex"); err != nil {
t.Fatal(err)
}
n := 0
ents, _ := os.ReadDir(dumps)
for _, e := range ents {
if strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
n++
}
}
if n != maxUndoCopiesPerApp {
t.Fatalf("the cap must hold on the already-current path too: %d undo copies survived, want %d", n, maxUndoCopiesPerApp)
}
// And the app's own dump is not a prune candidate.
if _, err := os.Stat(filepath.Join(dumps, "ex-postgres.sql")); err != nil {
t.Fatal("the app's own dump was pruned")
}
}