Files
felhom-controller/controller/internal/backup/r379_rollback_test.go
T
admin 968c968559
gates / gates (push) Successful in 11s
R-361: the safety dump destroyed the app's own database backup
writeSafetyDump called DumpOne into the app's OWN unit dir and renamed the result
to pre-restore-* afterwards. DumpOne writes <stack>-<dbtype>.sql - the app's
canonical dump - so every safety dump overwrote the app's real backup and then
moved it away, leaving the app with no database backup until the next nightly
run. A local restore-from-unit in that window tells the customer the app never
had a database.

The comment beside it asserted the rename meant it 'can never overwrite the app's
real dump'. False as written, and believed for four months. Measured live before
the fix: docmost and bookstack each held only pre-restore-* files and no
canonical dump.

DumpOneTo takes the final path and derives its own .tmp from it. DumpOne keeps
its signature and calls it with the canonical name. writeSafetyDump asks for its
own name directly; the rename is gone; the comment now states the invariant and
how it is enforced.

db_dumps no longer lists the undo copies. All three consumers of Manifest.DBDumps
were grepped and named - all inside recovery_unit.go, none reads it for recovery.
The files are neither deleted nor hidden.

Tests 1485 -> 1493. FIVE red-proofs, TWO PASSED first time and both are reported:
the behavioural tests inject the dump seam so a mutation inside DumpOneTo was
invisible, and 1.3 had no test at all. Guards added at the layer each defect
lives in; both mutations then convicted.
2026-08-22 23:39:59 +02:00

419 lines
17 KiB
Go

package backup
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
)
// R-379 / R-380. When an off-site database replay failed, the app was restarted onto a HALF-WRITTEN
// database and the customer was shown the undo copy's filename — a file nothing in the product could
// apply. Measured live on demo-hp 2026-08-22: Postgres left emptied and crash-looping, MariaDB left
// partly applied while `docker inspect` reported health=healthy. Both are the same failure — a half
// state — and the only difference was whether it looked broken.
//
// The fix puts the customer's own pre-restore copy back automatically. These tests pin that, the
// double-failure hold, and the two absence claims that go with them.
// rollbackFixture is reconFixture with the replay failing and the rollback injectable.
func rollbackFixture(t *testing.T, rollbackErr error) (*Manager, *recordingProvider, *int) {
t.Helper()
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
return errors.New("replay blew up")
}
calls := 0
m.SetRollbackImportFn(func(_ context.Context, _ DiscoveredDB, path string) error {
calls++
// The rollback must be handed THIS RUN's undo file, not the scratch's dump.
if !strings.Contains(filepath.Base(path), preRestoreDumpPrefix) {
t.Errorf("rollback was handed %q — that is not an undo copy", filepath.Base(path))
}
return rollbackErr
})
return m, prov, &calls
}
// SCENARIO A — replay fails, rollback succeeds. The app comes back and the outcome records it.
//
// WRONG OUTCOMES PINNED: the app left down; or a message that reports the failure and omits that the
// data is back — the omission of a GAIN is as misleading as the omission of a loss, and here it is
// the difference between a customer who panics and one who does not.
func TestR379_ScenarioA_RollbackSucceeds_AppComesBack(t *testing.T) {
m, prov, calls := rollbackFixture(t, nil)
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err == nil {
t.Fatal("a failed replay must still be surfaced as a failure")
}
if *calls != 1 {
t.Fatalf("the undo copy must be re-applied exactly once, got %d", *calls)
}
if !res.RolledBack {
t.Error("the result must record that the rollback happened")
}
if !prov.fullStarted {
t.Fatal("after a successful rollback the app must be started — it is in a known good state")
}
// The customer sentence says BOTH things.
low := err.Error()
if !strings.Contains(low, "sikertelen") {
t.Errorf("the message must say the restore failed; got: %v", err)
}
if !strings.Contains(low, "visszakerültek") {
t.Errorf("the message must say the data is back as it was; got: %v", err)
}
// And no hold was written — the app is fine.
if held, _ := m.RestoreHoldFor("immich"); held {
t.Error("a recovered app must NOT be held")
}
}
// SCENARIO C — replay fails AND rollback fails. The app is held, not started.
//
// OPERATOR RULING 2026-08-22: a running app on a half-written database lets the customer type into
// it and makes the damage permanent.
//
// WRONG OUTCOMES PINNED: the app started anyway; or the app-stop marker left active, which would
// have Recover() start the broken app at the next controller boot, quietly, hours later.
func TestR379_ScenarioC_BothFail_AppIsHeldNotStarted(t *testing.T) {
m, prov, calls := rollbackFixture(t, errors.New("rollback blew up too"))
notified := 0
m.SetRestoreHoldNotify(func(string, error, error) { notified++ })
_, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err == nil {
t.Fatal("a double failure must be surfaced")
}
if *calls != 1 {
t.Fatalf("the rollback must have been attempted, got %d calls", *calls)
}
// THE OBSERVABLE THAT MATTERS.
if prov.fullStarted {
t.Fatal("the app was STARTED onto a half-written database — the exact outcome the hold exists to prevent")
}
held, why := m.RestoreHoldFor("immich")
if !held {
t.Fatal("the hold must be persisted, or nothing will refuse a restart later")
}
if why == "" || !strings.Contains(why, "kapcsolat") {
t.Errorf("the hold's reason must name a route the customer can take; got %q", why)
}
if notified != 1 {
t.Errorf("the operator must be told exactly once, got %d", notified)
}
if !strings.Contains(err.Error(), "LEÁLLÍTVA") {
t.Errorf("the customer must be told the app was deliberately stopped; got: %v", err)
}
}
// SCENARIO E — the replay SUCCEEDS. Nothing new may happen.
//
// WRONG OUTCOME PINNED: a rollback firing on a successful restore would overwrite the restored data
// with the pre-restore state — a silent, total loss of the thing the customer asked for.
func TestR379_ScenarioE_SuccessfulReplay_NoRollbackNoHold(t *testing.T) {
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
rolled := 0
m.SetRollbackImportFn(func(context.Context, DiscoveredDB, string) error { rolled++; return nil })
held := 0
m.SetRestoreHoldNotify(func(string, error, error) { held++ })
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err != nil {
t.Fatalf("a clean restore must succeed: %v", err)
}
if rolled != 0 {
t.Fatalf("a rollback fired on a SUCCESSFUL restore — the restored data would be overwritten; calls=%d", rolled)
}
if res.RolledBack {
t.Error("a successful restore must not report a rollback")
}
if held != 0 {
t.Errorf("no hold may be written on success, got %d", held)
}
if isHeld, _ := m.RestoreHoldFor("immich"); isHeld {
t.Error("a successful restore must leave no hold")
}
if !prov.fullStarted {
t.Error("a successful restore still starts the app")
}
}
// SCENARIO E, POSITIVE CONTROL. "No rollback fired" is an absence claim, so prove the counter can
// count: the same seam, driven through the failure path, must register.
func TestR379_ScenarioE_PositiveControl_TheCounterCanCount(t *testing.T) {
m, _, calls := rollbackFixture(t, nil)
if _, err := m.ReconstituteFromOffsite(context.Background(), "immich", false); err == nil {
t.Fatal("fixture: the replay must fail here")
}
if *calls == 0 {
t.Fatal("the rollback counter never increments — Scenario E's zero would have proven nothing")
}
}
// SCENARIO F — an app with no database. Untouched in every respect.
func TestR379_ScenarioF_NoDatabase_Unchanged(t *testing.T) {
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", "")
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return nil, nil }
rolled := 0
m.SetRollbackImportFn(func(context.Context, DiscoveredDB, string) error { rolled++; return nil })
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err != nil {
t.Fatalf("a no-DB app must restore: %v", err)
}
if rolled != 0 || res.RolledBack {
t.Error("a no-DB app has nothing to undo and nothing to roll back")
}
if res.SafetyDump != "" {
t.Errorf("a no-DB app takes no undo copy, got %q", res.SafetyDump)
}
if held, _ := m.RestoreHoldFor("immich"); held {
t.Error("a no-DB app must never be held")
}
if !prov.fullStarted {
t.Error("a no-DB app still starts")
}
}
// SCENARIO G — two databases, one replay fails. The WHOLE undo set is re-applied.
//
// WRONG OUTCOME PINNED: only the first. writeSafetyDump used to return one path for an app with two
// databases, so a rollback built on that value would restore one and leave the other half-written —
// this defect, one database over.
func TestR379_ScenarioG_TwoDatabases_WholeSetRolledBack(t *testing.T) {
m, _, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
two := []DiscoveredDB{
{StackName: "immich", DBType: DBTypePostgres, ContainerName: "immich-postgres", ContainerID: "a"},
{StackName: "immich", DBType: DBTypeMariaDB, ContainerName: "immich-maria", ContainerID: "b"},
}
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return two, nil }
m.safetyDumpFn = func(_ context.Context, db DiscoveredDB, finalPath string) DumpResult {
// R-361: the seam takes the FINAL PATH now.
if err := os.MkdirAll(filepath.Dir(finalPath), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(finalPath, []byte(pgDump(1)), 0o644); err != nil {
t.Fatal(err)
}
return DumpResult{DB: db, FilePath: finalPath, Size: 42}
}
m.importDBDump = func(context.Context, DiscoveredDB, string) error { return errors.New("replay failed") }
var rolled []string
m.SetRollbackImportFn(func(_ context.Context, db DiscoveredDB, path string) error {
rolled = append(rolled, db.ContainerName+":"+filepath.Base(path))
return nil
})
if _, err := m.ReconstituteFromOffsite(context.Background(), "immich", false); err == nil {
t.Fatal("the replay failure must surface")
}
if len(rolled) != 2 {
t.Fatalf("BOTH databases must be rolled back, got %d: %v", len(rolled), rolled)
}
// Each database got its OWN undo file, matched by identity rather than by prefix.
if !strings.Contains(rolled[0], "immich-postgres:") || !strings.Contains(rolled[1], "immich-maria:") {
t.Errorf("each database must get its own undo copy, got %v", rolled)
}
for _, r := range rolled {
if !strings.Contains(r, preRestoreDumpPrefix) {
t.Errorf("rollback used a non-undo file: %s", r)
}
}
}
// The undo set is matched on THIS RUN's stamp. Three undo copies from three different runs
// accumulated on docmost in one afternoon; a prefix match would replay an arbitrary older state.
func TestR379_UndoSetIsThisRunOnly(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", DBType: DBTypePostgres, ContainerName: "app-postgres", ContainerID: "c"}}, nil
}
m.safetyDumpFn = func(_ context.Context, db DiscoveredDB, finalPath string) DumpResult {
// R-361: the seam takes the FINAL PATH now.
if err := os.MkdirAll(filepath.Dir(finalPath), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(finalPath, []byte("x"), 0o644); err != nil {
t.Fatal(err)
}
return DumpResult{DB: db, FilePath: finalPath, Size: 1}
}
// An OLDER undo copy from a previous run, already on disk.
dumpDir := AppDBDumpPath(nsRoot, "app")
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
t.Fatal(err)
}
stale := filepath.Join(dumpDir, preRestoreDumpPrefix+"20200101T000000Z-app-postgres.sql")
if err := os.WriteFile(stale, []byte("STALE"), 0o644); err != nil {
t.Fatal(err)
}
set, err := m.writeSafetyDump(context.Background(), "app", nsRoot)
if err != nil {
t.Fatal(err)
}
if len(set.Files) != 1 {
t.Fatalf("one database → one undo file, got %d", len(set.Files))
}
if strings.Contains(set.Files[0].Path, "20200101") {
t.Fatal("the set picked up an OLDER run's undo copy — it must carry only this run's stamp")
}
if set.Stamp == "" || strings.Contains(set.Files[0].Path, set.Stamp) == false {
t.Errorf("the file must carry this run's stamp %q, got %q", set.Stamp, set.Files[0].Path)
}
}
// pruneUndoCopies keeps the newest N and never the oldest — ordered by the STAMP, not by mtime.
func TestR379_PruneKeepsTheNewestByStamp(t *testing.T) {
m := newSafetyTestManager()
dir := t.TempDir()
stamps := []string{"20260101T000000Z", "20260201T000000Z", "20260301T000000Z", "20260401T000000Z", "20260501T000000Z"}
for _, st := range stamps {
if err := os.WriteFile(filepath.Join(dir, preRestoreDumpPrefix+st+"-app-postgres.sql"), []byte("x"), 0o644); err != nil {
t.Fatal(err)
}
}
// The app's OWN dump must never be a candidate.
own := filepath.Join(dir, "app-postgres.sql")
if err := os.WriteFile(own, []byte("own"), 0o644); err != nil {
t.Fatal(err)
}
m.pruneUndoCopies(dir, "app")
if _, err := os.Stat(own); err != nil {
t.Fatal("the app's own dump was pruned — only undo copies are candidates")
}
for _, st := range stamps[len(stamps)-maxUndoCopiesPerApp:] {
if _, err := os.Stat(filepath.Join(dir, preRestoreDumpPrefix+st+"-app-postgres.sql")); err != nil {
t.Errorf("the newest %d must survive; %s is gone", maxUndoCopiesPerApp, st)
}
}
for _, st := range stamps[:len(stamps)-maxUndoCopiesPerApp] {
if _, err := os.Stat(filepath.Join(dir, preRestoreDumpPrefix+st+"-app-postgres.sql")); err == nil {
t.Errorf("the oldest must be pruned; %s survived", st)
}
}
}
// R-381 — the customer sentence must not carry the engine's output.
//
// MEASURED 2026-08-22: 407 bytes (Postgres, with a caret diagram and `exit status 3`) and 615 bytes
// (MariaDB, whose middle was an `INSERT INTO migrations VALUES (...)` listing — ROWS OUT OF THE
// CUSTOMER'S OWN DATABASE, HTML-escaped, on their dashboard).
func TestR381_CustomerMessageCarriesNoEngineOutput(t *testing.T) {
engineNoise := "ERROR: syntax error at end of input\nLINE 1: COPY public.felhom_r356b_discriminator \n ^ — exit status 3"
m, _, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
return errors.New(engineNoise)
}
m.SetRollbackImportFn(func(context.Context, DiscoveredDB, string) error { return nil })
_, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err == nil {
t.Fatal("the failure must surface")
}
msg := err.Error()
for _, leak := range []string{"ERROR: syntax", "LINE 1:", "COPY public.", "exit status"} {
if strings.Contains(msg, leak) {
t.Errorf("the customer message leaks engine output %q; got: %s", leak, msg)
}
}
if len(msg) > 320 {
t.Errorf("the customer message is %d bytes — it is a sentence, not a transcript: %s", len(msg), msg)
}
// POSITIVE CONTROL: the noise really was in the error the code received, so the absence above is
// the message being clean rather than the noise never existing.
if !strings.Contains(engineNoise, "exit status") {
t.Fatal("fixture: the planted noise does not contain the marker this test greps for")
}
}
// THE DEFECT THE LIVE WALK FOUND, and the unit tests could not.
//
// The undo FILE is stable; the container it must be poured into is NOT. writeSafetyDump captures its
// DiscoveredDB BEFORE the stop, and by rollback time the stack has been stopped and the DB service
// re-created with a new container id. Measured on demo-hp 2026-08-22: captured `9adbc14f9af6` at
// 16:05:44, re-created as `309795897b82` at 16:05:47, and the rollback's docker exec against the dead
// id sat in waitDBReady until it timed out 30 s later — so the app was HELD for an infrastructure
// reason while its data was perfectly recoverable.
//
// Every other test here injects the import seam and never looks at container identity, which is
// exactly why none of them saw it. This one asserts the identity handed to the import.
func TestR379_RollbackUsesTheLiveContainerNotTheCapturedOne(t *testing.T) {
m, _, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
const capturedID, liveID = "9adbc14f9af6", "309795897b82"
calls := 0
// Discovery returns the CAPTURED id first (safety-dump time), then the LIVE one — the real
// sequence, because the DB service is re-created in between.
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
calls++
id := liveID
if calls == 1 {
id = capturedID
}
return []DiscoveredDB{{StackName: "immich", DBType: DBTypePostgres, ContainerName: "immich-postgres", ContainerID: id}}, nil
}
m.importDBDump = func(context.Context, DiscoveredDB, string) error { return errors.New("replay failed") }
var rolledInto []string
m.SetRollbackImportFn(func(_ context.Context, db DiscoveredDB, _ string) error {
rolledInto = append(rolledInto, db.ContainerID)
return nil
})
if _, err := m.ReconstituteFromOffsite(context.Background(), "immich", false); err == nil {
t.Fatal("the replay failure must surface")
}
if len(rolledInto) != 1 {
t.Fatalf("the rollback must have run once, got %v", rolledInto)
}
if rolledInto[0] == capturedID {
t.Fatalf("the rollback used the CAPTURED container id %s — that container no longer exists by rollback time; "+
"it must re-discover and use the live one", capturedID)
}
if rolledInto[0] != liveID {
t.Fatalf("the rollback must target the LIVE container %s, got %s", liveID, rolledInto[0])
}
}
// Fail closed: if the database's container cannot be found at rollback time, say so rather than pour
// the undo into something unidentified.
func TestR379_RollbackFailsClosedWhenTheContainerIsGone(t *testing.T) {
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", pgDump(1))
// Three discovery calls happen, in this order: writeSafetyDump, reimportDBDumpsFrom (the
// replay), then rollbackSafetyDump. The container vanishes only for the THIRD.
calls := 0
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
calls++
if calls >= 3 {
return nil, nil // vanished by rollback time
}
return []DiscoveredDB{{StackName: "immich", DBType: DBTypePostgres, ContainerName: "immich-postgres", ContainerID: "a"}}, nil
}
m.importDBDump = func(context.Context, DiscoveredDB, string) error { return errors.New("replay failed") }
rolled := 0
m.SetRollbackImportFn(func(context.Context, DiscoveredDB, string) error { rolled++; return nil })
_, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err == nil {
t.Fatal("a rollback with no container must surface")
}
if rolled != 0 {
t.Errorf("nothing may be imported when the target cannot be identified, got %d", rolled)
}
if prov.fullStarted {
t.Error("a failed rollback must leave the app held, not started")
}
if held, _ := m.RestoreHoldFor("immich"); !held {
t.Error("the app must be held when the rollback could not run")
}
}