package backup import ( "context" "errors" "io" "log" "path/filepath" "strings" "testing" ) // R-638 option A (09-update-architecture §3 decision 154) — a replay must meet the COPY's own schema. // // The loader overlays a copy on the live database: it only removes what the copy knows about. So a // replay is only safe when the database it lands in holds the copy's own files, not a newer app's // migrated schema. Measured 2026-10-06 on scratch 9202: the unit restore after a docmost 0.95 → 0.96 // migration put back exactly the copy's 42 tables — the main path is safe because it restores the // volume FIRST and replays with ONLY the database up. These tests pin the two side paths that did not: // // - Slice 1, the no-manifest fallback RestoreApp: it started the WHOLE stack at the CURRENT // definition before the replay, so a newer app could migrate the old data underneath it. // - Slice 2, the unit restore whose volume leg failed: it replayed anyway, over a database volume // that is not the copy's own. // // All Docker-reaching seams are injected (volumeReplayFrom, discoverDBs, importDBDump); no test here // touches a daemon. // r638FallbackFixture builds a Manager whose app "app" has NO recovery unit (so RestoreFromRecoveryUnit // takes the RestoreApp fallback), a LIVE compose with the given body, and — optionally — a replayable // dump at the app's on-drive dump path, which is where RestoreApp replays from. func r638FallbackFixture(t *testing.T, compose string, withDump bool) (*Manager, *fakeRecoveryProvider, *[]string) { t.Helper() tmp := t.TempDir() drive := filepath.Join(tmp, "drive") stackDir := filepath.Join(tmp, "stack") mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"), compose) if withDump { mustWrite(t, filepath.Join(AppDBDumpPath(drive, "app"), "app-postgres.sql"), pgDump(1)) } prov := &fakeRecoveryProvider{hdd: drive, running: true, info: RecoveryInfo{StackDir: stackDir}} m := &Manager{ logger: log.New(io.Discard, "", 0), systemDataPath: filepath.Join(tmp, "sys"), // != drive ⇒ nsRoot = drive stackProvider: prov, } // The volume leg is a seam here so the test can never reach Docker, even if a tar appears. m.volumeReplayFrom = func(string, string) (int, error) { return 0, nil } db := DiscoveredDB{StackName: "app", ContainerName: "immich-postgres", DBType: DBTypePostgres} m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{db}, nil } var imported []string m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error { imported = append(imported, p) return nil } return m, prov, &imported } // --- Slice 1: the no-manifest fallback ------------------------------------------------------------ // TestR638_FallbackReplaysBeforeTheAppStarts is the slice-1 assertion. It captures the provider's // state AT THE MOMENT the importer fires: the database service must be up and the full stack must // NOT be — a full start at the current definition is exactly the window in which a newer app // migrates the restored data before the old copy is poured over it. // // COMPANION RED-PROOF: on the pre-fix RestoreApp (StartStack before reimportDBDumpsCtx) this fails on // `the FULL stack was already up when the replay fired` — saved in // audits/design-build-2026-10-06/B/red-slice1-fallback-order.txt. func TestR638_FallbackReplaysBeforeTheAppStarts(t *testing.T) { m, prov, imported := r638FallbackFixture(t, immichLikeCompose, true) var dbUpAtReplay, fullUpAtReplay bool var callsAtReplay string m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error { dbUpAtReplay = len(prov.gotServices) > 0 fullUpAtReplay = prov.fullStarted callsAtReplay = strings.Join(prov.calls, ",") *imported = append(*imported, p) return nil } if err := m.RestoreApp("app", ""); err != nil { t.Fatalf("fallback restore: %v", err) } if len(*imported) != 1 { t.Fatalf("expected exactly one replay, got %v", *imported) } if fullUpAtReplay { t.Fatalf("the FULL stack was already up when the replay fired (calls before the replay: %q) — a newer app can migrate the restored data before the copy is loaded (R-638)", callsAtReplay) } if !dbUpAtReplay { t.Fatal("the database service was NOT started before the replay — the importer has no container to talk to") } if got := strings.Join(prov.gotServices, ","); got != "immich-postgres" { t.Fatalf("DB-only phase started %q, want only the database service immich-postgres", got) } if got := strings.Join(prov.calls, ","); got != "stop,startsvc:immich-postgres,start" { t.Fatalf("sequence = %q, want stop → volumes → db-only start → replay → full start", got) } } // TestR638_FallbackThroughUnitRestoreKeepsTheOrder reaches the fallback the way production does: a // unit restore that finds no manifest. Same ordering, and the counts stay UNKNOWN (not zero). func TestR638_FallbackThroughUnitRestoreKeepsTheOrder(t *testing.T) { m, prov, imported := r638FallbackFixture(t, immichLikeCompose, true) res, err := m.RestoreFromRecoveryUnit("app") if err != nil { t.Fatalf("restore: %v", err) } if !res.CountsUnknown { t.Error("the fallback's counts must stay UNKNOWN, not zero") } if len(*imported) != 1 { t.Fatalf("expected exactly one replay, got %v", *imported) } if got := strings.Join(prov.calls, ","); got != "stop,startsvc:immich-postgres,start" { t.Fatalf("sequence = %q, want stop → db-only start → replay → full start", got) } } // TestR638_FallbackNoDumpTakesOneFullStart is the negative: with nothing to replay there is no // DB-only window, and the flow is the old stop → volumes → full start. func TestR638_FallbackNoDumpTakesOneFullStart(t *testing.T) { m, prov, imported := r638FallbackFixture(t, immichLikeCompose, false) if err := m.RestoreApp("app", ""); err != nil { t.Fatalf("fallback restore: %v", err) } if len(prov.gotServices) != 0 { t.Fatalf("the DB-only phase ran with nothing to replay: %v", prov.gotServices) } if got := strings.Join(prov.calls, ","); got != "stop,start" { t.Fatalf("sequence = %q, want the unchanged stop → full start", got) } if len(*imported) != 0 { t.Fatalf("nothing should have been replayed, got %v", *imported) } } // TestR638_FallbackRefusesWhenNoDBServiceIdentifiable: a dump exists but the live compose names no // database service that could be started alone. The only alternative is the old full start + replay // — the R-638 shape — so it refuses, with ZERO mutations, exactly as the unit and off-site paths do. func TestR638_FallbackRefusesWhenNoDBServiceIdentifiable(t *testing.T) { m, prov, imported := r638FallbackFixture(t, noDBCompose, true) volTouched := false m.volumeReplayFrom = func(string, string) (int, error) { volTouched = true; return 0, nil } err := m.RestoreApp("app", "") if err == nil { t.Fatal("expected a refusal: a dump exists but no database service can be started for it") } if !strings.Contains(err.Error(), "nem azonosítható") { t.Fatalf("refusal must say the database service could not be identified, got: %v", err) } if len(prov.calls) != 0 { t.Fatalf("ZERO mutations required, but the provider was called: %v", prov.calls) } if volTouched { t.Fatal("volumes were replaced despite the refusal") } if len(*imported) != 0 { t.Fatalf("a replay happened despite the refusal: %v", *imported) } } // TestR638_FallbackVolumeFailureSkipsTheReplay is slice 2's rule on the fallback path: when the // volume leg failed, the database volume is not known to hold the copy's own files, so the copy is // NOT poured over it. The failure is still returned and the app is still brought back up. func TestR638_FallbackVolumeFailureSkipsTheReplay(t *testing.T) { m, prov, imported := r638FallbackFixture(t, immichLikeCompose, true) m.volumeReplayFrom = func(string, string) (int, error) { return 0, errors.New("failed to restore 1 volume(s): [app_db]") } err := m.RestoreApp("app", "") if err == nil || !strings.Contains(err.Error(), "completed with data errors") || !strings.Contains(err.Error(), "app_db") { t.Fatalf("the volume failure must surface as the data-error outcome naming the volume, got: %v", err) } if len(*imported) != 0 { t.Fatalf("the importer ran after a failed volume leg: %v", *imported) } if !prov.fullStarted { t.Fatal("the app was left down after the failure — today's failure path brings it back up") } if got := strings.Join(prov.calls, ","); got != "stop,start" { t.Fatalf("sequence = %q, want stop → full start (no DB-only window, no replay)", got) } } // --- Slice 2: unit restore with a failed volume leg ----------------------------------------------- // TestR638_UnitVolumeFailureNeverCallsTheImporter is the slice-2 assertion. A volume leg that errors // means the database's volume may still be the LIVE (possibly newer, migrated) one, or a half-filled // one — the copy overlays it and leaves whatever it does not know about. So the importer must not run. // // What must NOT change from today's failure path: the error reaches the caller as "completed with data // errors" (no silent success), the definition is the unit's, and the app is brought back up. // // COMPANION RED-PROOF: on the pre-fix restore_unit.go (dataErr set, then fall through to the replay) // this fails on `the importer ran after a failed volume leg` — saved in // audits/design-build-2026-10-06/B/red-slice2-no-replay-after-volume-failure.txt. func TestR638_UnitVolumeFailureNeverCallsTheImporter(t *testing.T) { m, prov, imported := r47UnitFixture(t, immichLikeCompose, true) m.volumeReplayFrom = func(string, string) (int, error) { return 1, errors.New("failed to restore 1 volume(s): [app_immich_postgres_data]") } res, err := m.RestoreFromRecoveryUnit("app") if len(*imported) != 0 { t.Fatalf("the importer ran after a failed volume leg: %v", *imported) } if err == nil { t.Fatal("a failed volume leg must be surfaced, not reported as success") } if !strings.Contains(err.Error(), "completed with data errors") || !strings.Contains(err.Error(), "app_immich_postgres_data") { t.Fatalf("the pre-existing outcome must be preserved and name the failed volume, got: %v", err) } if res.VolumesReplayed != 1 { t.Errorf("the partial volume count must still be reported, got %d", res.VolumesReplayed) } if res.DBsReplayed != 0 { t.Errorf("no database was replayed, but the result says %d", res.DBsReplayed) } if prov.gotEnv == nil { t.Error("the unit's definition must still be written, as on today's failure path") } if !prov.fullStarted { t.Fatal("the app was left down — today's failure path brings it back up") } if got := strings.Join(prov.calls, ","); got != "stop,recreate,start" { t.Fatalf("sequence = %q, want stop → recreate → full start (no DB-only window, no replay)", got) } } // TestR638_UnitVolumeSuccessStillReplays is the positive control for slice 2: the skip is keyed on the // volume FAILURE, not on the presence of volumes — a clean volume leg still replays. func TestR638_UnitVolumeSuccessStillReplays(t *testing.T) { m, prov, imported := r47UnitFixture(t, immichLikeCompose, true) m.volumeReplayFrom = func(string, string) (int, error) { return 1, nil } res, err := m.RestoreFromRecoveryUnit("app") if err != nil { t.Fatalf("restore: %v", err) } if len(*imported) != 1 || res.DBsReplayed != 1 { t.Fatalf("a clean volume leg must still replay: imported=%v replayed=%d", *imported, res.DBsReplayed) } if got := strings.Join(prov.calls, ","); got != "stop,recreate,startsvc:immich-postgres,start" { t.Fatalf("sequence = %q", got) } }