package backup import ( "context" "io" "log" "os" "path/filepath" "strings" "testing" ) // R-102 Group A — the recovery-unit restore can be pointed at a unit ANYWHERE. // // The defect: every reader of a recovery unit could only name a path under `backups/primary/` // (appbackup/paths.go joined it literally), while Tier-2 mirrored each app's whole unit to // `/backups/secondary//recovery-unit/` on every run. So the mirror was written nightly for // months and read by nothing — and it was unreadable in exactly the failure Tier-2 exists for, where // the primary drive and its unit are gone (07-backup-architecture §6.3, §7.2). // // Every test below asserts WHICH DIRECTORY was read, not merely that a restore succeeded. A test that // only checked `err == nil` would have passed against the pre-fix code, because the pre-fix code read // a real unit — just never the one that survives. // r102Unit captures a REAL recovery unit for "app" onto its own fresh drive, with an identifying // SUBDOMAIN, a portable DB password and a portable data key, plus optional volume tars and a DB dump. // // The unit is produced by the production CaptureRecoveryUnit, not hand-written: the capture side and // the restore side must meet at real bytes, so a change to the on-disk shape cannot pass by having a // test agree with itself (the reason captureFixtureUnit is built the same way). func r102Unit(t *testing.T, marker string, volTars []string, dbDump string) (drive, unitDir string) { t.Helper() tmp := t.TempDir() drive = filepath.Join(tmp, "drive") return drive, r102UnitOnDrive(t, drive, marker, volTars, dbDump) } // r102UnitOnDrive is r102Unit against a drive the caller already owns — the Tier-2 fixture needs the // unit to sit on the SAME drive the Tier-2 run will mirror FROM. func r102UnitOnDrive(t *testing.T, drive, marker string, volTars []string, dbDump string) (unitDir string) { t.Helper() tmp := t.TempDir() stackDir := filepath.Join(tmp, "stack") if err := os.MkdirAll(stackDir, 0o755); err != nil { t.Fatal(err) } mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"), "services:\n app:\n image: example/app:1\n db:\n image: postgres:16\n") mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: App\n") mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: "+marker+"\n") info := RecoveryInfo{ StackDir: stackDir, DisplayName: "App", ImagePins: []string{"example/app:1"}, NonSecretEnv: map[string]string{"SUBDOMAIN": marker}, SecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"}, DataKeyEnvVars: []string{"SECRET_KEY"}, PortableSecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"}, PortableSecrets: map[string]string{"DB_PASSWORD": "pw-" + marker, "SECRET_KEY": "key-" + marker}, } unitDir = RecoveryUnitPath(drive, "app") // The dumps are written BEFORE the capture so the manifest enumerates them — a manifest that // lists nothing is the R-353 "the backup held only settings" shape, which is a different case. for _, v := range volTars { mustWrite(t, filepath.Join(UnitVolumeDumpDir(unitDir), v), "tar:"+v+":"+marker) } if dbDump != "" { mustWrite(t, filepath.Join(UnitDBDumpDir(unitDir), "app-postgres.sql"), dbDump) } m := &Manager{ logger: log.New(io.Discard, "", 0), systemDataPath: filepath.Join(tmp, "system"), stackProvider: &fakeRecoveryProvider{info: info, hdd: drive}, version: "vtest", } if err := m.CaptureRecoveryUnit("app"); err != nil { t.Fatalf("capture fixture %q: %v", marker, err) } return unitDir } // r102Manager builds a restore-side Manager whose LIVE drive is liveDrive, with the Docker and DB // seams injected so no daemon is touched. The returned slices record the directories each leg read. func r102Manager(t *testing.T, liveDrive string) (m *Manager, fake *fakeRecoveryProvider, volDirs, dbPaths *[]string) { t.Helper() fake = &fakeRecoveryProvider{hdd: liveDrive, running: true} m = &Manager{ logger: log.New(io.Discard, "", 0), systemDataPath: filepath.Join(liveDrive, "..", "sys"), stackProvider: fake, } var vd, dp []string m.volumeReplayFrom = func(_, dumpDir string) (int, error) { vd = append(vd, dumpDir) entries, err := os.ReadDir(dumpDir) if err != nil { return 0, nil } n := 0 for _, e := range entries { if strings.HasSuffix(e.Name(), ".tar") { n++ } } return n, nil } m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil } m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error { dp = append(dp, p) return nil } return m, fake, &vd, &dp } // A2 — TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir. // // Two complete units exist on disk with DIFFERENT contents. The restore is pointed at the second one // and must read every leg — env, secrets, volume tars, DB dump — out of THAT directory. The first // unit is the app's own primary unit and is present and perfectly readable, which is what makes the // assertion mean something: a restore that ignored unitDir would still succeed, and would silently // return the wrong copy's data. func TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir(t *testing.T) { primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1)) _, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar", "vol_b.tar"}, pgDump(2)) m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive) res, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit) if err != nil { t.Fatalf("restore from the named unit: %v", err) } if fake.gotEnv == nil { t.Fatal("recreate was never called — the restore did not reach the redeploy") } if got := fake.gotEnv["SUBDOMAIN"]; got != "mirror" { t.Errorf("config came from the WRONG unit: SUBDOMAIN=%q, want %q", got, "mirror") } if got := fake.gotEnv["SECRET_KEY"]; got != "key-mirror" { t.Errorf("data key came from the WRONG unit: %q", got) } if got := fake.gotEnv["DB_PASSWORD"]; got != "pw-mirror" { t.Errorf("DB password came from the WRONG unit: %q", got) } if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(mirrorUnit) { t.Errorf("volume tars read from %v, want %q", *volDirs, UnitVolumeDumpDir(mirrorUnit)) } if res.VolumesReplayed != 2 { t.Errorf("VolumesReplayed=%d, want 2 (the mirror holds two tars; the primary holds one)", res.VolumesReplayed) } if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(mirrorUnit) { t.Errorf("DB dump read from %v, want a file under %q", *dbPaths, UnitDBDumpDir(mirrorUnit)) } // The negative control: nothing was read out of the primary unit. for _, d := range *volDirs { if strings.HasPrefix(d, primaryUnit) { t.Errorf("a leg was read from the PRIMARY unit %q despite a mirror being named", d) } } } // A3 — TestR102_PrimaryPathIsUnchanged. The one-argument wrapper must still resolve to the app's own // drive, `backups/primary/`. Asserted on the DIRECTORIES the legs actually read, not on a // path expression re-derived in the test. func TestR102_PrimaryPathIsUnchanged(t *testing.T) { primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1)) m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive) if _, err := m.RestoreFromRecoveryUnit("app"); err != nil { t.Fatalf("primary restore: %v", err) } if got := fake.gotEnv["SUBDOMAIN"]; got != "primary" { t.Errorf("SUBDOMAIN=%q, want the primary unit's %q", got, "primary") } if !strings.Contains(primaryUnit, filepath.Join("backups", "primary", "app")) { t.Fatalf("fixture is not where the primary unit belongs: %q", primaryUnit) } if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(primaryUnit) { t.Errorf("volume dir = %v, want %q", *volDirs, UnitVolumeDumpDir(primaryUnit)) } if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(primaryUnit) { t.Errorf("db dump dir = %v, want %q", *dbPaths, UnitDBDumpDir(primaryUnit)) } } // A4 — TestR102_LiveDestinationIsUnchanged. THE SOURCE MOVES; THE DESTINATION DOES NOT. // // The restore reads a mirror that lives on a different drive entirely, and must still write the app // back to its own live namespace: the definition through RecreateStackDefinitionFromUnit (whose // compose dir must be the MIRROR's, since that is the definition being restored), and the data into // the LIVE Docker volumes and the LIVE database container. A restore that also relocated the app's // data would be a migration, not a restore. // // The observable for "the destination did not move" is that nothing under the mirror's own drive was // written to: the mirror tree is byte-identical before and after. func TestR102_LiveDestinationIsUnchanged(t *testing.T) { primaryDrive, _ := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1)) mirrorDrive, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2)) before := fingerprintTree(t, mirrorDrive) m, fake, _, _ := r102Manager(t, primaryDrive) var recreateComposeDir string m.stackProvider = &r102RecordingProvider{fakeRecoveryProvider: fake, composeDirOut: &recreateComposeDir} if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil { t.Fatalf("restore: %v", err) } if recreateComposeDir != UnitComposeDir(mirrorUnit) { t.Errorf("recreate read compose from %q, want the mirror's %q", recreateComposeDir, UnitComposeDir(mirrorUnit)) } if after := fingerprintTree(t, mirrorDrive); after != before { t.Error("the SOURCE mirror was written to — a restore reads its source and never writes it") } // The live drive is where the app lives, and the restore must not have relocated it: the app's // own drive path is still the one the manager resolves for it. if got := m.GetAppDrivePath("app"); got != primaryDrive { t.Errorf("the app's live drive moved to %q, want %q", got, primaryDrive) } } // r102RecordingProvider captures the compose directory RecreateStackDefinitionFromUnit is handed — // the one place the restore names a source directory to the guest side. type r102RecordingProvider struct { *fakeRecoveryProvider composeDirOut *string } func (f *r102RecordingProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error { *f.composeDirOut = composeDir return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv) } // A5 — TestR102_MutationOrderIsPreserved. R-47's ordering is pinned and R-102 must not have moved it: // stop → volumes → recreate → DB-only start → replay → full start. The DB-only phase exists because // starting the whole stack first let the application rebuild schema underneath the replay (H4, // DIAG-immich-restore-round2-2026-07-19). // // The state AT REPLAY TIME is asserted, not just the call sequence — H4 looked correctly ordered and // the app was up. // // Red-proof (recorded in REPORT.md): swap the volume replay and the recreate step in // RestoreFromRecoveryUnitAt → this test fails on the sequence assertion. func TestR102_MutationOrderIsPreserved(t *testing.T) { primaryDrive, _ := r102Unit(t, "primary", nil, pgDump(1)) _, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2)) m, fake, _, _ := r102Manager(t, primaryDrive) var volumesDoneAtRecreate, fullUpAtReplay bool var volumesReplayed bool m.volumeReplayFrom = func(_, _ string) (int, error) { volumesReplayed = true if fake.gotEnv != nil { t.Error("volumes were replayed AFTER the definition was recreated") } return 1, nil } prov := &r102OrderProvider{fakeRecoveryProvider: fake, volumesReplayed: &volumesReplayed, seen: &volumesDoneAtRecreate} m.stackProvider = prov m.importDBDump = func(context.Context, DiscoveredDB, string) error { fullUpAtReplay = fake.fullStarted return nil } if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil { t.Fatalf("restore: %v", err) } if got := strings.Join(fake.calls, ","); got != "stop,recreate,startsvc:db,start" { t.Fatalf("sequence = %q, want stop → recreate → db-only start → (replay) → full start", got) } if !volumesDoneAtRecreate { t.Error("the volume replay had not run when the definition was recreated") } if fullUpAtReplay { t.Error("the FULL stack was already up when the DB replay fired — this is H4 exactly (R-47)") } } // r102OrderProvider records whether the volume replay had already happened by the time the definition // was recreated — the ordering fact the call log alone cannot carry. type r102OrderProvider struct { *fakeRecoveryProvider volumesReplayed *bool seen *bool } func (f *r102OrderProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error { *f.seen = *f.volumesReplayed return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv) } // A6 — TestR102_MissingManifestInMirrorFailsClosed. A DIRECTORY IS NOT A PACKAGE. // // `/backups/secondary//recovery-unit/` can exist and be useless: a copy interrupted // mid-run, or a tree whose manifest.json never landed. The Tier-2 unit restore must refuse it and // must leave the app running — a restore armed over an unopenable unit would stop the app, replay // nothing, and rewrite its definition from an empty capture. Same lesson as R-358 one tier over. // // Placed with Group A because it is the fail-closed half of the parameterised unit; the route it // exercises is Part 1.3's. func TestR102_MissingManifestInMirrorFailsClosed(t *testing.T) { for _, tc := range []struct { name string setup func(t *testing.T, unitDir string) }{ {"manifest absent", func(t *testing.T, unitDir string) { if err := os.Remove(UnitManifestFile(unitDir)); err != nil { t.Fatal(err) } }}, {"manifest unparseable", func(t *testing.T, unitDir string) { mustWrite(t, UnitManifestFile(unitDir), "{ this is not json") }}, {"recovery-unit absent entirely", func(t *testing.T, unitDir string) { if err := os.RemoveAll(unitDir); err != nil { t.Fatal(err) } }}, } { t.Run(tc.name, func(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) tc.setup(t, tier2UnitDir(f.destBase)) cov, covErr := f.m.Tier2RestoreCoverage("app") if covErr != nil { t.Fatalf("coverage: %v", covErr) } if cov.CanRestoreUnit() { t.Error("CanRestoreUnit() said yes over an unopenable mirror — the surface would offer the action") } _, err := f.m.RestoreTier2Unit("app") if err == nil { t.Fatal("the restore ran over an unopenable mirror") } if !strings.Contains(err.Error(), ErrTier2NoUnitInCopy.Error()) { t.Errorf("refusal = %v, want ErrTier2NoUnitInCopy", err) } if f.fake.stopped { t.Error("the app was STOPPED despite the refusal — the outage this gate exists to avoid") } if f.fake.gotEnv != nil { t.Error("the app's definition was rewritten despite the refusal") } }) } } // TestR102_UnopenableMirrorIsStillDisclosedAsUnread is the other side of A6, and the reason // UnitRestorable is a second field rather than a widening of HasUnit: a half-copied mirror cannot be // restored, but it IS captured data the file restore is not looking at, so the disclosure must stand. func TestR102_UnopenableMirrorIsStillDisclosedAsUnread(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) mustWrite(t, UnitManifestFile(tier2UnitDir(f.destBase)), "{ not json") cov, err := f.m.Tier2RestoreCoverage("app") if err != nil { t.Fatalf("coverage: %v", err) } if !cov.HasUnit { t.Error("HasUnit went false for a mirror that exists — the file restore would stop disclosing unread data") } if cov.UnitRestorable { t.Error("UnitRestorable stayed true over an unparseable manifest") } }