From 0f9b796615d3e662fc010b747fb5048f36faffb7 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Mon, 31 Aug 2026 11:30:34 +0200 Subject: [PATCH] R-102: the recovery unit on the second drive becomes a way back Tier-2 mirrors each app's whole recovery unit to /backups/secondary//recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable). --- .../internal/backup/r102_tier2_unit_test.go | 298 ++++++++++++++ .../internal/backup/r102_unit_at_test.go | 369 ++++++++++++++++++ controller/internal/backup/restore_db.go | 17 +- controller/internal/backup/restore_unit.go | 64 ++- controller/internal/backup/tier2_restore.go | 124 +++++- 5 files changed, 856 insertions(+), 16 deletions(-) create mode 100644 controller/internal/backup/r102_tier2_unit_test.go create mode 100644 controller/internal/backup/r102_unit_at_test.go diff --git a/controller/internal/backup/r102_tier2_unit_test.go b/controller/internal/backup/r102_tier2_unit_test.go new file mode 100644 index 0000000..f720625 --- /dev/null +++ b/controller/internal/backup/r102_tier2_unit_test.go @@ -0,0 +1,298 @@ +package backup + +import ( + "context" + "io" + "log" + "os" + "path/filepath" + "strings" + "testing" + + "gitea.dooplex.hu/admin/felhom-controller/internal/config" + "gitea.dooplex.hu/admin/felhom-controller/internal/settings" +) + +// R-102 Group B — the Tier-2 route: the mirror on the SECOND DRIVE becomes a way back. +// +// The mirror in every test here is written by the PRODUCTION Tier-2 run (RunTier2 with copyTree +// standing in for rsync, the seam tier2_v2_test.go already uses), not by hand. That matters: the +// claim is "the copy Tier-2 actually writes is the copy this restore reads", and a hand-built +// fixture could only prove that the restore agrees with the test's idea of the layout. + +// r102T2 is a fully-wired Tier-2 fixture: a live drive carrying the app's PRIMARY recovery unit, a +// second drive carrying the mirror Tier-2 just wrote, and a restore-side Manager with the Docker and +// DB seams injected so no daemon is touched. +type r102T2 struct { + m *Manager + fake *fakeRecoveryProvider + liveDrive string + destDrive string + destBase string + primaryUnit string + volDirs *[]string + dbPaths *[]string +} + +// r102Tier2Fixture captures a unit on the live drive, runs the REAL RunTier2 to mirror it onto the +// second drive, then returns a Manager ready to restore. The mirror's contents differ from the +// primary's afterwards when the caller mutates one of them — B1 and B2 both depend on being able to +// tell the two copies apart. +func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 { + t.Helper() + tmp := t.TempDir() + live := filepath.Join(tmp, "usb") + dest := filepath.Join(tmp, "flash") + sys := filepath.Join(tmp, "sys") + + primaryUnit := r102UnitOnDrive(t, live, "primary", volTars, dbDump) + + sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0)) + if err != nil { + t.Fatal(err) + } + if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil { + t.Fatal(err) + } + fake := &fakeRecoveryProvider{hdd: live, running: true} + cfg := &config.Config{} + cfg.Paths.SystemDataPath = sys + cfg.Paths.DataDir = filepath.Join(tmp, "data") + m := NewManager(cfg, sett, log.New(io.Discard, "", 0)) + m.stackProvider = fake + m.systemDataPath = sys + m.tier2Mirror = copyTree + m.tier2SSDFits = func(string, int64) bool { return true } + m.samePhysicalDevice = oneDrivePerSubtree + + if err := m.RunTier2("app"); err != nil { + t.Fatalf("RunTier2 (the code that writes the mirror this task reads): %v", err) + } + destBase := filepath.Join(dest, "backups", "secondary", "app") + if _, sErr := os.Stat(UnitManifestFile(tier2UnitDir(destBase))); sErr != nil { + t.Fatalf("Tier-2 did not mirror the unit — the fixture proves nothing: %v", sErr) + } + + var vd, dp []string + m.volumeReplayFrom = func(_, dumpDir string) (int, error) { + vd = append(vd, dumpDir) + entries, rErr := os.ReadDir(dumpDir) + if rErr != nil { + return 0, nil + } + n := 0 + for _, e := range entries { + if strings.HasSuffix(e.Name(), ".tar") { + n++ + } + } + return n, nil + } + m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { + return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil + } + m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error { + dp = append(dp, p) + return nil + } + return &r102T2{m: m, fake: fake, liveDrive: live, destDrive: dest, destBase: destBase, + primaryUnit: primaryUnit, volDirs: &vd, dbPaths: &dp} +} + +// markMirror rewrites the MIRROR's app.yaml so the two copies are distinguishable. It edits the copy, +// never the primary, so a restore that read the primary would return the pre-edit value. +func (f *r102T2) markMirror(t *testing.T, marker string) { + t.Helper() + p := filepath.Join(UnitComposeDir(tier2UnitDir(f.destBase)), "app.yaml") + b, err := os.ReadFile(p) + if err != nil { + t.Fatal(err) + } + out := strings.Replace(string(b), "SUBDOMAIN: primary", "SUBDOMAIN: "+marker, 1) + if out == string(b) { + t.Fatalf("mirror app.yaml did not carry the expected marker; contents:\n%s", b) + } + mustWrite(t, p, out) +} + +// B1 — TestR102_Tier2UnitRestoreReadsTheSecondaryMirror. +func TestR102_Tier2UnitRestoreReadsTheSecondaryMirror(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) + f.markMirror(t, "from-the-mirror") + + res, err := f.m.RestoreTier2Unit("app") + if err != nil { + t.Fatalf("Tier-2 unit restore: %v", err) + } + if got := f.fake.gotEnv["SUBDOMAIN"]; got != "from-the-mirror" { + t.Errorf("config came from the PRIMARY unit: SUBDOMAIN=%q", got) + } + wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase)) + if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol { + t.Errorf("volume tars read from %v, want the mirror's %q", *f.volDirs, wantVol) + } + wantDB := UnitDBDumpDir(tier2UnitDir(f.destBase)) + if len(*f.dbPaths) != 1 || filepath.Dir((*f.dbPaths)[0]) != wantDB { + t.Errorf("DB dump read from %v, want a file under the mirror's %q", *f.dbPaths, wantDB) + } + if res.VolumesReplayed != 1 || res.DBsReplayed != 1 { + t.Errorf("nothing came back: volumes=%d dbs=%d", res.VolumesReplayed, res.DBsReplayed) + } + if res.ManifestVolumes != 1 || res.ManifestDBs != 1 { + t.Errorf("the mirror's manifest did not enumerate its dumps: %+v", res) + } +} + +// B2 — TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT. +// +// THE ACCEPTANCE TEST. Tier-2 exists for the loss of the primary drive, and in that failure the +// primary recovery unit is GONE. A test that leaves the primary in place proves the code compiles, +// not that the mirror is a route (07-backup-architecture §7.2). +// +// The primary unit is not merely emptied — the whole `backups/` tree on the live drive is removed and +// then made unreadable, so any code that tried to fall back to it would error rather than silently +// find nothing. +// +// Red-proof (recorded in REPORT.md): point RestoreTier2Unit back at the primary unit path → this test +// fails with "no readable recovery unit ... falling back to volume-only restore" behaviour and the +// mirror's marker never reaching the redeploy. +func TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1)) + f.markMirror(t, "only-the-mirror-survives") + + // The primary drive's whole backup tree is destroyed, then the parent is made unreadable so a + // fallback read cannot quietly return "nothing here". + primaryBackups := filepath.Join(f.liveDrive, "backups") + if err := os.RemoveAll(primaryBackups); err != nil { + t.Fatal(err) + } + if err := os.MkdirAll(primaryBackups, 0o000); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = os.Chmod(primaryBackups, 0o755) }) + if _, err := os.Stat(f.primaryUnit); err == nil { + t.Fatalf("the primary unit is still readable at %q — the test would prove nothing", f.primaryUnit) + } + + res, err := f.m.RestoreTier2Unit("app") + if err != nil { + t.Fatalf("the restore must succeed from the mirror ALONE — this is the whole point of R-102: %v", err) + } + if got := f.fake.gotEnv["SUBDOMAIN"]; got != "only-the-mirror-survives" { + t.Errorf("SUBDOMAIN=%q — the mirror was not the source", got) + } + if got := f.fake.gotEnv["SECRET_KEY"]; got != "key-primary" { + t.Errorf("the data-encrypting key did not come back from the mirrored unit: %q", got) + } + if res.VolumesReplayed != 2 { + t.Errorf("VolumesReplayed=%d, want 2 from the mirror", res.VolumesReplayed) + } + if res.DBsReplayed != 1 { + t.Errorf("DBsReplayed=%d, want 1 from the mirror", res.DBsReplayed) + } + wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase)) + if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol { + t.Errorf("volume source = %v, want %q", *f.volDirs, wantVol) + } +} + +// B3 — TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag. Every restore in this manager shares one +// running flag. The Tier-2 unit route must be inside it, not beside it (R-351b). +func TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "") + + // A backup/restore is already in flight. + if err := f.m.acquireRunning(); err != nil { + t.Fatal(err) + } + _, err := f.m.RestoreTier2Unit("app") + if err == nil { + t.Fatal("a second restore started while one was in flight") + } + if !strings.Contains(err.Error(), "already in progress") { + t.Errorf("refusal = %v, want the single-writer refusal", err) + } + if f.fake.stopped { + t.Error("the app was stopped by a restore that should never have started") + } + f.m.releaseRunning() + + // And the flag is RELEASED afterwards, or the next restore would be refused forever. + if _, err := f.m.RestoreTier2Unit("app"); err != nil { + t.Fatalf("restore after the flag cleared: %v", err) + } + if err := f.m.acquireRunning(); err != nil { + t.Errorf("the running flag was not released after the Tier-2 unit restore: %v", err) + } + f.m.releaseRunning() +} + +// B4 — TestR102_NoTier2CopyIsAnHonestRefusal. No recorded copy at all: refuse with the reason the +// file restore already uses, and take no outage. +func TestR102_NoTier2CopyIsAnHonestRefusal(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "") + // Forget the recorded copy entirely — the "this app has never had a Tier-2 run" shape. + if err := f.m.settings.SetCrossDriveConfig("app", &settings.CrossDriveBackup{}); err != nil { + t.Fatal(err) + } + _, err := f.m.RestoreTier2Unit("app") + if err == nil { + t.Fatal("expected a refusal with no recorded copy") + } + if !strings.Contains(err.Error(), errNoTier2Copy.Error()) { + t.Errorf("refusal = %v, want errNoTier2Copy", err) + } + if f.fake.stopped { + t.Error("the app was stopped despite the refusal") + } +} + +// B5 — TestR102_CopyAgeIsCarriedToTheSurface. The action overwrites live data with a copy of a +// certain age, so the age has to reach the surface. R-101 governs WHICH date: the success anchor, +// never the attempt clock, and Tier2CopyDate says which one it returned. +func TestR102_CopyAgeIsCarriedToTheSurface(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "") + + cov, err := f.m.Tier2RestoreCoverage("app") + if err != nil { + t.Fatalf("coverage: %v", err) + } + if cov.CopyLastSuccess == "" { + t.Fatal("the successful Tier-2 run left no LastSuccess for the surface to name") + } + date, proven := cov.Tier2CopyDate() + if !proven || date != cov.CopyLastSuccess { + t.Errorf("Tier2CopyDate() = (%q,%v), want the success anchor %q", date, proven, cov.CopyLastSuccess) + } + + // R-101's other half: a later FAILED attempt must not become the date shown. LastRun advances, + // LastSuccess does not, and the surface must keep naming the copy that actually exists. + older := cov.CopyLastSuccess + if err := f.m.settings.UpdateCrossDriveStatus("app", func(c *settings.CrossDriveBackup) { + c.LastRun = "2099-01-01T00:00:00Z" + c.LastStatus = "error" + }); err != nil { + t.Fatal(err) + } + cov2, err := f.m.Tier2RestoreCoverage("app") + if err != nil { + t.Fatalf("coverage after a failed attempt: %v", err) + } + date2, proven2 := cov2.Tier2CopyDate() + if !proven2 || date2 != older { + t.Errorf("after a failed attempt the surface would name %q (proven=%v); want the last SUCCESS %q", date2, proven2, older) + } +} + +// TestR102_Tier2UnitRestoreDoesNotWriteTheMirror — a restore reads its source. If the Tier-2 copy +// were mutated, the second drive would stop being a way back the moment it was used once. +func TestR102_Tier2UnitRestoreDoesNotWriteTheMirror(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) + before := fingerprintTree(t, f.destBase) + if _, err := f.m.RestoreTier2Unit("app"); err != nil { + t.Fatalf("restore: %v", err) + } + if after := fingerprintTree(t, f.destBase); after != before { + t.Error("the Tier-2 copy was written to by a restore that only reads it") + } +} diff --git a/controller/internal/backup/r102_unit_at_test.go b/controller/internal/backup/r102_unit_at_test.go new file mode 100644 index 0000000..a99416a --- /dev/null +++ b/controller/internal/backup/r102_unit_at_test.go @@ -0,0 +1,369 @@ +package backup + +import ( + "context" + "io" + "log" + "os" + "path/filepath" + "strings" + "testing" +) + +// R-102 Group A — the recovery-unit restore can be pointed at a unit ANYWHERE. +// +// The defect: every reader of a recovery unit could only name a path under `backups/primary/` +// (appbackup/paths.go joined it literally), while Tier-2 mirrored each app's whole unit to +// `/backups/secondary//recovery-unit/` on every run. So the mirror was written nightly for +// months and read by nothing — and it was unreadable in exactly the failure Tier-2 exists for, where +// the primary drive and its unit are gone (07-backup-architecture §6.3, §7.2). +// +// Every test below asserts WHICH DIRECTORY was read, not merely that a restore succeeded. A test that +// only checked `err == nil` would have passed against the pre-fix code, because the pre-fix code read +// a real unit — just never the one that survives. + +// r102Unit captures a REAL recovery unit for "app" onto its own fresh drive, with an identifying +// SUBDOMAIN, a portable DB password and a portable data key, plus optional volume tars and a DB dump. +// +// The unit is produced by the production CaptureRecoveryUnit, not hand-written: the capture side and +// the restore side must meet at real bytes, so a change to the on-disk shape cannot pass by having a +// test agree with itself (the reason captureFixtureUnit is built the same way). +func r102Unit(t *testing.T, marker string, volTars []string, dbDump string) (drive, unitDir string) { + t.Helper() + tmp := t.TempDir() + drive = filepath.Join(tmp, "drive") + return drive, r102UnitOnDrive(t, drive, marker, volTars, dbDump) +} + +// r102UnitOnDrive is r102Unit against a drive the caller already owns — the Tier-2 fixture needs the +// unit to sit on the SAME drive the Tier-2 run will mirror FROM. +func r102UnitOnDrive(t *testing.T, drive, marker string, volTars []string, dbDump string) (unitDir string) { + t.Helper() + tmp := t.TempDir() + stackDir := filepath.Join(tmp, "stack") + if err := os.MkdirAll(stackDir, 0o755); err != nil { + t.Fatal(err) + } + mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"), + "services:\n app:\n image: example/app:1\n db:\n image: postgres:16\n") + mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: App\n") + mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: "+marker+"\n") + + info := RecoveryInfo{ + StackDir: stackDir, + DisplayName: "App", + ImagePins: []string{"example/app:1"}, + NonSecretEnv: map[string]string{"SUBDOMAIN": marker}, + SecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"}, + DataKeyEnvVars: []string{"SECRET_KEY"}, + PortableSecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"}, + PortableSecrets: map[string]string{"DB_PASSWORD": "pw-" + marker, "SECRET_KEY": "key-" + marker}, + } + unitDir = RecoveryUnitPath(drive, "app") + // The dumps are written BEFORE the capture so the manifest enumerates them — a manifest that + // lists nothing is the R-353 "the backup held only settings" shape, which is a different case. + for _, v := range volTars { + mustWrite(t, filepath.Join(UnitVolumeDumpDir(unitDir), v), "tar:"+v+":"+marker) + } + if dbDump != "" { + mustWrite(t, filepath.Join(UnitDBDumpDir(unitDir), "app-postgres.sql"), dbDump) + } + + m := &Manager{ + logger: log.New(io.Discard, "", 0), + systemDataPath: filepath.Join(tmp, "system"), + stackProvider: &fakeRecoveryProvider{info: info, hdd: drive}, + version: "vtest", + } + if err := m.CaptureRecoveryUnit("app"); err != nil { + t.Fatalf("capture fixture %q: %v", marker, err) + } + return unitDir +} + +// r102Manager builds a restore-side Manager whose LIVE drive is liveDrive, with the Docker and DB +// seams injected so no daemon is touched. The returned slices record the directories each leg read. +func r102Manager(t *testing.T, liveDrive string) (m *Manager, fake *fakeRecoveryProvider, volDirs, dbPaths *[]string) { + t.Helper() + fake = &fakeRecoveryProvider{hdd: liveDrive, running: true} + m = &Manager{ + logger: log.New(io.Discard, "", 0), + systemDataPath: filepath.Join(liveDrive, "..", "sys"), + stackProvider: fake, + } + var vd, dp []string + m.volumeReplayFrom = func(_, dumpDir string) (int, error) { + vd = append(vd, dumpDir) + entries, err := os.ReadDir(dumpDir) + if err != nil { + return 0, nil + } + n := 0 + for _, e := range entries { + if strings.HasSuffix(e.Name(), ".tar") { + n++ + } + } + return n, nil + } + m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { + return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil + } + m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error { + dp = append(dp, p) + return nil + } + return m, fake, &vd, &dp +} + +// A2 — TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir. +// +// Two complete units exist on disk with DIFFERENT contents. The restore is pointed at the second one +// and must read every leg — env, secrets, volume tars, DB dump — out of THAT directory. The first +// unit is the app's own primary unit and is present and perfectly readable, which is what makes the +// assertion mean something: a restore that ignored unitDir would still succeed, and would silently +// return the wrong copy's data. +func TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir(t *testing.T) { + primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1)) + _, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar", "vol_b.tar"}, pgDump(2)) + + m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive) + + res, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit) + if err != nil { + t.Fatalf("restore from the named unit: %v", err) + } + if fake.gotEnv == nil { + t.Fatal("recreate was never called — the restore did not reach the redeploy") + } + if got := fake.gotEnv["SUBDOMAIN"]; got != "mirror" { + t.Errorf("config came from the WRONG unit: SUBDOMAIN=%q, want %q", got, "mirror") + } + if got := fake.gotEnv["SECRET_KEY"]; got != "key-mirror" { + t.Errorf("data key came from the WRONG unit: %q", got) + } + if got := fake.gotEnv["DB_PASSWORD"]; got != "pw-mirror" { + t.Errorf("DB password came from the WRONG unit: %q", got) + } + if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(mirrorUnit) { + t.Errorf("volume tars read from %v, want %q", *volDirs, UnitVolumeDumpDir(mirrorUnit)) + } + if res.VolumesReplayed != 2 { + t.Errorf("VolumesReplayed=%d, want 2 (the mirror holds two tars; the primary holds one)", res.VolumesReplayed) + } + if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(mirrorUnit) { + t.Errorf("DB dump read from %v, want a file under %q", *dbPaths, UnitDBDumpDir(mirrorUnit)) + } + // The negative control: nothing was read out of the primary unit. + for _, d := range *volDirs { + if strings.HasPrefix(d, primaryUnit) { + t.Errorf("a leg was read from the PRIMARY unit %q despite a mirror being named", d) + } + } +} + +// A3 — TestR102_PrimaryPathIsUnchanged. The one-argument wrapper must still resolve to the app's own +// drive, `backups/primary/`. Asserted on the DIRECTORIES the legs actually read, not on a +// path expression re-derived in the test. +func TestR102_PrimaryPathIsUnchanged(t *testing.T) { + primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1)) + m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive) + + if _, err := m.RestoreFromRecoveryUnit("app"); err != nil { + t.Fatalf("primary restore: %v", err) + } + if got := fake.gotEnv["SUBDOMAIN"]; got != "primary" { + t.Errorf("SUBDOMAIN=%q, want the primary unit's %q", got, "primary") + } + if !strings.Contains(primaryUnit, filepath.Join("backups", "primary", "app")) { + t.Fatalf("fixture is not where the primary unit belongs: %q", primaryUnit) + } + if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(primaryUnit) { + t.Errorf("volume dir = %v, want %q", *volDirs, UnitVolumeDumpDir(primaryUnit)) + } + if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(primaryUnit) { + t.Errorf("db dump dir = %v, want %q", *dbPaths, UnitDBDumpDir(primaryUnit)) + } +} + +// A4 — TestR102_LiveDestinationIsUnchanged. THE SOURCE MOVES; THE DESTINATION DOES NOT. +// +// The restore reads a mirror that lives on a different drive entirely, and must still write the app +// back to its own live namespace: the definition through RecreateStackDefinitionFromUnit (whose +// compose dir must be the MIRROR's, since that is the definition being restored), and the data into +// the LIVE Docker volumes and the LIVE database container. A restore that also relocated the app's +// data would be a migration, not a restore. +// +// The observable for "the destination did not move" is that nothing under the mirror's own drive was +// written to: the mirror tree is byte-identical before and after. +func TestR102_LiveDestinationIsUnchanged(t *testing.T) { + primaryDrive, _ := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1)) + mirrorDrive, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2)) + + before := fingerprintTree(t, mirrorDrive) + + m, fake, _, _ := r102Manager(t, primaryDrive) + var recreateComposeDir string + m.stackProvider = &r102RecordingProvider{fakeRecoveryProvider: fake, composeDirOut: &recreateComposeDir} + + if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil { + t.Fatalf("restore: %v", err) + } + if recreateComposeDir != UnitComposeDir(mirrorUnit) { + t.Errorf("recreate read compose from %q, want the mirror's %q", recreateComposeDir, UnitComposeDir(mirrorUnit)) + } + if after := fingerprintTree(t, mirrorDrive); after != before { + t.Error("the SOURCE mirror was written to — a restore reads its source and never writes it") + } + // The live drive is where the app lives, and the restore must not have relocated it: the app's + // own drive path is still the one the manager resolves for it. + if got := m.GetAppDrivePath("app"); got != primaryDrive { + t.Errorf("the app's live drive moved to %q, want %q", got, primaryDrive) + } +} + +// r102RecordingProvider captures the compose directory RecreateStackDefinitionFromUnit is handed — +// the one place the restore names a source directory to the guest side. +type r102RecordingProvider struct { + *fakeRecoveryProvider + composeDirOut *string +} + +func (f *r102RecordingProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error { + *f.composeDirOut = composeDir + return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv) +} + +// A5 — TestR102_MutationOrderIsPreserved. R-47's ordering is pinned and R-102 must not have moved it: +// stop → volumes → recreate → DB-only start → replay → full start. The DB-only phase exists because +// starting the whole stack first let the application rebuild schema underneath the replay (H4, +// DIAG-immich-restore-round2-2026-07-19). +// +// The state AT REPLAY TIME is asserted, not just the call sequence — H4 looked correctly ordered and +// the app was up. +// +// Red-proof (recorded in REPORT.md): swap the volume replay and the recreate step in +// RestoreFromRecoveryUnitAt → this test fails on the sequence assertion. +func TestR102_MutationOrderIsPreserved(t *testing.T) { + primaryDrive, _ := r102Unit(t, "primary", nil, pgDump(1)) + _, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2)) + + m, fake, _, _ := r102Manager(t, primaryDrive) + + var volumesDoneAtRecreate, fullUpAtReplay bool + var volumesReplayed bool + m.volumeReplayFrom = func(_, _ string) (int, error) { + volumesReplayed = true + if fake.gotEnv != nil { + t.Error("volumes were replayed AFTER the definition was recreated") + } + return 1, nil + } + prov := &r102OrderProvider{fakeRecoveryProvider: fake, volumesReplayed: &volumesReplayed, seen: &volumesDoneAtRecreate} + m.stackProvider = prov + m.importDBDump = func(context.Context, DiscoveredDB, string) error { + fullUpAtReplay = fake.fullStarted + return nil + } + + if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil { + t.Fatalf("restore: %v", err) + } + if got := strings.Join(fake.calls, ","); got != "stop,recreate,startsvc:db,start" { + t.Fatalf("sequence = %q, want stop → recreate → db-only start → (replay) → full start", got) + } + if !volumesDoneAtRecreate { + t.Error("the volume replay had not run when the definition was recreated") + } + if fullUpAtReplay { + t.Error("the FULL stack was already up when the DB replay fired — this is H4 exactly (R-47)") + } +} + +// r102OrderProvider records whether the volume replay had already happened by the time the definition +// was recreated — the ordering fact the call log alone cannot carry. +type r102OrderProvider struct { + *fakeRecoveryProvider + volumesReplayed *bool + seen *bool +} + +func (f *r102OrderProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error { + *f.seen = *f.volumesReplayed + return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv) +} + +// A6 — TestR102_MissingManifestInMirrorFailsClosed. A DIRECTORY IS NOT A PACKAGE. +// +// `/backups/secondary//recovery-unit/` can exist and be useless: a copy interrupted +// mid-run, or a tree whose manifest.json never landed. The Tier-2 unit restore must refuse it and +// must leave the app running — a restore armed over an unopenable unit would stop the app, replay +// nothing, and rewrite its definition from an empty capture. Same lesson as R-358 one tier over. +// +// Placed with Group A because it is the fail-closed half of the parameterised unit; the route it +// exercises is Part 1.3's. +func TestR102_MissingManifestInMirrorFailsClosed(t *testing.T) { + for _, tc := range []struct { + name string + setup func(t *testing.T, unitDir string) + }{ + {"manifest absent", func(t *testing.T, unitDir string) { + if err := os.Remove(UnitManifestFile(unitDir)); err != nil { + t.Fatal(err) + } + }}, + {"manifest unparseable", func(t *testing.T, unitDir string) { + mustWrite(t, UnitManifestFile(unitDir), "{ this is not json") + }}, + {"recovery-unit absent entirely", func(t *testing.T, unitDir string) { + if err := os.RemoveAll(unitDir); err != nil { + t.Fatal(err) + } + }}, + } { + t.Run(tc.name, func(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) + tc.setup(t, tier2UnitDir(f.destBase)) + + cov, covErr := f.m.Tier2RestoreCoverage("app") + if covErr != nil { + t.Fatalf("coverage: %v", covErr) + } + if cov.CanRestoreUnit() { + t.Error("CanRestoreUnit() said yes over an unopenable mirror — the surface would offer the action") + } + _, err := f.m.RestoreTier2Unit("app") + if err == nil { + t.Fatal("the restore ran over an unopenable mirror") + } + if !strings.Contains(err.Error(), ErrTier2NoUnitInCopy.Error()) { + t.Errorf("refusal = %v, want ErrTier2NoUnitInCopy", err) + } + if f.fake.stopped { + t.Error("the app was STOPPED despite the refusal — the outage this gate exists to avoid") + } + if f.fake.gotEnv != nil { + t.Error("the app's definition was rewritten despite the refusal") + } + }) + } +} + +// TestR102_UnopenableMirrorIsStillDisclosedAsUnread is the other side of A6, and the reason +// UnitRestorable is a second field rather than a widening of HasUnit: a half-copied mirror cannot be +// restored, but it IS captured data the file restore is not looking at, so the disclosure must stand. +func TestR102_UnopenableMirrorIsStillDisclosedAsUnread(t *testing.T) { + f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) + mustWrite(t, UnitManifestFile(tier2UnitDir(f.destBase)), "{ not json") + + cov, err := f.m.Tier2RestoreCoverage("app") + if err != nil { + t.Fatalf("coverage: %v", err) + } + if !cov.HasUnit { + t.Error("HasUnit went false for a mirror that exists — the file restore would stop disclosing unread data") + } + if cov.UnitRestorable { + t.Error("UnitRestorable stayed true over an unparseable manifest") + } +} diff --git a/controller/internal/backup/restore_db.go b/controller/internal/backup/restore_db.go index 2fb99fc..0ed576b 100644 --- a/controller/internal/backup/restore_db.go +++ b/controller/internal/backup/restore_db.go @@ -89,10 +89,25 @@ func (m *Manager) reimportDBDumpsFrom(ctx context.Context, stackName, dumpDir st return imported, nil } +// dbReimportTimeout bounds a DB replay so a stuck import cannot hang a restore indefinitely. Named +// once because both bounded entry points below must agree: two paths that differ in how long they +// let a wedged import hold the restore are two different products (R-102 added the second one). +const dbReimportTimeout = 35 * time.Minute + // reimportDBDumpsCtx is a small helper that runs reimportDBDumps with a bounded context so a stuck DB // import cannot hang the restore indefinitely. func (m *Manager) reimportDBDumpsCtx(stackName, nsRoot string) (int, error) { - ctx, cancel := context.WithTimeout(context.Background(), 35*time.Minute) + ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout) defer cancel() return m.reimportDBDumps(ctx, stackName, nsRoot) } + +// reimportDBDumpsAtCtx is reimportDBDumpsCtx with an EXPLICIT dump directory — the bounded-context +// twin of reimportDBDumpsFrom, added for R-102 so the Tier-2 unit restore can replay out of the +// secondary mirror. Same discovery/import seams, same failure semantics, same bound; only the source +// directory differs. +func (m *Manager) reimportDBDumpsAtCtx(stackName, dumpDir string) (int, error) { + ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout) + defer cancel() + return m.reimportDBDumpsFrom(ctx, stackName, dumpDir) +} diff --git a/controller/internal/backup/restore_unit.go b/controller/internal/backup/restore_unit.go index 07ffa76..fac3a27 100644 --- a/controller/internal/backup/restore_unit.go +++ b/controller/internal/backup/restore_unit.go @@ -178,7 +178,36 @@ type UnitRestoreResult struct { // D5: this no longer needs the guest. A restore with the guest's app.yaml absent succeeds, which is // pinned by TestRestoreFromRecoveryUnitWithGuestAbsent — the withheld class is regenerated (O4) and // only a data key missing from BOTH sources still refuses. +// +// R-102: this is now the thin caller. It names the PRIMARY unit — the app's own drive, +// backups/primary/ — and hands it to RestoreFromRecoveryUnitAt, which holds the whole body. +// An unresolvable drive path is still refused inside …At, in the same place and with the same +// message, so the order of the checks a caller can observe is unchanged. func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, error) { + return m.RestoreFromRecoveryUnitAt(stackName, RecoveryUnitPath(m.namespaceRoot(m.GetAppDrivePath(stackName)), stackName)) +} + +// RestoreFromRecoveryUnitAt is RestoreFromRecoveryUnit with an EXPLICIT recovery-unit directory. +// +// R-102. ONE implementation, two callers — the same rule restoreDockerVolumesFrom states beside +// itself in restore.go, and for the same reason: a second copy of this body is exactly how the local +// path and the off-site path drifted apart until nothing compared them. +// +// The reason it exists: Tier-2 mirrors the app's whole recovery unit to +// /backups/secondary//recovery-unit/ on every run, and until now every reader of a unit +// could only name a path under backups/primary/. So in the one failure Tier-2 exists for — the +// primary drive is lost, taking the primary unit with it — the surviving copy could not be opened by +// any action in the product (07-backup-architecture §6.3, §7.2). +// +// THE SOURCE MOVES; THE DESTINATION DOES NOT. unitDir changes only where the manifest, the compose +// capture, the .sql dumps and the volume tars are READ from. The app's data is written back to the +// live Docker volumes and the live database container, and its definition to the guest, exactly as +// before — a restore that also relocated the app's data would be a migration, not a restore. +// +// Everything else is pinned and unchanged: the mutation order (stop → volumes → recreate → DB-only +// start → replay → start, R-47), the secret reconciliation with unit-over-guest precedence and the +// fail-closed data-key gate, and the no-unit fallback to RestoreApp with its CountsUnknown handling. +func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRestoreResult, error) { var res UnitRestoreResult if m.stackProvider == nil { return res, fmt.Errorf("stack provider not configured") @@ -197,15 +226,18 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, m.mu.Unlock() }() + // The DESTINATION side, and it is deliberately still resolved here: RestoreApp (the no-unit + // fallback below) needs it, and 07-backup-architecture §6.3 records that the restore destination + // is resolved by the same rule as the capture destination. It is no longer used to derive any + // SOURCE path — that is what unitDir is for. drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { return res, fmt.Errorf("cannot determine drive path for %s", stackName) } - nsRoot := m.namespaceRoot(drivePath) - manifest := readManifest(RecoveryUnitManifestPath(nsRoot, stackName)) + manifest := readManifest(UnitManifestFile(unitDir)) if manifest == nil { - m.logger.Printf("[WARN] [backup] No recovery unit for %s — falling back to volume-only restore", stackName) + m.logger.Printf("[WARN] [backup] No readable recovery unit for %s at %s — falling back to volume-only restore", stackName, unitDir) m.mu.Lock() m.running = false // RestoreApp re-acquires the running flag m.mu.Unlock() @@ -220,7 +252,7 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, // was just parsed above, so the claim and the outcome are counted from the same document. res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps) - composeDir := RecoveryUnitComposePath(nsRoot, stackName) + composeDir := UnitComposeDir(unitDir) nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars) // D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The @@ -275,8 +307,11 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, stackName, len(unresolved), unresolved) } } - m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit: images=%d, secrets recovered=%d/%d, data_keys=%d", - stackName, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars)) + // R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with + // two answers, and "an absent log line is not evidence" — the drill reads this line to prove the + // secondary mirror, not the primary unit, was the source. It is a path, never a secret. + m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit %s: images=%d, secrets recovered=%d/%d, data_keys=%d", + stackName, unitDir, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars)) // R-47: which compose service holds the database, and is there anything to replay? Resolved from // the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed @@ -286,7 +321,8 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, // "cannot tell" is not "no database" — leave it empty and let the gate decide. m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr) } - hasDumps := hasReplayableDump(AppDBDumpPath(nsRoot, stackName)) + dbDumpDir := UnitDBDumpDir(unitDir) + hasDumps := hasReplayableDump(dbDumpDir) if hasDumps && len(dbServices) == 0 { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName) return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stackName) @@ -302,7 +338,17 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, // R-353: the count is captured even when the replay errors — a partial replay is a fact the // customer's sentence has to be built from, and discarding it on the error path is how Scenario C // would end up wearing Scenario B's wording. - replayed, volErr := m.restoreDockerVolumes(stackName, drivePath) + // R-354's volume-REPLAY seam, reused here rather than a second one being invented. It defaults to + // the real restoreDockerVolumesFrom, so production behaviour is byte-for-byte what it was; what it + // buys is that R-102's acceptance test can assert WHICH directory the tars came out of without a + // Docker daemon. For 40 of the 53 catalogue apps that archive is the entire dataset, so "the + // mirror was the source" has to be provable for the volume leg too, not only for the env and the + // database. + volReplay := m.volumeReplayFrom + if volReplay == nil { + volReplay = m.restoreDockerVolumesFrom + } + replayed, volErr := volReplay(stackName, UnitVolumeDumpDir(unitDir)) res.VolumesReplayed = replayed if volErr != nil { m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, volErr) @@ -322,7 +368,7 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, if dataErr == nil { dataErr = err } - } else if n, err := m.reimportDBDumpsCtx(stackName, nsRoot); err != nil { + } else if n, err := m.reimportDBDumpsAtCtx(stackName, dbDumpDir); err != nil { res.DBsReplayed = n // partial credit: whatever imported before the failure really did import m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err) if dataErr == nil { diff --git a/controller/internal/backup/tier2_restore.go b/controller/internal/backup/tier2_restore.go index bd5fa1d..56c3d94 100644 --- a/controller/internal/backup/tier2_restore.go +++ b/controller/internal/backup/tier2_restore.go @@ -39,6 +39,12 @@ var ( // are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action // that does work, instead of taking an outage and reporting "no missing files". ErrTier2NoRestorableData = errors.New("ennek az alkalmazásnak az adatai nem ebből a másolatból állíthatók vissza") + // ErrTier2NoUnitInCopy (R-102) — the recorded Tier-2 copy holds no OPENABLE recovery unit: either + // recovery-unit/ is absent, or it is a directory without a readable manifest.json. Exported so the + // handler can refuse before beginning any op. FAIL CLOSED is the whole point of the second half: + // a directory that exists is not a package, and reading a half-copied mirror as if it were one is + // how a restore would overwrite live data with nothing. + ErrTier2NoUnitInCopy = errors.New("a másodlagos másolatban nincs megnyitható mentési egység ehhez az alkalmazáshoz") ) // Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1 @@ -50,14 +56,64 @@ var ( // tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in // which case the restore is a guaranteed no-op no matter how much data was lost. type Tier2Coverage struct { - Legs []string // subtrees this restore reads and that exist in the copy: "hdd", "userdata" - HasUnit bool // recovery-unit/ present — captured, but NOT restorable by this path + Legs []string // subtrees the FILE restore reads and that exist in the copy: "hdd", "userdata" + HasUnit bool // recovery-unit/ present as a DIRECTORY — the disclosure fact, see below + + // UnitRestorable (R-102) — the mirror is a real PACKAGE, not merely a directory: recovery-unit/ + // exists AND carries a manifest.json that parses. This is the gate for the UNIT restore. + // + // It is a SECOND FIELD and not a widening of HasUnit, and the distinction is load-bearing in both + // directions. HasUnit answers "is there captured data this FILE restore is not looking at?" — the + // question tier2UnitNotCoveredMsg is appended for, and the honest answer for a half-copied mirror + // is still yes. UnitRestorable answers "can the unit restore open this?" — and for that same + // half-copied mirror the answer is no. Collapsing them would either silence a true disclosure or + // arm a restore over an unopenable package. + UnitRestorable bool + + // CopyLastRun / CopyLastSuccess — WHEN the copy this restore would read was written, so the + // surface can name the date before it overwrites anything with it (Scenario E). Filled only by + // Tier2RestoreCoverage, which is the path that holds the settings; tier2CoverageAt is a pure + // filesystem inspection and leaves them empty. They are STRINGS in the recorded RFC3339 form, + // carried verbatim — no formatting decision is taken in this package. + // + // R-101 applies here exactly as it does on the backup card: CopyLastRun is the ATTEMPT clock and + // CopyLastSuccess is the only evidence a copy was actually made. A surface that shows one must + // not present it as the other. + CopyLastRun string + CopyLastSuccess string } -// CanRestore reports whether the restore has any subtree to read at all. +// CanRestore reports whether the FILE restore has any subtree to read at all. +// +// R-102/R-103: this answers exactly one question and must keep answering only that one. Widening it +// to include the unit is R-356 arriving a second time — there, ONE predicate meant both "has this app +// a drive?" and "is this app installed?", and it refused 40 running apps for months while telling +// their owners to reinstall them somewhere those apps never offer. Two questions, two predicates. func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 } -// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat — no side effects. +// CanRestoreUnit reports whether the UNIT restore can run from this copy — the second predicate. +func (c Tier2Coverage) CanRestoreUnit() bool { return c.UnitRestorable } + +// tier2UnitDir returns the recovery-unit directory inside a resolved Tier-2 copy. ONE expression of +// where Tier-2 puts the mirror; tier2.go writes it at the same relative name ("Unit leg (always)"). +func tier2UnitDir(destBase string) string { + return filepath.Join(destBase, "recovery-unit") +} + +// tier2UnitIsOpenable reports whether a mirrored unit directory is a PACKAGE and not just a +// directory. Fail-closed by construction: the manifest must be present AND parse (readManifest +// returns nil for both a missing file and malformed JSON), because an unopenable unit that armed a +// restore would stop the app, replay nothing, and rewrite its definition from an empty capture. +// +// This is the R-358 lesson one tier over — the off-site scratch marker — arriving on Tier-2. +func tier2UnitIsOpenable(unitDir string) bool { + if fi, err := os.Stat(unitDir); err != nil || !fi.IsDir() { + return false + } + return readManifest(UnitManifestFile(unitDir)) != nil +} + +// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat/read — no side effects. func tier2CoverageAt(destBase string) Tier2Coverage { var c Tier2Coverage for _, leg := range []string{"hdd", "userdata"} { @@ -65,9 +121,11 @@ func tier2CoverageAt(destBase string) Tier2Coverage { c.Legs = append(c.Legs, leg) } } - if fi, err := os.Stat(filepath.Join(destBase, "recovery-unit")); err == nil && fi.IsDir() { + unitDir := tier2UnitDir(destBase) + if fi, err := os.Stat(unitDir); err == nil && fi.IsDir() { c.HasUnit = true } + c.UnitRestorable = tier2UnitIsOpenable(unitDir) return c } @@ -79,7 +137,61 @@ func (m *Manager) Tier2RestoreCoverage(stackName string) (Tier2Coverage, error) if err != nil { return Tier2Coverage{}, err } - return tier2CoverageAt(destBase), nil + cov := tier2CoverageAt(destBase) + // R-102: carry WHEN the copy was written, so the surface can name the date on an action that + // overwrites live data with it. Read from the same recorded config tier2RecordedCopyDir just + // resolved the path from, so the date and the directory cannot describe different runs. + if m.settings != nil { + if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil { + cov.CopyLastRun, cov.CopyLastSuccess = cfg.LastRun, cfg.LastSuccess + } + } + return cov, nil +} + +// RestoreTier2Unit runs the FULL recovery-unit restore from the app's Tier-2 copy on the SECOND +// DRIVE — R-102, and the reason this task exists. +// +// Tier-2 has mirrored each app's whole recovery unit to +// /backups/secondary//recovery-unit/ on every run for months, and no code path read it: +// every reader of a unit could only name a path under backups/primary/. So in the exact failure +// Tier-2 exists for — the primary drive is lost, and the primary unit with it — the surviving copy +// was unopenable by any customer action (07-backup-architecture §6.3, §7.2). +// +// It is NOT the additive file restore beside it. This one OVERWRITES: named volumes are recreated +// from the mirror's tars and the database is replayed from the mirror's dump. The surface must carry +// that difference; see tier2UnitConfirm in the web package. +// +// THE SINGLE-WRITER FLAG IS TAKEN INSIDE RestoreFromRecoveryUnitAt, exactly as on the primary path — +// do NOT add an acquireRunning() here. A second acquire would refuse the restore it is guarding. +func (m *Manager) RestoreTier2Unit(stackName string) (UnitRestoreResult, error) { + destBase, err := m.tier2RecordedCopyDir(stackName) + if err != nil { + return UnitRestoreResult{}, err + } + unitDir := tier2UnitDir(destBase) + if !tier2UnitIsOpenable(unitDir) { + m.logger.Printf("[WARN] [backup] Tier-2 unit restore refused for %s: no openable recovery unit in the recorded copy — the app was NOT stopped", stackName) + return UnitRestoreResult{}, ErrTier2NoUnitInCopy + } + // WARN, not INFO: this is the destructive one of the two Tier-2 restores. The unit directory is a + // path and never a secret, and naming it is what makes "the SECONDARY mirror was the source" a + // positive observable in the log rather than an absence to be argued from. + m.logger.Printf("[WARN] [backup] Tier-2 UNIT restore for %s from the secondary mirror %s — this OVERWRITES live app data", stackName, unitDir) + return m.RestoreFromRecoveryUnitAt(stackName, unitDir) +} + +// Tier2CopyDate returns the date the surface should name for this app's Tier-2 copy, preferring the +// last SUCCESS over the last ATTEMPT (R-101: a timestamp that records "we tried" cannot answer "did +// it work"), and reports whether the returned value is a proven success. +// +// It exists so the confirm text and the outcome sentence cannot disagree about which copy is being +// restored: one resolver, two readers. +func (c Tier2Coverage) Tier2CopyDate() (date string, proven bool) { + if c.CopyLastSuccess != "" { + return c.CopyLastSuccess, true + } + return c.CopyLastRun, false } // tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every