package backup import ( "context" "io" "log" "os" "path/filepath" "strings" "testing" "gitea.dooplex.hu/admin/felhom-controller/internal/config" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // R-102 Group B — the Tier-2 route: the mirror on the SECOND DRIVE becomes a way back. // // The mirror in every test here is written by the PRODUCTION Tier-2 run (RunTier2 with copyTree // standing in for rsync, the seam tier2_v2_test.go already uses), not by hand. That matters: the // claim is "the copy Tier-2 actually writes is the copy this restore reads", and a hand-built // fixture could only prove that the restore agrees with the test's idea of the layout. // r102T2 is a fully-wired Tier-2 fixture: a live drive carrying the app's PRIMARY recovery unit, a // second drive carrying the mirror Tier-2 just wrote, and a restore-side Manager with the Docker and // DB seams injected so no daemon is touched. type r102T2 struct { m *Manager fake *fakeRecoveryProvider liveDrive string destDrive string destBase string primaryUnit string volDirs *[]string dbPaths *[]string } // r102Tier2Fixture captures a unit on the live drive, runs the REAL RunTier2 to mirror it onto the // second drive, then returns a Manager ready to restore. The mirror's contents differ from the // primary's afterwards when the caller mutates one of them — B1 and B2 both depend on being able to // tell the two copies apart. func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 { t.Helper() tmp := t.TempDir() live := filepath.Join(tmp, "usb") dest := filepath.Join(tmp, "flash") sys := filepath.Join(tmp, "sys") primaryUnit := r102UnitOnDrive(t, live, "primary", volTars, dbDump) sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0)) if err != nil { t.Fatal(err) } if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil { t.Fatal(err) } fake := &fakeRecoveryProvider{hdd: live, running: true} cfg := &config.Config{} cfg.Paths.SystemDataPath = sys cfg.Paths.DataDir = filepath.Join(tmp, "data") m := NewManager(cfg, sett, log.New(io.Discard, "", 0)) m.stackProvider = fake m.systemDataPath = sys m.tier2Mirror = copyTree m.tier2SSDFits = func(string, int64) bool { return true } m.samePhysicalDevice = oneDrivePerSubtree if err := m.RunTier2("app"); err != nil { t.Fatalf("RunTier2 (the code that writes the mirror this task reads): %v", err) } destBase := filepath.Join(dest, "backups", "secondary", "app") if _, sErr := os.Stat(UnitManifestFile(tier2UnitDir(destBase))); sErr != nil { t.Fatalf("Tier-2 did not mirror the unit — the fixture proves nothing: %v", sErr) } var vd, dp []string m.volumeReplayFrom = func(_, dumpDir string) (int, error) { vd = append(vd, dumpDir) entries, rErr := os.ReadDir(dumpDir) if rErr != nil { return 0, nil } n := 0 for _, e := range entries { if strings.HasSuffix(e.Name(), ".tar") { n++ } } return n, nil } m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil } m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error { dp = append(dp, p) return nil } return &r102T2{m: m, fake: fake, liveDrive: live, destDrive: dest, destBase: destBase, primaryUnit: primaryUnit, volDirs: &vd, dbPaths: &dp} } // markMirror rewrites the MIRROR's app.yaml so the two copies are distinguishable. It edits the copy, // never the primary, so a restore that read the primary would return the pre-edit value. func (f *r102T2) markMirror(t *testing.T, marker string) { t.Helper() p := filepath.Join(UnitComposeDir(tier2UnitDir(f.destBase)), "app.yaml") b, err := os.ReadFile(p) if err != nil { t.Fatal(err) } out := strings.Replace(string(b), "SUBDOMAIN: primary", "SUBDOMAIN: "+marker, 1) if out == string(b) { t.Fatalf("mirror app.yaml did not carry the expected marker; contents:\n%s", b) } mustWrite(t, p, out) } // B1 — TestR102_Tier2UnitRestoreReadsTheSecondaryMirror. func TestR102_Tier2UnitRestoreReadsTheSecondaryMirror(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) f.markMirror(t, "from-the-mirror") res, err := f.m.RestoreTier2Unit("app") if err != nil { t.Fatalf("Tier-2 unit restore: %v", err) } if got := f.fake.gotEnv["SUBDOMAIN"]; got != "from-the-mirror" { t.Errorf("config came from the PRIMARY unit: SUBDOMAIN=%q", got) } wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase)) if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol { t.Errorf("volume tars read from %v, want the mirror's %q", *f.volDirs, wantVol) } wantDB := UnitDBDumpDir(tier2UnitDir(f.destBase)) if len(*f.dbPaths) != 1 || filepath.Dir((*f.dbPaths)[0]) != wantDB { t.Errorf("DB dump read from %v, want a file under the mirror's %q", *f.dbPaths, wantDB) } if res.VolumesReplayed != 1 || res.DBsReplayed != 1 { t.Errorf("nothing came back: volumes=%d dbs=%d", res.VolumesReplayed, res.DBsReplayed) } if res.ManifestVolumes != 1 || res.ManifestDBs != 1 { t.Errorf("the mirror's manifest did not enumerate its dumps: %+v", res) } } // B2 — TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT. // // THE ACCEPTANCE TEST. Tier-2 exists for the loss of the primary drive, and in that failure the // primary recovery unit is GONE. A test that leaves the primary in place proves the code compiles, // not that the mirror is a route (07-backup-architecture §7.2). // // The primary unit is not merely emptied — the whole `backups/` tree on the live drive is removed and // then made unreadable, so any code that tried to fall back to it would error rather than silently // find nothing. // // Red-proof (recorded in REPORT.md): point RestoreTier2Unit back at the primary unit path → this test // fails with "no readable recovery unit ... falling back to volume-only restore" behaviour and the // mirror's marker never reaching the redeploy. func TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1)) f.markMirror(t, "only-the-mirror-survives") // The primary drive's whole backup tree is destroyed, then the parent is made unreadable so a // fallback read cannot quietly return "nothing here". primaryBackups := filepath.Join(f.liveDrive, "backups") if err := os.RemoveAll(primaryBackups); err != nil { t.Fatal(err) } if err := os.MkdirAll(primaryBackups, 0o000); err != nil { t.Fatal(err) } t.Cleanup(func() { _ = os.Chmod(primaryBackups, 0o755) }) if _, err := os.Stat(f.primaryUnit); err == nil { t.Fatalf("the primary unit is still readable at %q — the test would prove nothing", f.primaryUnit) } res, err := f.m.RestoreTier2Unit("app") if err != nil { t.Fatalf("the restore must succeed from the mirror ALONE — this is the whole point of R-102: %v", err) } if got := f.fake.gotEnv["SUBDOMAIN"]; got != "only-the-mirror-survives" { t.Errorf("SUBDOMAIN=%q — the mirror was not the source", got) } if got := f.fake.gotEnv["SECRET_KEY"]; got != "key-primary" { t.Errorf("the data-encrypting key did not come back from the mirrored unit: %q", got) } if res.VolumesReplayed != 2 { t.Errorf("VolumesReplayed=%d, want 2 from the mirror", res.VolumesReplayed) } if res.DBsReplayed != 1 { t.Errorf("DBsReplayed=%d, want 1 from the mirror", res.DBsReplayed) } wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase)) if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol { t.Errorf("volume source = %v, want %q", *f.volDirs, wantVol) } } // B3 — TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag. Every restore in this manager shares one // running flag. The Tier-2 unit route must be inside it, not beside it (R-351b). func TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "") // A backup/restore is already in flight. if err := f.m.acquireRunning(); err != nil { t.Fatal(err) } _, err := f.m.RestoreTier2Unit("app") if err == nil { t.Fatal("a second restore started while one was in flight") } if !strings.Contains(err.Error(), "already in progress") { t.Errorf("refusal = %v, want the single-writer refusal", err) } if f.fake.stopped { t.Error("the app was stopped by a restore that should never have started") } f.m.releaseRunning() // And the flag is RELEASED afterwards, or the next restore would be refused forever. if _, err := f.m.RestoreTier2Unit("app"); err != nil { t.Fatalf("restore after the flag cleared: %v", err) } if err := f.m.acquireRunning(); err != nil { t.Errorf("the running flag was not released after the Tier-2 unit restore: %v", err) } f.m.releaseRunning() } // B4 — TestR102_NoTier2CopyIsAnHonestRefusal. No recorded copy at all: refuse with the reason the // file restore already uses, and take no outage. func TestR102_NoTier2CopyIsAnHonestRefusal(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "") // Forget the recorded copy entirely — the "this app has never had a Tier-2 run" shape. if err := f.m.settings.SetCrossDriveConfig("app", &settings.CrossDriveBackup{}); err != nil { t.Fatal(err) } _, err := f.m.RestoreTier2Unit("app") if err == nil { t.Fatal("expected a refusal with no recorded copy") } if !strings.Contains(err.Error(), errNoTier2Copy.Error()) { t.Errorf("refusal = %v, want errNoTier2Copy", err) } if f.fake.stopped { t.Error("the app was stopped despite the refusal") } } // B5 — TestR102_CopyAgeIsCarriedToTheSurface. The action overwrites live data with a copy of a // certain age, so the age has to reach the surface. R-101 governs WHICH date: the success anchor, // never the attempt clock, and Tier2CopyDate says which one it returned. func TestR102_CopyAgeIsCarriedToTheSurface(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "") cov, err := f.m.Tier2RestoreCoverage("app") if err != nil { t.Fatalf("coverage: %v", err) } if cov.CopyLastSuccess == "" { t.Fatal("the successful Tier-2 run left no LastSuccess for the surface to name") } date, proven := cov.Tier2CopyDate() if !proven || date != cov.CopyLastSuccess { t.Errorf("Tier2CopyDate() = (%q,%v), want the success anchor %q", date, proven, cov.CopyLastSuccess) } // R-101's other half: a later FAILED attempt must not become the date shown. LastRun advances, // LastSuccess does not, and the surface must keep naming the copy that actually exists. older := cov.CopyLastSuccess if err := f.m.settings.UpdateCrossDriveStatus("app", func(c *settings.CrossDriveBackup) { c.LastRun = "2099-01-01T00:00:00Z" c.LastStatus = "error" }); err != nil { t.Fatal(err) } cov2, err := f.m.Tier2RestoreCoverage("app") if err != nil { t.Fatalf("coverage after a failed attempt: %v", err) } date2, proven2 := cov2.Tier2CopyDate() if !proven2 || date2 != older { t.Errorf("after a failed attempt the surface would name %q (proven=%v); want the last SUCCESS %q", date2, proven2, older) } } // TestR102_Tier2UnitRestoreDoesNotWriteTheMirror — a restore reads its source. If the Tier-2 copy // were mutated, the second drive would stop being a way back the moment it was used once. func TestR102_Tier2UnitRestoreDoesNotWriteTheMirror(t *testing.T) { f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1)) before := fingerprintTree(t, f.destBase) if _, err := f.m.RestoreTier2Unit("app"); err != nil { t.Fatalf("restore: %v", err) } if after := fingerprintTree(t, f.destBase); after != before { t.Error("the Tier-2 copy was written to by a restore that only reads it") } }