3c6b49b31c
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
303 lines
12 KiB
Go
303 lines
12 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"io"
|
|
"log"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
|
)
|
|
|
|
// R-102 Group B — the Tier-2 route: the mirror on the SECOND DRIVE becomes a way back.
|
|
//
|
|
// The mirror in every test here is written by the PRODUCTION Tier-2 run (RunTier2 with copyTree
|
|
// standing in for rsync, the seam tier2_v2_test.go already uses), not by hand. That matters: the
|
|
// claim is "the copy Tier-2 actually writes is the copy this restore reads", and a hand-built
|
|
// fixture could only prove that the restore agrees with the test's idea of the layout.
|
|
|
|
// r102T2 is a fully-wired Tier-2 fixture: a live drive carrying the app's PRIMARY recovery unit, a
|
|
// second drive carrying the mirror Tier-2 just wrote, and a restore-side Manager with the Docker and
|
|
// DB seams injected so no daemon is touched.
|
|
type r102T2 struct {
|
|
m *Manager
|
|
fake *fakeRecoveryProvider
|
|
liveDrive string
|
|
destDrive string
|
|
destBase string
|
|
primaryUnit string
|
|
volDirs *[]string
|
|
dbPaths *[]string
|
|
}
|
|
|
|
// r102Tier2Fixture captures a unit on the live drive, runs the REAL RunTier2 to mirror it onto the
|
|
// second drive, then returns a Manager ready to restore. The mirror's contents differ from the
|
|
// primary's afterwards when the caller mutates one of them — B1 and B2 both depend on being able to
|
|
// tell the two copies apart.
|
|
func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 {
|
|
t.Helper()
|
|
tmp := t.TempDir()
|
|
live := filepath.Join(tmp, "usb")
|
|
dest := filepath.Join(tmp, "flash")
|
|
sys := filepath.Join(tmp, "sys")
|
|
|
|
primaryUnit := r102UnitOnDrive(t, live, "primary", volTars, dbDump)
|
|
|
|
sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0))
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// A registered drive is a directory that EXISTS (R-668, v0.269.0: a missing one is no candidate).
|
|
if err := os.MkdirAll(dest, 0o755); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
fake := &fakeRecoveryProvider{hdd: live, running: true}
|
|
cfg := &config.Config{}
|
|
cfg.Paths.SystemDataPath = sys
|
|
cfg.Paths.DataDir = filepath.Join(tmp, "data")
|
|
m := NewManager(cfg, sett, log.New(io.Discard, "", 0))
|
|
m.stackProvider = fake
|
|
m.systemDataPath = sys
|
|
m.tier2Mirror = copyTree
|
|
m.tier2SSDFits = func(string, int64) bool { return true }
|
|
m.samePhysicalDevice = oneDrivePerSubtree
|
|
|
|
if err := m.RunTier2("app"); err != nil {
|
|
t.Fatalf("RunTier2 (the code that writes the mirror this task reads): %v", err)
|
|
}
|
|
destBase := filepath.Join(dest, "backups", "secondary", "app")
|
|
if _, sErr := os.Stat(UnitManifestFile(tier2UnitDir(destBase))); sErr != nil {
|
|
t.Fatalf("Tier-2 did not mirror the unit — the fixture proves nothing: %v", sErr)
|
|
}
|
|
|
|
var vd, dp []string
|
|
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
|
|
vd = append(vd, dumpDir)
|
|
entries, rErr := os.ReadDir(dumpDir)
|
|
if rErr != nil {
|
|
return 0, nil
|
|
}
|
|
n := 0
|
|
for _, e := range entries {
|
|
if strings.HasSuffix(e.Name(), ".tar") {
|
|
n++
|
|
}
|
|
}
|
|
return n, nil
|
|
}
|
|
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
|
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
|
}
|
|
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
|
dp = append(dp, p)
|
|
return nil
|
|
}
|
|
return &r102T2{m: m, fake: fake, liveDrive: live, destDrive: dest, destBase: destBase,
|
|
primaryUnit: primaryUnit, volDirs: &vd, dbPaths: &dp}
|
|
}
|
|
|
|
// markMirror rewrites the MIRROR's app.yaml so the two copies are distinguishable. It edits the copy,
|
|
// never the primary, so a restore that read the primary would return the pre-edit value.
|
|
func (f *r102T2) markMirror(t *testing.T, marker string) {
|
|
t.Helper()
|
|
p := filepath.Join(UnitComposeDir(tier2UnitDir(f.destBase)), "app.yaml")
|
|
b, err := os.ReadFile(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
out := strings.Replace(string(b), "SUBDOMAIN: primary", "SUBDOMAIN: "+marker, 1)
|
|
if out == string(b) {
|
|
t.Fatalf("mirror app.yaml did not carry the expected marker; contents:\n%s", b)
|
|
}
|
|
mustWrite(t, p, out)
|
|
}
|
|
|
|
// B1 — TestR102_Tier2UnitRestoreReadsTheSecondaryMirror.
|
|
func TestR102_Tier2UnitRestoreReadsTheSecondaryMirror(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
|
f.markMirror(t, "from-the-mirror")
|
|
|
|
res, err := f.m.RestoreTier2Unit("app")
|
|
if err != nil {
|
|
t.Fatalf("Tier-2 unit restore: %v", err)
|
|
}
|
|
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "from-the-mirror" {
|
|
t.Errorf("config came from the PRIMARY unit: SUBDOMAIN=%q", got)
|
|
}
|
|
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
|
|
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
|
|
t.Errorf("volume tars read from %v, want the mirror's %q", *f.volDirs, wantVol)
|
|
}
|
|
wantDB := UnitDBDumpDir(tier2UnitDir(f.destBase))
|
|
if len(*f.dbPaths) != 1 || filepath.Dir((*f.dbPaths)[0]) != wantDB {
|
|
t.Errorf("DB dump read from %v, want a file under the mirror's %q", *f.dbPaths, wantDB)
|
|
}
|
|
if res.VolumesReplayed != 1 || res.DBsReplayed != 1 {
|
|
t.Errorf("nothing came back: volumes=%d dbs=%d", res.VolumesReplayed, res.DBsReplayed)
|
|
}
|
|
if res.ManifestVolumes != 1 || res.ManifestDBs != 1 {
|
|
t.Errorf("the mirror's manifest did not enumerate its dumps: %+v", res)
|
|
}
|
|
}
|
|
|
|
// B2 — TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT.
|
|
//
|
|
// THE ACCEPTANCE TEST. Tier-2 exists for the loss of the primary drive, and in that failure the
|
|
// primary recovery unit is GONE. A test that leaves the primary in place proves the code compiles,
|
|
// not that the mirror is a route (07-backup-architecture §7.2).
|
|
//
|
|
// The primary unit is not merely emptied — the whole `backups/` tree on the live drive is removed and
|
|
// then made unreadable, so any code that tried to fall back to it would error rather than silently
|
|
// find nothing.
|
|
//
|
|
// Red-proof (recorded in REPORT.md): point RestoreTier2Unit back at the primary unit path → this test
|
|
// fails with "no readable recovery unit ... falling back to volume-only restore" behaviour and the
|
|
// mirror's marker never reaching the redeploy.
|
|
func TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1))
|
|
f.markMirror(t, "only-the-mirror-survives")
|
|
|
|
// The primary drive's whole backup tree is destroyed, then the parent is made unreadable so a
|
|
// fallback read cannot quietly return "nothing here".
|
|
primaryBackups := filepath.Join(f.liveDrive, "backups")
|
|
if err := os.RemoveAll(primaryBackups); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := os.MkdirAll(primaryBackups, 0o000); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
t.Cleanup(func() { _ = os.Chmod(primaryBackups, 0o755) })
|
|
if _, err := os.Stat(f.primaryUnit); err == nil {
|
|
t.Fatalf("the primary unit is still readable at %q — the test would prove nothing", f.primaryUnit)
|
|
}
|
|
|
|
res, err := f.m.RestoreTier2Unit("app")
|
|
if err != nil {
|
|
t.Fatalf("the restore must succeed from the mirror ALONE — this is the whole point of R-102: %v", err)
|
|
}
|
|
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "only-the-mirror-survives" {
|
|
t.Errorf("SUBDOMAIN=%q — the mirror was not the source", got)
|
|
}
|
|
if got := f.fake.gotEnv["SECRET_KEY"]; got != "key-primary" {
|
|
t.Errorf("the data-encrypting key did not come back from the mirrored unit: %q", got)
|
|
}
|
|
if res.VolumesReplayed != 2 {
|
|
t.Errorf("VolumesReplayed=%d, want 2 from the mirror", res.VolumesReplayed)
|
|
}
|
|
if res.DBsReplayed != 1 {
|
|
t.Errorf("DBsReplayed=%d, want 1 from the mirror", res.DBsReplayed)
|
|
}
|
|
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
|
|
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
|
|
t.Errorf("volume source = %v, want %q", *f.volDirs, wantVol)
|
|
}
|
|
}
|
|
|
|
// B3 — TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag. Every restore in this manager shares one
|
|
// running flag. The Tier-2 unit route must be inside it, not beside it (R-351b).
|
|
func TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
|
|
|
// A backup/restore is already in flight.
|
|
if err := f.m.acquireRunning(); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
_, err := f.m.RestoreTier2Unit("app")
|
|
if err == nil {
|
|
t.Fatal("a second restore started while one was in flight")
|
|
}
|
|
if !strings.Contains(err.Error(), "already in progress") {
|
|
t.Errorf("refusal = %v, want the single-writer refusal", err)
|
|
}
|
|
if f.fake.stopped {
|
|
t.Error("the app was stopped by a restore that should never have started")
|
|
}
|
|
f.m.releaseRunning()
|
|
|
|
// And the flag is RELEASED afterwards, or the next restore would be refused forever.
|
|
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
|
|
t.Fatalf("restore after the flag cleared: %v", err)
|
|
}
|
|
if err := f.m.acquireRunning(); err != nil {
|
|
t.Errorf("the running flag was not released after the Tier-2 unit restore: %v", err)
|
|
}
|
|
f.m.releaseRunning()
|
|
}
|
|
|
|
// B4 — TestR102_NoTier2CopyIsAnHonestRefusal. No recorded copy at all: refuse with the reason the
|
|
// file restore already uses, and take no outage.
|
|
func TestR102_NoTier2CopyIsAnHonestRefusal(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
|
// Forget the recorded copy entirely — the "this app has never had a Tier-2 run" shape.
|
|
if err := f.m.settings.SetCrossDriveConfig("app", &settings.CrossDriveBackup{}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
_, err := f.m.RestoreTier2Unit("app")
|
|
if err == nil {
|
|
t.Fatal("expected a refusal with no recorded copy")
|
|
}
|
|
if !strings.Contains(err.Error(), errNoTier2Copy.Error()) {
|
|
t.Errorf("refusal = %v, want errNoTier2Copy", err)
|
|
}
|
|
if f.fake.stopped {
|
|
t.Error("the app was stopped despite the refusal")
|
|
}
|
|
}
|
|
|
|
// B5 — TestR102_CopyAgeIsCarriedToTheSurface. The action overwrites live data with a copy of a
|
|
// certain age, so the age has to reach the surface. R-101 governs WHICH date: the success anchor,
|
|
// never the attempt clock, and Tier2CopyDate says which one it returned.
|
|
func TestR102_CopyAgeIsCarriedToTheSurface(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
|
|
|
cov, err := f.m.Tier2RestoreCoverage("app")
|
|
if err != nil {
|
|
t.Fatalf("coverage: %v", err)
|
|
}
|
|
if cov.CopyLastSuccess == "" {
|
|
t.Fatal("the successful Tier-2 run left no LastSuccess for the surface to name")
|
|
}
|
|
date, proven := cov.Tier2CopyDate()
|
|
if !proven || date != cov.CopyLastSuccess {
|
|
t.Errorf("Tier2CopyDate() = (%q,%v), want the success anchor %q", date, proven, cov.CopyLastSuccess)
|
|
}
|
|
|
|
// R-101's other half: a later FAILED attempt must not become the date shown. LastRun advances,
|
|
// LastSuccess does not, and the surface must keep naming the copy that actually exists.
|
|
older := cov.CopyLastSuccess
|
|
if err := f.m.settings.UpdateCrossDriveStatus("app", func(c *settings.CrossDriveBackup) {
|
|
c.LastRun = "2099-01-01T00:00:00Z"
|
|
c.LastStatus = "error"
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
cov2, err := f.m.Tier2RestoreCoverage("app")
|
|
if err != nil {
|
|
t.Fatalf("coverage after a failed attempt: %v", err)
|
|
}
|
|
date2, proven2 := cov2.Tier2CopyDate()
|
|
if !proven2 || date2 != older {
|
|
t.Errorf("after a failed attempt the surface would name %q (proven=%v); want the last SUCCESS %q", date2, proven2, older)
|
|
}
|
|
}
|
|
|
|
// TestR102_Tier2UnitRestoreDoesNotWriteTheMirror — a restore reads its source. If the Tier-2 copy
|
|
// were mutated, the second drive would stop being a way back the moment it was used once.
|
|
func TestR102_Tier2UnitRestoreDoesNotWriteTheMirror(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
|
before := fingerprintTree(t, f.destBase)
|
|
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
|
|
t.Fatalf("restore: %v", err)
|
|
}
|
|
if after := fingerprintTree(t, f.destBase); after != before {
|
|
t.Error("the Tier-2 copy was written to by a restore that only reads it")
|
|
}
|
|
}
|