R-102: the recovery unit on the second drive becomes a way back
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable).
This commit is contained in:
@@ -0,0 +1,298 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-102 Group B — the Tier-2 route: the mirror on the SECOND DRIVE becomes a way back.
|
||||
//
|
||||
// The mirror in every test here is written by the PRODUCTION Tier-2 run (RunTier2 with copyTree
|
||||
// standing in for rsync, the seam tier2_v2_test.go already uses), not by hand. That matters: the
|
||||
// claim is "the copy Tier-2 actually writes is the copy this restore reads", and a hand-built
|
||||
// fixture could only prove that the restore agrees with the test's idea of the layout.
|
||||
|
||||
// r102T2 is a fully-wired Tier-2 fixture: a live drive carrying the app's PRIMARY recovery unit, a
|
||||
// second drive carrying the mirror Tier-2 just wrote, and a restore-side Manager with the Docker and
|
||||
// DB seams injected so no daemon is touched.
|
||||
type r102T2 struct {
|
||||
m *Manager
|
||||
fake *fakeRecoveryProvider
|
||||
liveDrive string
|
||||
destDrive string
|
||||
destBase string
|
||||
primaryUnit string
|
||||
volDirs *[]string
|
||||
dbPaths *[]string
|
||||
}
|
||||
|
||||
// r102Tier2Fixture captures a unit on the live drive, runs the REAL RunTier2 to mirror it onto the
|
||||
// second drive, then returns a Manager ready to restore. The mirror's contents differ from the
|
||||
// primary's afterwards when the caller mutates one of them — B1 and B2 both depend on being able to
|
||||
// tell the two copies apart.
|
||||
func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 {
|
||||
t.Helper()
|
||||
tmp := t.TempDir()
|
||||
live := filepath.Join(tmp, "usb")
|
||||
dest := filepath.Join(tmp, "flash")
|
||||
sys := filepath.Join(tmp, "sys")
|
||||
|
||||
primaryUnit := r102UnitOnDrive(t, live, "primary", volTars, dbDump)
|
||||
|
||||
sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
fake := &fakeRecoveryProvider{hdd: live, running: true}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.SystemDataPath = sys
|
||||
cfg.Paths.DataDir = filepath.Join(tmp, "data")
|
||||
m := NewManager(cfg, sett, log.New(io.Discard, "", 0))
|
||||
m.stackProvider = fake
|
||||
m.systemDataPath = sys
|
||||
m.tier2Mirror = copyTree
|
||||
m.tier2SSDFits = func(string, int64) bool { return true }
|
||||
m.samePhysicalDevice = oneDrivePerSubtree
|
||||
|
||||
if err := m.RunTier2("app"); err != nil {
|
||||
t.Fatalf("RunTier2 (the code that writes the mirror this task reads): %v", err)
|
||||
}
|
||||
destBase := filepath.Join(dest, "backups", "secondary", "app")
|
||||
if _, sErr := os.Stat(UnitManifestFile(tier2UnitDir(destBase))); sErr != nil {
|
||||
t.Fatalf("Tier-2 did not mirror the unit — the fixture proves nothing: %v", sErr)
|
||||
}
|
||||
|
||||
var vd, dp []string
|
||||
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
|
||||
vd = append(vd, dumpDir)
|
||||
entries, rErr := os.ReadDir(dumpDir)
|
||||
if rErr != nil {
|
||||
return 0, nil
|
||||
}
|
||||
n := 0
|
||||
for _, e := range entries {
|
||||
if strings.HasSuffix(e.Name(), ".tar") {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
||||
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
||||
}
|
||||
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||
dp = append(dp, p)
|
||||
return nil
|
||||
}
|
||||
return &r102T2{m: m, fake: fake, liveDrive: live, destDrive: dest, destBase: destBase,
|
||||
primaryUnit: primaryUnit, volDirs: &vd, dbPaths: &dp}
|
||||
}
|
||||
|
||||
// markMirror rewrites the MIRROR's app.yaml so the two copies are distinguishable. It edits the copy,
|
||||
// never the primary, so a restore that read the primary would return the pre-edit value.
|
||||
func (f *r102T2) markMirror(t *testing.T, marker string) {
|
||||
t.Helper()
|
||||
p := filepath.Join(UnitComposeDir(tier2UnitDir(f.destBase)), "app.yaml")
|
||||
b, err := os.ReadFile(p)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
out := strings.Replace(string(b), "SUBDOMAIN: primary", "SUBDOMAIN: "+marker, 1)
|
||||
if out == string(b) {
|
||||
t.Fatalf("mirror app.yaml did not carry the expected marker; contents:\n%s", b)
|
||||
}
|
||||
mustWrite(t, p, out)
|
||||
}
|
||||
|
||||
// B1 — TestR102_Tier2UnitRestoreReadsTheSecondaryMirror.
|
||||
func TestR102_Tier2UnitRestoreReadsTheSecondaryMirror(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
||||
f.markMirror(t, "from-the-mirror")
|
||||
|
||||
res, err := f.m.RestoreTier2Unit("app")
|
||||
if err != nil {
|
||||
t.Fatalf("Tier-2 unit restore: %v", err)
|
||||
}
|
||||
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "from-the-mirror" {
|
||||
t.Errorf("config came from the PRIMARY unit: SUBDOMAIN=%q", got)
|
||||
}
|
||||
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
|
||||
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
|
||||
t.Errorf("volume tars read from %v, want the mirror's %q", *f.volDirs, wantVol)
|
||||
}
|
||||
wantDB := UnitDBDumpDir(tier2UnitDir(f.destBase))
|
||||
if len(*f.dbPaths) != 1 || filepath.Dir((*f.dbPaths)[0]) != wantDB {
|
||||
t.Errorf("DB dump read from %v, want a file under the mirror's %q", *f.dbPaths, wantDB)
|
||||
}
|
||||
if res.VolumesReplayed != 1 || res.DBsReplayed != 1 {
|
||||
t.Errorf("nothing came back: volumes=%d dbs=%d", res.VolumesReplayed, res.DBsReplayed)
|
||||
}
|
||||
if res.ManifestVolumes != 1 || res.ManifestDBs != 1 {
|
||||
t.Errorf("the mirror's manifest did not enumerate its dumps: %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// B2 — TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT.
|
||||
//
|
||||
// THE ACCEPTANCE TEST. Tier-2 exists for the loss of the primary drive, and in that failure the
|
||||
// primary recovery unit is GONE. A test that leaves the primary in place proves the code compiles,
|
||||
// not that the mirror is a route (07-backup-architecture §7.2).
|
||||
//
|
||||
// The primary unit is not merely emptied — the whole `backups/` tree on the live drive is removed and
|
||||
// then made unreadable, so any code that tried to fall back to it would error rather than silently
|
||||
// find nothing.
|
||||
//
|
||||
// Red-proof (recorded in REPORT.md): point RestoreTier2Unit back at the primary unit path → this test
|
||||
// fails with "no readable recovery unit ... falling back to volume-only restore" behaviour and the
|
||||
// mirror's marker never reaching the redeploy.
|
||||
func TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1))
|
||||
f.markMirror(t, "only-the-mirror-survives")
|
||||
|
||||
// The primary drive's whole backup tree is destroyed, then the parent is made unreadable so a
|
||||
// fallback read cannot quietly return "nothing here".
|
||||
primaryBackups := filepath.Join(f.liveDrive, "backups")
|
||||
if err := os.RemoveAll(primaryBackups); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.MkdirAll(primaryBackups, 0o000); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { _ = os.Chmod(primaryBackups, 0o755) })
|
||||
if _, err := os.Stat(f.primaryUnit); err == nil {
|
||||
t.Fatalf("the primary unit is still readable at %q — the test would prove nothing", f.primaryUnit)
|
||||
}
|
||||
|
||||
res, err := f.m.RestoreTier2Unit("app")
|
||||
if err != nil {
|
||||
t.Fatalf("the restore must succeed from the mirror ALONE — this is the whole point of R-102: %v", err)
|
||||
}
|
||||
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "only-the-mirror-survives" {
|
||||
t.Errorf("SUBDOMAIN=%q — the mirror was not the source", got)
|
||||
}
|
||||
if got := f.fake.gotEnv["SECRET_KEY"]; got != "key-primary" {
|
||||
t.Errorf("the data-encrypting key did not come back from the mirrored unit: %q", got)
|
||||
}
|
||||
if res.VolumesReplayed != 2 {
|
||||
t.Errorf("VolumesReplayed=%d, want 2 from the mirror", res.VolumesReplayed)
|
||||
}
|
||||
if res.DBsReplayed != 1 {
|
||||
t.Errorf("DBsReplayed=%d, want 1 from the mirror", res.DBsReplayed)
|
||||
}
|
||||
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
|
||||
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
|
||||
t.Errorf("volume source = %v, want %q", *f.volDirs, wantVol)
|
||||
}
|
||||
}
|
||||
|
||||
// B3 — TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag. Every restore in this manager shares one
|
||||
// running flag. The Tier-2 unit route must be inside it, not beside it (R-351b).
|
||||
func TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
||||
|
||||
// A backup/restore is already in flight.
|
||||
if err := f.m.acquireRunning(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, err := f.m.RestoreTier2Unit("app")
|
||||
if err == nil {
|
||||
t.Fatal("a second restore started while one was in flight")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "already in progress") {
|
||||
t.Errorf("refusal = %v, want the single-writer refusal", err)
|
||||
}
|
||||
if f.fake.stopped {
|
||||
t.Error("the app was stopped by a restore that should never have started")
|
||||
}
|
||||
f.m.releaseRunning()
|
||||
|
||||
// And the flag is RELEASED afterwards, or the next restore would be refused forever.
|
||||
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
|
||||
t.Fatalf("restore after the flag cleared: %v", err)
|
||||
}
|
||||
if err := f.m.acquireRunning(); err != nil {
|
||||
t.Errorf("the running flag was not released after the Tier-2 unit restore: %v", err)
|
||||
}
|
||||
f.m.releaseRunning()
|
||||
}
|
||||
|
||||
// B4 — TestR102_NoTier2CopyIsAnHonestRefusal. No recorded copy at all: refuse with the reason the
|
||||
// file restore already uses, and take no outage.
|
||||
func TestR102_NoTier2CopyIsAnHonestRefusal(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
||||
// Forget the recorded copy entirely — the "this app has never had a Tier-2 run" shape.
|
||||
if err := f.m.settings.SetCrossDriveConfig("app", &settings.CrossDriveBackup{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, err := f.m.RestoreTier2Unit("app")
|
||||
if err == nil {
|
||||
t.Fatal("expected a refusal with no recorded copy")
|
||||
}
|
||||
if !strings.Contains(err.Error(), errNoTier2Copy.Error()) {
|
||||
t.Errorf("refusal = %v, want errNoTier2Copy", err)
|
||||
}
|
||||
if f.fake.stopped {
|
||||
t.Error("the app was stopped despite the refusal")
|
||||
}
|
||||
}
|
||||
|
||||
// B5 — TestR102_CopyAgeIsCarriedToTheSurface. The action overwrites live data with a copy of a
|
||||
// certain age, so the age has to reach the surface. R-101 governs WHICH date: the success anchor,
|
||||
// never the attempt clock, and Tier2CopyDate says which one it returned.
|
||||
func TestR102_CopyAgeIsCarriedToTheSurface(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
||||
|
||||
cov, err := f.m.Tier2RestoreCoverage("app")
|
||||
if err != nil {
|
||||
t.Fatalf("coverage: %v", err)
|
||||
}
|
||||
if cov.CopyLastSuccess == "" {
|
||||
t.Fatal("the successful Tier-2 run left no LastSuccess for the surface to name")
|
||||
}
|
||||
date, proven := cov.Tier2CopyDate()
|
||||
if !proven || date != cov.CopyLastSuccess {
|
||||
t.Errorf("Tier2CopyDate() = (%q,%v), want the success anchor %q", date, proven, cov.CopyLastSuccess)
|
||||
}
|
||||
|
||||
// R-101's other half: a later FAILED attempt must not become the date shown. LastRun advances,
|
||||
// LastSuccess does not, and the surface must keep naming the copy that actually exists.
|
||||
older := cov.CopyLastSuccess
|
||||
if err := f.m.settings.UpdateCrossDriveStatus("app", func(c *settings.CrossDriveBackup) {
|
||||
c.LastRun = "2099-01-01T00:00:00Z"
|
||||
c.LastStatus = "error"
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cov2, err := f.m.Tier2RestoreCoverage("app")
|
||||
if err != nil {
|
||||
t.Fatalf("coverage after a failed attempt: %v", err)
|
||||
}
|
||||
date2, proven2 := cov2.Tier2CopyDate()
|
||||
if !proven2 || date2 != older {
|
||||
t.Errorf("after a failed attempt the surface would name %q (proven=%v); want the last SUCCESS %q", date2, proven2, older)
|
||||
}
|
||||
}
|
||||
|
||||
// TestR102_Tier2UnitRestoreDoesNotWriteTheMirror — a restore reads its source. If the Tier-2 copy
|
||||
// were mutated, the second drive would stop being a way back the moment it was used once.
|
||||
func TestR102_Tier2UnitRestoreDoesNotWriteTheMirror(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
||||
before := fingerprintTree(t, f.destBase)
|
||||
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
|
||||
t.Fatalf("restore: %v", err)
|
||||
}
|
||||
if after := fingerprintTree(t, f.destBase); after != before {
|
||||
t.Error("the Tier-2 copy was written to by a restore that only reads it")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user