Files
felhom-controller/controller/internal/backup/r102_tier2_unit_test.go
T
admin 0f9b796615 R-102: the recovery unit on the second drive becomes a way back
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on
every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the
primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any
action in the product (07-backup-architecture 6.3, 7.2).

Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is
the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the
DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all
unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the
fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched.
reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now
named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second
one invented, which is what lets the acceptance test assert the volume leg's source directory.

Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a
parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken
inside RestoreFromRecoveryUnitAt, not beside it.

Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it
still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which
refused 40 running apps for months.

Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with
the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'.
Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the
primary -> fails with the mirror never reaching the redeploy, and with permission denied once the
primary tree is unreadable).
2026-08-31 11:30:34 +02:00

299 lines
12 KiB
Go

package backup
import (
"context"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// R-102 Group B — the Tier-2 route: the mirror on the SECOND DRIVE becomes a way back.
//
// The mirror in every test here is written by the PRODUCTION Tier-2 run (RunTier2 with copyTree
// standing in for rsync, the seam tier2_v2_test.go already uses), not by hand. That matters: the
// claim is "the copy Tier-2 actually writes is the copy this restore reads", and a hand-built
// fixture could only prove that the restore agrees with the test's idea of the layout.
// r102T2 is a fully-wired Tier-2 fixture: a live drive carrying the app's PRIMARY recovery unit, a
// second drive carrying the mirror Tier-2 just wrote, and a restore-side Manager with the Docker and
// DB seams injected so no daemon is touched.
type r102T2 struct {
m *Manager
fake *fakeRecoveryProvider
liveDrive string
destDrive string
destBase string
primaryUnit string
volDirs *[]string
dbPaths *[]string
}
// r102Tier2Fixture captures a unit on the live drive, runs the REAL RunTier2 to mirror it onto the
// second drive, then returns a Manager ready to restore. The mirror's contents differ from the
// primary's afterwards when the caller mutates one of them — B1 and B2 both depend on being able to
// tell the two copies apart.
func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 {
t.Helper()
tmp := t.TempDir()
live := filepath.Join(tmp, "usb")
dest := filepath.Join(tmp, "flash")
sys := filepath.Join(tmp, "sys")
primaryUnit := r102UnitOnDrive(t, live, "primary", volTars, dbDump)
sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0))
if err != nil {
t.Fatal(err)
}
if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil {
t.Fatal(err)
}
fake := &fakeRecoveryProvider{hdd: live, running: true}
cfg := &config.Config{}
cfg.Paths.SystemDataPath = sys
cfg.Paths.DataDir = filepath.Join(tmp, "data")
m := NewManager(cfg, sett, log.New(io.Discard, "", 0))
m.stackProvider = fake
m.systemDataPath = sys
m.tier2Mirror = copyTree
m.tier2SSDFits = func(string, int64) bool { return true }
m.samePhysicalDevice = oneDrivePerSubtree
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2 (the code that writes the mirror this task reads): %v", err)
}
destBase := filepath.Join(dest, "backups", "secondary", "app")
if _, sErr := os.Stat(UnitManifestFile(tier2UnitDir(destBase))); sErr != nil {
t.Fatalf("Tier-2 did not mirror the unit — the fixture proves nothing: %v", sErr)
}
var vd, dp []string
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
vd = append(vd, dumpDir)
entries, rErr := os.ReadDir(dumpDir)
if rErr != nil {
return 0, nil
}
n := 0
for _, e := range entries {
if strings.HasSuffix(e.Name(), ".tar") {
n++
}
}
return n, nil
}
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
}
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
dp = append(dp, p)
return nil
}
return &r102T2{m: m, fake: fake, liveDrive: live, destDrive: dest, destBase: destBase,
primaryUnit: primaryUnit, volDirs: &vd, dbPaths: &dp}
}
// markMirror rewrites the MIRROR's app.yaml so the two copies are distinguishable. It edits the copy,
// never the primary, so a restore that read the primary would return the pre-edit value.
func (f *r102T2) markMirror(t *testing.T, marker string) {
t.Helper()
p := filepath.Join(UnitComposeDir(tier2UnitDir(f.destBase)), "app.yaml")
b, err := os.ReadFile(p)
if err != nil {
t.Fatal(err)
}
out := strings.Replace(string(b), "SUBDOMAIN: primary", "SUBDOMAIN: "+marker, 1)
if out == string(b) {
t.Fatalf("mirror app.yaml did not carry the expected marker; contents:\n%s", b)
}
mustWrite(t, p, out)
}
// B1 — TestR102_Tier2UnitRestoreReadsTheSecondaryMirror.
func TestR102_Tier2UnitRestoreReadsTheSecondaryMirror(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
f.markMirror(t, "from-the-mirror")
res, err := f.m.RestoreTier2Unit("app")
if err != nil {
t.Fatalf("Tier-2 unit restore: %v", err)
}
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "from-the-mirror" {
t.Errorf("config came from the PRIMARY unit: SUBDOMAIN=%q", got)
}
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
t.Errorf("volume tars read from %v, want the mirror's %q", *f.volDirs, wantVol)
}
wantDB := UnitDBDumpDir(tier2UnitDir(f.destBase))
if len(*f.dbPaths) != 1 || filepath.Dir((*f.dbPaths)[0]) != wantDB {
t.Errorf("DB dump read from %v, want a file under the mirror's %q", *f.dbPaths, wantDB)
}
if res.VolumesReplayed != 1 || res.DBsReplayed != 1 {
t.Errorf("nothing came back: volumes=%d dbs=%d", res.VolumesReplayed, res.DBsReplayed)
}
if res.ManifestVolumes != 1 || res.ManifestDBs != 1 {
t.Errorf("the mirror's manifest did not enumerate its dumps: %+v", res)
}
}
// B2 — TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT.
//
// THE ACCEPTANCE TEST. Tier-2 exists for the loss of the primary drive, and in that failure the
// primary recovery unit is GONE. A test that leaves the primary in place proves the code compiles,
// not that the mirror is a route (07-backup-architecture §7.2).
//
// The primary unit is not merely emptied — the whole `backups/` tree on the live drive is removed and
// then made unreadable, so any code that tried to fall back to it would error rather than silently
// find nothing.
//
// Red-proof (recorded in REPORT.md): point RestoreTier2Unit back at the primary unit path → this test
// fails with "no readable recovery unit ... falling back to volume-only restore" behaviour and the
// mirror's marker never reaching the redeploy.
func TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1))
f.markMirror(t, "only-the-mirror-survives")
// The primary drive's whole backup tree is destroyed, then the parent is made unreadable so a
// fallback read cannot quietly return "nothing here".
primaryBackups := filepath.Join(f.liveDrive, "backups")
if err := os.RemoveAll(primaryBackups); err != nil {
t.Fatal(err)
}
if err := os.MkdirAll(primaryBackups, 0o000); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = os.Chmod(primaryBackups, 0o755) })
if _, err := os.Stat(f.primaryUnit); err == nil {
t.Fatalf("the primary unit is still readable at %q — the test would prove nothing", f.primaryUnit)
}
res, err := f.m.RestoreTier2Unit("app")
if err != nil {
t.Fatalf("the restore must succeed from the mirror ALONE — this is the whole point of R-102: %v", err)
}
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "only-the-mirror-survives" {
t.Errorf("SUBDOMAIN=%q — the mirror was not the source", got)
}
if got := f.fake.gotEnv["SECRET_KEY"]; got != "key-primary" {
t.Errorf("the data-encrypting key did not come back from the mirrored unit: %q", got)
}
if res.VolumesReplayed != 2 {
t.Errorf("VolumesReplayed=%d, want 2 from the mirror", res.VolumesReplayed)
}
if res.DBsReplayed != 1 {
t.Errorf("DBsReplayed=%d, want 1 from the mirror", res.DBsReplayed)
}
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
t.Errorf("volume source = %v, want %q", *f.volDirs, wantVol)
}
}
// B3 — TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag. Every restore in this manager shares one
// running flag. The Tier-2 unit route must be inside it, not beside it (R-351b).
func TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
// A backup/restore is already in flight.
if err := f.m.acquireRunning(); err != nil {
t.Fatal(err)
}
_, err := f.m.RestoreTier2Unit("app")
if err == nil {
t.Fatal("a second restore started while one was in flight")
}
if !strings.Contains(err.Error(), "already in progress") {
t.Errorf("refusal = %v, want the single-writer refusal", err)
}
if f.fake.stopped {
t.Error("the app was stopped by a restore that should never have started")
}
f.m.releaseRunning()
// And the flag is RELEASED afterwards, or the next restore would be refused forever.
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
t.Fatalf("restore after the flag cleared: %v", err)
}
if err := f.m.acquireRunning(); err != nil {
t.Errorf("the running flag was not released after the Tier-2 unit restore: %v", err)
}
f.m.releaseRunning()
}
// B4 — TestR102_NoTier2CopyIsAnHonestRefusal. No recorded copy at all: refuse with the reason the
// file restore already uses, and take no outage.
func TestR102_NoTier2CopyIsAnHonestRefusal(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
// Forget the recorded copy entirely — the "this app has never had a Tier-2 run" shape.
if err := f.m.settings.SetCrossDriveConfig("app", &settings.CrossDriveBackup{}); err != nil {
t.Fatal(err)
}
_, err := f.m.RestoreTier2Unit("app")
if err == nil {
t.Fatal("expected a refusal with no recorded copy")
}
if !strings.Contains(err.Error(), errNoTier2Copy.Error()) {
t.Errorf("refusal = %v, want errNoTier2Copy", err)
}
if f.fake.stopped {
t.Error("the app was stopped despite the refusal")
}
}
// B5 — TestR102_CopyAgeIsCarriedToTheSurface. The action overwrites live data with a copy of a
// certain age, so the age has to reach the surface. R-101 governs WHICH date: the success anchor,
// never the attempt clock, and Tier2CopyDate says which one it returned.
func TestR102_CopyAgeIsCarriedToTheSurface(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
cov, err := f.m.Tier2RestoreCoverage("app")
if err != nil {
t.Fatalf("coverage: %v", err)
}
if cov.CopyLastSuccess == "" {
t.Fatal("the successful Tier-2 run left no LastSuccess for the surface to name")
}
date, proven := cov.Tier2CopyDate()
if !proven || date != cov.CopyLastSuccess {
t.Errorf("Tier2CopyDate() = (%q,%v), want the success anchor %q", date, proven, cov.CopyLastSuccess)
}
// R-101's other half: a later FAILED attempt must not become the date shown. LastRun advances,
// LastSuccess does not, and the surface must keep naming the copy that actually exists.
older := cov.CopyLastSuccess
if err := f.m.settings.UpdateCrossDriveStatus("app", func(c *settings.CrossDriveBackup) {
c.LastRun = "2099-01-01T00:00:00Z"
c.LastStatus = "error"
}); err != nil {
t.Fatal(err)
}
cov2, err := f.m.Tier2RestoreCoverage("app")
if err != nil {
t.Fatalf("coverage after a failed attempt: %v", err)
}
date2, proven2 := cov2.Tier2CopyDate()
if !proven2 || date2 != older {
t.Errorf("after a failed attempt the surface would name %q (proven=%v); want the last SUCCESS %q", date2, proven2, older)
}
}
// TestR102_Tier2UnitRestoreDoesNotWriteTheMirror — a restore reads its source. If the Tier-2 copy
// were mutated, the second drive would stop being a way back the moment it was used once.
func TestR102_Tier2UnitRestoreDoesNotWriteTheMirror(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
before := fingerprintTree(t, f.destBase)
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
t.Fatalf("restore: %v", err)
}
if after := fingerprintTree(t, f.destBase); after != before {
t.Error("the Tier-2 copy was written to by a restore that only reads it")
}
}