0f9b796615
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable).
370 lines
16 KiB
Go
370 lines
16 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"io"
|
|
"log"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// R-102 Group A — the recovery-unit restore can be pointed at a unit ANYWHERE.
|
|
//
|
|
// The defect: every reader of a recovery unit could only name a path under `backups/primary/`
|
|
// (appbackup/paths.go joined it literally), while Tier-2 mirrored each app's whole unit to
|
|
// `<dest>/backups/secondary/<app>/recovery-unit/` on every run. So the mirror was written nightly for
|
|
// months and read by nothing — and it was unreadable in exactly the failure Tier-2 exists for, where
|
|
// the primary drive and its unit are gone (07-backup-architecture §6.3, §7.2).
|
|
//
|
|
// Every test below asserts WHICH DIRECTORY was read, not merely that a restore succeeded. A test that
|
|
// only checked `err == nil` would have passed against the pre-fix code, because the pre-fix code read
|
|
// a real unit — just never the one that survives.
|
|
|
|
// r102Unit captures a REAL recovery unit for "app" onto its own fresh drive, with an identifying
|
|
// SUBDOMAIN, a portable DB password and a portable data key, plus optional volume tars and a DB dump.
|
|
//
|
|
// The unit is produced by the production CaptureRecoveryUnit, not hand-written: the capture side and
|
|
// the restore side must meet at real bytes, so a change to the on-disk shape cannot pass by having a
|
|
// test agree with itself (the reason captureFixtureUnit is built the same way).
|
|
func r102Unit(t *testing.T, marker string, volTars []string, dbDump string) (drive, unitDir string) {
|
|
t.Helper()
|
|
tmp := t.TempDir()
|
|
drive = filepath.Join(tmp, "drive")
|
|
return drive, r102UnitOnDrive(t, drive, marker, volTars, dbDump)
|
|
}
|
|
|
|
// r102UnitOnDrive is r102Unit against a drive the caller already owns — the Tier-2 fixture needs the
|
|
// unit to sit on the SAME drive the Tier-2 run will mirror FROM.
|
|
func r102UnitOnDrive(t *testing.T, drive, marker string, volTars []string, dbDump string) (unitDir string) {
|
|
t.Helper()
|
|
tmp := t.TempDir()
|
|
stackDir := filepath.Join(tmp, "stack")
|
|
if err := os.MkdirAll(stackDir, 0o755); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"),
|
|
"services:\n app:\n image: example/app:1\n db:\n image: postgres:16\n")
|
|
mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: App\n")
|
|
mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: "+marker+"\n")
|
|
|
|
info := RecoveryInfo{
|
|
StackDir: stackDir,
|
|
DisplayName: "App",
|
|
ImagePins: []string{"example/app:1"},
|
|
NonSecretEnv: map[string]string{"SUBDOMAIN": marker},
|
|
SecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"},
|
|
DataKeyEnvVars: []string{"SECRET_KEY"},
|
|
PortableSecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"},
|
|
PortableSecrets: map[string]string{"DB_PASSWORD": "pw-" + marker, "SECRET_KEY": "key-" + marker},
|
|
}
|
|
unitDir = RecoveryUnitPath(drive, "app")
|
|
// The dumps are written BEFORE the capture so the manifest enumerates them — a manifest that
|
|
// lists nothing is the R-353 "the backup held only settings" shape, which is a different case.
|
|
for _, v := range volTars {
|
|
mustWrite(t, filepath.Join(UnitVolumeDumpDir(unitDir), v), "tar:"+v+":"+marker)
|
|
}
|
|
if dbDump != "" {
|
|
mustWrite(t, filepath.Join(UnitDBDumpDir(unitDir), "app-postgres.sql"), dbDump)
|
|
}
|
|
|
|
m := &Manager{
|
|
logger: log.New(io.Discard, "", 0),
|
|
systemDataPath: filepath.Join(tmp, "system"),
|
|
stackProvider: &fakeRecoveryProvider{info: info, hdd: drive},
|
|
version: "vtest",
|
|
}
|
|
if err := m.CaptureRecoveryUnit("app"); err != nil {
|
|
t.Fatalf("capture fixture %q: %v", marker, err)
|
|
}
|
|
return unitDir
|
|
}
|
|
|
|
// r102Manager builds a restore-side Manager whose LIVE drive is liveDrive, with the Docker and DB
|
|
// seams injected so no daemon is touched. The returned slices record the directories each leg read.
|
|
func r102Manager(t *testing.T, liveDrive string) (m *Manager, fake *fakeRecoveryProvider, volDirs, dbPaths *[]string) {
|
|
t.Helper()
|
|
fake = &fakeRecoveryProvider{hdd: liveDrive, running: true}
|
|
m = &Manager{
|
|
logger: log.New(io.Discard, "", 0),
|
|
systemDataPath: filepath.Join(liveDrive, "..", "sys"),
|
|
stackProvider: fake,
|
|
}
|
|
var vd, dp []string
|
|
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
|
|
vd = append(vd, dumpDir)
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
return 0, nil
|
|
}
|
|
n := 0
|
|
for _, e := range entries {
|
|
if strings.HasSuffix(e.Name(), ".tar") {
|
|
n++
|
|
}
|
|
}
|
|
return n, nil
|
|
}
|
|
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
|
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
|
}
|
|
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
|
dp = append(dp, p)
|
|
return nil
|
|
}
|
|
return m, fake, &vd, &dp
|
|
}
|
|
|
|
// A2 — TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir.
|
|
//
|
|
// Two complete units exist on disk with DIFFERENT contents. The restore is pointed at the second one
|
|
// and must read every leg — env, secrets, volume tars, DB dump — out of THAT directory. The first
|
|
// unit is the app's own primary unit and is present and perfectly readable, which is what makes the
|
|
// assertion mean something: a restore that ignored unitDir would still succeed, and would silently
|
|
// return the wrong copy's data.
|
|
func TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir(t *testing.T) {
|
|
primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
|
|
_, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar", "vol_b.tar"}, pgDump(2))
|
|
|
|
m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive)
|
|
|
|
res, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit)
|
|
if err != nil {
|
|
t.Fatalf("restore from the named unit: %v", err)
|
|
}
|
|
if fake.gotEnv == nil {
|
|
t.Fatal("recreate was never called — the restore did not reach the redeploy")
|
|
}
|
|
if got := fake.gotEnv["SUBDOMAIN"]; got != "mirror" {
|
|
t.Errorf("config came from the WRONG unit: SUBDOMAIN=%q, want %q", got, "mirror")
|
|
}
|
|
if got := fake.gotEnv["SECRET_KEY"]; got != "key-mirror" {
|
|
t.Errorf("data key came from the WRONG unit: %q", got)
|
|
}
|
|
if got := fake.gotEnv["DB_PASSWORD"]; got != "pw-mirror" {
|
|
t.Errorf("DB password came from the WRONG unit: %q", got)
|
|
}
|
|
if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(mirrorUnit) {
|
|
t.Errorf("volume tars read from %v, want %q", *volDirs, UnitVolumeDumpDir(mirrorUnit))
|
|
}
|
|
if res.VolumesReplayed != 2 {
|
|
t.Errorf("VolumesReplayed=%d, want 2 (the mirror holds two tars; the primary holds one)", res.VolumesReplayed)
|
|
}
|
|
if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(mirrorUnit) {
|
|
t.Errorf("DB dump read from %v, want a file under %q", *dbPaths, UnitDBDumpDir(mirrorUnit))
|
|
}
|
|
// The negative control: nothing was read out of the primary unit.
|
|
for _, d := range *volDirs {
|
|
if strings.HasPrefix(d, primaryUnit) {
|
|
t.Errorf("a leg was read from the PRIMARY unit %q despite a mirror being named", d)
|
|
}
|
|
}
|
|
}
|
|
|
|
// A3 — TestR102_PrimaryPathIsUnchanged. The one-argument wrapper must still resolve to the app's own
|
|
// drive, `backups/primary/<stack>`. Asserted on the DIRECTORIES the legs actually read, not on a
|
|
// path expression re-derived in the test.
|
|
func TestR102_PrimaryPathIsUnchanged(t *testing.T) {
|
|
primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
|
|
m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive)
|
|
|
|
if _, err := m.RestoreFromRecoveryUnit("app"); err != nil {
|
|
t.Fatalf("primary restore: %v", err)
|
|
}
|
|
if got := fake.gotEnv["SUBDOMAIN"]; got != "primary" {
|
|
t.Errorf("SUBDOMAIN=%q, want the primary unit's %q", got, "primary")
|
|
}
|
|
if !strings.Contains(primaryUnit, filepath.Join("backups", "primary", "app")) {
|
|
t.Fatalf("fixture is not where the primary unit belongs: %q", primaryUnit)
|
|
}
|
|
if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(primaryUnit) {
|
|
t.Errorf("volume dir = %v, want %q", *volDirs, UnitVolumeDumpDir(primaryUnit))
|
|
}
|
|
if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(primaryUnit) {
|
|
t.Errorf("db dump dir = %v, want %q", *dbPaths, UnitDBDumpDir(primaryUnit))
|
|
}
|
|
}
|
|
|
|
// A4 — TestR102_LiveDestinationIsUnchanged. THE SOURCE MOVES; THE DESTINATION DOES NOT.
|
|
//
|
|
// The restore reads a mirror that lives on a different drive entirely, and must still write the app
|
|
// back to its own live namespace: the definition through RecreateStackDefinitionFromUnit (whose
|
|
// compose dir must be the MIRROR's, since that is the definition being restored), and the data into
|
|
// the LIVE Docker volumes and the LIVE database container. A restore that also relocated the app's
|
|
// data would be a migration, not a restore.
|
|
//
|
|
// The observable for "the destination did not move" is that nothing under the mirror's own drive was
|
|
// written to: the mirror tree is byte-identical before and after.
|
|
func TestR102_LiveDestinationIsUnchanged(t *testing.T) {
|
|
primaryDrive, _ := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
|
|
mirrorDrive, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2))
|
|
|
|
before := fingerprintTree(t, mirrorDrive)
|
|
|
|
m, fake, _, _ := r102Manager(t, primaryDrive)
|
|
var recreateComposeDir string
|
|
m.stackProvider = &r102RecordingProvider{fakeRecoveryProvider: fake, composeDirOut: &recreateComposeDir}
|
|
|
|
if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil {
|
|
t.Fatalf("restore: %v", err)
|
|
}
|
|
if recreateComposeDir != UnitComposeDir(mirrorUnit) {
|
|
t.Errorf("recreate read compose from %q, want the mirror's %q", recreateComposeDir, UnitComposeDir(mirrorUnit))
|
|
}
|
|
if after := fingerprintTree(t, mirrorDrive); after != before {
|
|
t.Error("the SOURCE mirror was written to — a restore reads its source and never writes it")
|
|
}
|
|
// The live drive is where the app lives, and the restore must not have relocated it: the app's
|
|
// own drive path is still the one the manager resolves for it.
|
|
if got := m.GetAppDrivePath("app"); got != primaryDrive {
|
|
t.Errorf("the app's live drive moved to %q, want %q", got, primaryDrive)
|
|
}
|
|
}
|
|
|
|
// r102RecordingProvider captures the compose directory RecreateStackDefinitionFromUnit is handed —
|
|
// the one place the restore names a source directory to the guest side.
|
|
type r102RecordingProvider struct {
|
|
*fakeRecoveryProvider
|
|
composeDirOut *string
|
|
}
|
|
|
|
func (f *r102RecordingProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error {
|
|
*f.composeDirOut = composeDir
|
|
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv)
|
|
}
|
|
|
|
// A5 — TestR102_MutationOrderIsPreserved. R-47's ordering is pinned and R-102 must not have moved it:
|
|
// stop → volumes → recreate → DB-only start → replay → full start. The DB-only phase exists because
|
|
// starting the whole stack first let the application rebuild schema underneath the replay (H4,
|
|
// DIAG-immich-restore-round2-2026-07-19).
|
|
//
|
|
// The state AT REPLAY TIME is asserted, not just the call sequence — H4 looked correctly ordered and
|
|
// the app was up.
|
|
//
|
|
// Red-proof (recorded in REPORT.md): swap the volume replay and the recreate step in
|
|
// RestoreFromRecoveryUnitAt → this test fails on the sequence assertion.
|
|
func TestR102_MutationOrderIsPreserved(t *testing.T) {
|
|
primaryDrive, _ := r102Unit(t, "primary", nil, pgDump(1))
|
|
_, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2))
|
|
|
|
m, fake, _, _ := r102Manager(t, primaryDrive)
|
|
|
|
var volumesDoneAtRecreate, fullUpAtReplay bool
|
|
var volumesReplayed bool
|
|
m.volumeReplayFrom = func(_, _ string) (int, error) {
|
|
volumesReplayed = true
|
|
if fake.gotEnv != nil {
|
|
t.Error("volumes were replayed AFTER the definition was recreated")
|
|
}
|
|
return 1, nil
|
|
}
|
|
prov := &r102OrderProvider{fakeRecoveryProvider: fake, volumesReplayed: &volumesReplayed, seen: &volumesDoneAtRecreate}
|
|
m.stackProvider = prov
|
|
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
|
|
fullUpAtReplay = fake.fullStarted
|
|
return nil
|
|
}
|
|
|
|
if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil {
|
|
t.Fatalf("restore: %v", err)
|
|
}
|
|
if got := strings.Join(fake.calls, ","); got != "stop,recreate,startsvc:db,start" {
|
|
t.Fatalf("sequence = %q, want stop → recreate → db-only start → (replay) → full start", got)
|
|
}
|
|
if !volumesDoneAtRecreate {
|
|
t.Error("the volume replay had not run when the definition was recreated")
|
|
}
|
|
if fullUpAtReplay {
|
|
t.Error("the FULL stack was already up when the DB replay fired — this is H4 exactly (R-47)")
|
|
}
|
|
}
|
|
|
|
// r102OrderProvider records whether the volume replay had already happened by the time the definition
|
|
// was recreated — the ordering fact the call log alone cannot carry.
|
|
type r102OrderProvider struct {
|
|
*fakeRecoveryProvider
|
|
volumesReplayed *bool
|
|
seen *bool
|
|
}
|
|
|
|
func (f *r102OrderProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error {
|
|
*f.seen = *f.volumesReplayed
|
|
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv)
|
|
}
|
|
|
|
// A6 — TestR102_MissingManifestInMirrorFailsClosed. A DIRECTORY IS NOT A PACKAGE.
|
|
//
|
|
// `<dest>/backups/secondary/<app>/recovery-unit/` can exist and be useless: a copy interrupted
|
|
// mid-run, or a tree whose manifest.json never landed. The Tier-2 unit restore must refuse it and
|
|
// must leave the app running — a restore armed over an unopenable unit would stop the app, replay
|
|
// nothing, and rewrite its definition from an empty capture. Same lesson as R-358 one tier over.
|
|
//
|
|
// Placed with Group A because it is the fail-closed half of the parameterised unit; the route it
|
|
// exercises is Part 1.3's.
|
|
func TestR102_MissingManifestInMirrorFailsClosed(t *testing.T) {
|
|
for _, tc := range []struct {
|
|
name string
|
|
setup func(t *testing.T, unitDir string)
|
|
}{
|
|
{"manifest absent", func(t *testing.T, unitDir string) {
|
|
if err := os.Remove(UnitManifestFile(unitDir)); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
}},
|
|
{"manifest unparseable", func(t *testing.T, unitDir string) {
|
|
mustWrite(t, UnitManifestFile(unitDir), "{ this is not json")
|
|
}},
|
|
{"recovery-unit absent entirely", func(t *testing.T, unitDir string) {
|
|
if err := os.RemoveAll(unitDir); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
}},
|
|
} {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
|
tc.setup(t, tier2UnitDir(f.destBase))
|
|
|
|
cov, covErr := f.m.Tier2RestoreCoverage("app")
|
|
if covErr != nil {
|
|
t.Fatalf("coverage: %v", covErr)
|
|
}
|
|
if cov.CanRestoreUnit() {
|
|
t.Error("CanRestoreUnit() said yes over an unopenable mirror — the surface would offer the action")
|
|
}
|
|
_, err := f.m.RestoreTier2Unit("app")
|
|
if err == nil {
|
|
t.Fatal("the restore ran over an unopenable mirror")
|
|
}
|
|
if !strings.Contains(err.Error(), ErrTier2NoUnitInCopy.Error()) {
|
|
t.Errorf("refusal = %v, want ErrTier2NoUnitInCopy", err)
|
|
}
|
|
if f.fake.stopped {
|
|
t.Error("the app was STOPPED despite the refusal — the outage this gate exists to avoid")
|
|
}
|
|
if f.fake.gotEnv != nil {
|
|
t.Error("the app's definition was rewritten despite the refusal")
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestR102_UnopenableMirrorIsStillDisclosedAsUnread is the other side of A6, and the reason
|
|
// UnitRestorable is a second field rather than a widening of HasUnit: a half-copied mirror cannot be
|
|
// restored, but it IS captured data the file restore is not looking at, so the disclosure must stand.
|
|
func TestR102_UnopenableMirrorIsStillDisclosedAsUnread(t *testing.T) {
|
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
|
mustWrite(t, UnitManifestFile(tier2UnitDir(f.destBase)), "{ not json")
|
|
|
|
cov, err := f.m.Tier2RestoreCoverage("app")
|
|
if err != nil {
|
|
t.Fatalf("coverage: %v", err)
|
|
}
|
|
if !cov.HasUnit {
|
|
t.Error("HasUnit went false for a mirror that exists — the file restore would stop disclosing unread data")
|
|
}
|
|
if cov.UnitRestorable {
|
|
t.Error("UnitRestorable stayed true over an unparseable manifest")
|
|
}
|
|
}
|