Files
felhom-controller/controller/internal/backup/r102_unit_at_test.go
T
admin 0f9b796615 R-102: the recovery unit on the second drive becomes a way back
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on
every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the
primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any
action in the product (07-backup-architecture 6.3, 7.2).

Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is
the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the
DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all
unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the
fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched.
reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now
named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second
one invented, which is what lets the acceptance test assert the volume leg's source directory.

Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a
parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken
inside RestoreFromRecoveryUnitAt, not beside it.

Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it
still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which
refused 40 running apps for months.

Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with
the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'.
Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the
primary -> fails with the mirror never reaching the redeploy, and with permission denied once the
primary tree is unreadable).
2026-08-31 11:30:34 +02:00

370 lines
16 KiB
Go

package backup
import (
"context"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
)
// R-102 Group A — the recovery-unit restore can be pointed at a unit ANYWHERE.
//
// The defect: every reader of a recovery unit could only name a path under `backups/primary/`
// (appbackup/paths.go joined it literally), while Tier-2 mirrored each app's whole unit to
// `<dest>/backups/secondary/<app>/recovery-unit/` on every run. So the mirror was written nightly for
// months and read by nothing — and it was unreadable in exactly the failure Tier-2 exists for, where
// the primary drive and its unit are gone (07-backup-architecture §6.3, §7.2).
//
// Every test below asserts WHICH DIRECTORY was read, not merely that a restore succeeded. A test that
// only checked `err == nil` would have passed against the pre-fix code, because the pre-fix code read
// a real unit — just never the one that survives.
// r102Unit captures a REAL recovery unit for "app" onto its own fresh drive, with an identifying
// SUBDOMAIN, a portable DB password and a portable data key, plus optional volume tars and a DB dump.
//
// The unit is produced by the production CaptureRecoveryUnit, not hand-written: the capture side and
// the restore side must meet at real bytes, so a change to the on-disk shape cannot pass by having a
// test agree with itself (the reason captureFixtureUnit is built the same way).
func r102Unit(t *testing.T, marker string, volTars []string, dbDump string) (drive, unitDir string) {
t.Helper()
tmp := t.TempDir()
drive = filepath.Join(tmp, "drive")
return drive, r102UnitOnDrive(t, drive, marker, volTars, dbDump)
}
// r102UnitOnDrive is r102Unit against a drive the caller already owns — the Tier-2 fixture needs the
// unit to sit on the SAME drive the Tier-2 run will mirror FROM.
func r102UnitOnDrive(t *testing.T, drive, marker string, volTars []string, dbDump string) (unitDir string) {
t.Helper()
tmp := t.TempDir()
stackDir := filepath.Join(tmp, "stack")
if err := os.MkdirAll(stackDir, 0o755); err != nil {
t.Fatal(err)
}
mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"),
"services:\n app:\n image: example/app:1\n db:\n image: postgres:16\n")
mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: App\n")
mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: "+marker+"\n")
info := RecoveryInfo{
StackDir: stackDir,
DisplayName: "App",
ImagePins: []string{"example/app:1"},
NonSecretEnv: map[string]string{"SUBDOMAIN": marker},
SecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"},
DataKeyEnvVars: []string{"SECRET_KEY"},
PortableSecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"},
PortableSecrets: map[string]string{"DB_PASSWORD": "pw-" + marker, "SECRET_KEY": "key-" + marker},
}
unitDir = RecoveryUnitPath(drive, "app")
// The dumps are written BEFORE the capture so the manifest enumerates them — a manifest that
// lists nothing is the R-353 "the backup held only settings" shape, which is a different case.
for _, v := range volTars {
mustWrite(t, filepath.Join(UnitVolumeDumpDir(unitDir), v), "tar:"+v+":"+marker)
}
if dbDump != "" {
mustWrite(t, filepath.Join(UnitDBDumpDir(unitDir), "app-postgres.sql"), dbDump)
}
m := &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(tmp, "system"),
stackProvider: &fakeRecoveryProvider{info: info, hdd: drive},
version: "vtest",
}
if err := m.CaptureRecoveryUnit("app"); err != nil {
t.Fatalf("capture fixture %q: %v", marker, err)
}
return unitDir
}
// r102Manager builds a restore-side Manager whose LIVE drive is liveDrive, with the Docker and DB
// seams injected so no daemon is touched. The returned slices record the directories each leg read.
func r102Manager(t *testing.T, liveDrive string) (m *Manager, fake *fakeRecoveryProvider, volDirs, dbPaths *[]string) {
t.Helper()
fake = &fakeRecoveryProvider{hdd: liveDrive, running: true}
m = &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(liveDrive, "..", "sys"),
stackProvider: fake,
}
var vd, dp []string
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
vd = append(vd, dumpDir)
entries, err := os.ReadDir(dumpDir)
if err != nil {
return 0, nil
}
n := 0
for _, e := range entries {
if strings.HasSuffix(e.Name(), ".tar") {
n++
}
}
return n, nil
}
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
}
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
dp = append(dp, p)
return nil
}
return m, fake, &vd, &dp
}
// A2 — TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir.
//
// Two complete units exist on disk with DIFFERENT contents. The restore is pointed at the second one
// and must read every leg — env, secrets, volume tars, DB dump — out of THAT directory. The first
// unit is the app's own primary unit and is present and perfectly readable, which is what makes the
// assertion mean something: a restore that ignored unitDir would still succeed, and would silently
// return the wrong copy's data.
func TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir(t *testing.T) {
primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
_, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar", "vol_b.tar"}, pgDump(2))
m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive)
res, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit)
if err != nil {
t.Fatalf("restore from the named unit: %v", err)
}
if fake.gotEnv == nil {
t.Fatal("recreate was never called — the restore did not reach the redeploy")
}
if got := fake.gotEnv["SUBDOMAIN"]; got != "mirror" {
t.Errorf("config came from the WRONG unit: SUBDOMAIN=%q, want %q", got, "mirror")
}
if got := fake.gotEnv["SECRET_KEY"]; got != "key-mirror" {
t.Errorf("data key came from the WRONG unit: %q", got)
}
if got := fake.gotEnv["DB_PASSWORD"]; got != "pw-mirror" {
t.Errorf("DB password came from the WRONG unit: %q", got)
}
if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(mirrorUnit) {
t.Errorf("volume tars read from %v, want %q", *volDirs, UnitVolumeDumpDir(mirrorUnit))
}
if res.VolumesReplayed != 2 {
t.Errorf("VolumesReplayed=%d, want 2 (the mirror holds two tars; the primary holds one)", res.VolumesReplayed)
}
if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(mirrorUnit) {
t.Errorf("DB dump read from %v, want a file under %q", *dbPaths, UnitDBDumpDir(mirrorUnit))
}
// The negative control: nothing was read out of the primary unit.
for _, d := range *volDirs {
if strings.HasPrefix(d, primaryUnit) {
t.Errorf("a leg was read from the PRIMARY unit %q despite a mirror being named", d)
}
}
}
// A3 — TestR102_PrimaryPathIsUnchanged. The one-argument wrapper must still resolve to the app's own
// drive, `backups/primary/<stack>`. Asserted on the DIRECTORIES the legs actually read, not on a
// path expression re-derived in the test.
func TestR102_PrimaryPathIsUnchanged(t *testing.T) {
primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive)
if _, err := m.RestoreFromRecoveryUnit("app"); err != nil {
t.Fatalf("primary restore: %v", err)
}
if got := fake.gotEnv["SUBDOMAIN"]; got != "primary" {
t.Errorf("SUBDOMAIN=%q, want the primary unit's %q", got, "primary")
}
if !strings.Contains(primaryUnit, filepath.Join("backups", "primary", "app")) {
t.Fatalf("fixture is not where the primary unit belongs: %q", primaryUnit)
}
if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(primaryUnit) {
t.Errorf("volume dir = %v, want %q", *volDirs, UnitVolumeDumpDir(primaryUnit))
}
if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(primaryUnit) {
t.Errorf("db dump dir = %v, want %q", *dbPaths, UnitDBDumpDir(primaryUnit))
}
}
// A4 — TestR102_LiveDestinationIsUnchanged. THE SOURCE MOVES; THE DESTINATION DOES NOT.
//
// The restore reads a mirror that lives on a different drive entirely, and must still write the app
// back to its own live namespace: the definition through RecreateStackDefinitionFromUnit (whose
// compose dir must be the MIRROR's, since that is the definition being restored), and the data into
// the LIVE Docker volumes and the LIVE database container. A restore that also relocated the app's
// data would be a migration, not a restore.
//
// The observable for "the destination did not move" is that nothing under the mirror's own drive was
// written to: the mirror tree is byte-identical before and after.
func TestR102_LiveDestinationIsUnchanged(t *testing.T) {
primaryDrive, _ := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
mirrorDrive, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2))
before := fingerprintTree(t, mirrorDrive)
m, fake, _, _ := r102Manager(t, primaryDrive)
var recreateComposeDir string
m.stackProvider = &r102RecordingProvider{fakeRecoveryProvider: fake, composeDirOut: &recreateComposeDir}
if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil {
t.Fatalf("restore: %v", err)
}
if recreateComposeDir != UnitComposeDir(mirrorUnit) {
t.Errorf("recreate read compose from %q, want the mirror's %q", recreateComposeDir, UnitComposeDir(mirrorUnit))
}
if after := fingerprintTree(t, mirrorDrive); after != before {
t.Error("the SOURCE mirror was written to — a restore reads its source and never writes it")
}
// The live drive is where the app lives, and the restore must not have relocated it: the app's
// own drive path is still the one the manager resolves for it.
if got := m.GetAppDrivePath("app"); got != primaryDrive {
t.Errorf("the app's live drive moved to %q, want %q", got, primaryDrive)
}
}
// r102RecordingProvider captures the compose directory RecreateStackDefinitionFromUnit is handed —
// the one place the restore names a source directory to the guest side.
type r102RecordingProvider struct {
*fakeRecoveryProvider
composeDirOut *string
}
func (f *r102RecordingProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error {
*f.composeDirOut = composeDir
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv)
}
// A5 — TestR102_MutationOrderIsPreserved. R-47's ordering is pinned and R-102 must not have moved it:
// stop → volumes → recreate → DB-only start → replay → full start. The DB-only phase exists because
// starting the whole stack first let the application rebuild schema underneath the replay (H4,
// DIAG-immich-restore-round2-2026-07-19).
//
// The state AT REPLAY TIME is asserted, not just the call sequence — H4 looked correctly ordered and
// the app was up.
//
// Red-proof (recorded in REPORT.md): swap the volume replay and the recreate step in
// RestoreFromRecoveryUnitAt → this test fails on the sequence assertion.
func TestR102_MutationOrderIsPreserved(t *testing.T) {
primaryDrive, _ := r102Unit(t, "primary", nil, pgDump(1))
_, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2))
m, fake, _, _ := r102Manager(t, primaryDrive)
var volumesDoneAtRecreate, fullUpAtReplay bool
var volumesReplayed bool
m.volumeReplayFrom = func(_, _ string) (int, error) {
volumesReplayed = true
if fake.gotEnv != nil {
t.Error("volumes were replayed AFTER the definition was recreated")
}
return 1, nil
}
prov := &r102OrderProvider{fakeRecoveryProvider: fake, volumesReplayed: &volumesReplayed, seen: &volumesDoneAtRecreate}
m.stackProvider = prov
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
fullUpAtReplay = fake.fullStarted
return nil
}
if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil {
t.Fatalf("restore: %v", err)
}
if got := strings.Join(fake.calls, ","); got != "stop,recreate,startsvc:db,start" {
t.Fatalf("sequence = %q, want stop → recreate → db-only start → (replay) → full start", got)
}
if !volumesDoneAtRecreate {
t.Error("the volume replay had not run when the definition was recreated")
}
if fullUpAtReplay {
t.Error("the FULL stack was already up when the DB replay fired — this is H4 exactly (R-47)")
}
}
// r102OrderProvider records whether the volume replay had already happened by the time the definition
// was recreated — the ordering fact the call log alone cannot carry.
type r102OrderProvider struct {
*fakeRecoveryProvider
volumesReplayed *bool
seen *bool
}
func (f *r102OrderProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error {
*f.seen = *f.volumesReplayed
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv)
}
// A6 — TestR102_MissingManifestInMirrorFailsClosed. A DIRECTORY IS NOT A PACKAGE.
//
// `<dest>/backups/secondary/<app>/recovery-unit/` can exist and be useless: a copy interrupted
// mid-run, or a tree whose manifest.json never landed. The Tier-2 unit restore must refuse it and
// must leave the app running — a restore armed over an unopenable unit would stop the app, replay
// nothing, and rewrite its definition from an empty capture. Same lesson as R-358 one tier over.
//
// Placed with Group A because it is the fail-closed half of the parameterised unit; the route it
// exercises is Part 1.3's.
func TestR102_MissingManifestInMirrorFailsClosed(t *testing.T) {
for _, tc := range []struct {
name string
setup func(t *testing.T, unitDir string)
}{
{"manifest absent", func(t *testing.T, unitDir string) {
if err := os.Remove(UnitManifestFile(unitDir)); err != nil {
t.Fatal(err)
}
}},
{"manifest unparseable", func(t *testing.T, unitDir string) {
mustWrite(t, UnitManifestFile(unitDir), "{ this is not json")
}},
{"recovery-unit absent entirely", func(t *testing.T, unitDir string) {
if err := os.RemoveAll(unitDir); err != nil {
t.Fatal(err)
}
}},
} {
t.Run(tc.name, func(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
tc.setup(t, tier2UnitDir(f.destBase))
cov, covErr := f.m.Tier2RestoreCoverage("app")
if covErr != nil {
t.Fatalf("coverage: %v", covErr)
}
if cov.CanRestoreUnit() {
t.Error("CanRestoreUnit() said yes over an unopenable mirror — the surface would offer the action")
}
_, err := f.m.RestoreTier2Unit("app")
if err == nil {
t.Fatal("the restore ran over an unopenable mirror")
}
if !strings.Contains(err.Error(), ErrTier2NoUnitInCopy.Error()) {
t.Errorf("refusal = %v, want ErrTier2NoUnitInCopy", err)
}
if f.fake.stopped {
t.Error("the app was STOPPED despite the refusal — the outage this gate exists to avoid")
}
if f.fake.gotEnv != nil {
t.Error("the app's definition was rewritten despite the refusal")
}
})
}
}
// TestR102_UnopenableMirrorIsStillDisclosedAsUnread is the other side of A6, and the reason
// UnitRestorable is a second field rather than a widening of HasUnit: a half-copied mirror cannot be
// restored, but it IS captured data the file restore is not looking at, so the disclosure must stand.
func TestR102_UnopenableMirrorIsStillDisclosedAsUnread(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
mustWrite(t, UnitManifestFile(tier2UnitDir(f.destBase)), "{ not json")
cov, err := f.m.Tier2RestoreCoverage("app")
if err != nil {
t.Fatalf("coverage: %v", err)
}
if !cov.HasUnit {
t.Error("HasUnit went false for a mirror that exists — the file restore would stop disclosing unread data")
}
if cov.UnitRestorable {
t.Error("UnitRestorable stayed true over an unparseable manifest")
}
}