R-102: the recovery unit on the second drive becomes a way back
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable).
This commit is contained in:
@@ -0,0 +1,298 @@
|
|||||||
|
package backup
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"io"
|
||||||
|
"log"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||||
|
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// R-102 Group B — the Tier-2 route: the mirror on the SECOND DRIVE becomes a way back.
|
||||||
|
//
|
||||||
|
// The mirror in every test here is written by the PRODUCTION Tier-2 run (RunTier2 with copyTree
|
||||||
|
// standing in for rsync, the seam tier2_v2_test.go already uses), not by hand. That matters: the
|
||||||
|
// claim is "the copy Tier-2 actually writes is the copy this restore reads", and a hand-built
|
||||||
|
// fixture could only prove that the restore agrees with the test's idea of the layout.
|
||||||
|
|
||||||
|
// r102T2 is a fully-wired Tier-2 fixture: a live drive carrying the app's PRIMARY recovery unit, a
|
||||||
|
// second drive carrying the mirror Tier-2 just wrote, and a restore-side Manager with the Docker and
|
||||||
|
// DB seams injected so no daemon is touched.
|
||||||
|
type r102T2 struct {
|
||||||
|
m *Manager
|
||||||
|
fake *fakeRecoveryProvider
|
||||||
|
liveDrive string
|
||||||
|
destDrive string
|
||||||
|
destBase string
|
||||||
|
primaryUnit string
|
||||||
|
volDirs *[]string
|
||||||
|
dbPaths *[]string
|
||||||
|
}
|
||||||
|
|
||||||
|
// r102Tier2Fixture captures a unit on the live drive, runs the REAL RunTier2 to mirror it onto the
|
||||||
|
// second drive, then returns a Manager ready to restore. The mirror's contents differ from the
|
||||||
|
// primary's afterwards when the caller mutates one of them — B1 and B2 both depend on being able to
|
||||||
|
// tell the two copies apart.
|
||||||
|
func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 {
|
||||||
|
t.Helper()
|
||||||
|
tmp := t.TempDir()
|
||||||
|
live := filepath.Join(tmp, "usb")
|
||||||
|
dest := filepath.Join(tmp, "flash")
|
||||||
|
sys := filepath.Join(tmp, "sys")
|
||||||
|
|
||||||
|
primaryUnit := r102UnitOnDrive(t, live, "primary", volTars, dbDump)
|
||||||
|
|
||||||
|
sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
fake := &fakeRecoveryProvider{hdd: live, running: true}
|
||||||
|
cfg := &config.Config{}
|
||||||
|
cfg.Paths.SystemDataPath = sys
|
||||||
|
cfg.Paths.DataDir = filepath.Join(tmp, "data")
|
||||||
|
m := NewManager(cfg, sett, log.New(io.Discard, "", 0))
|
||||||
|
m.stackProvider = fake
|
||||||
|
m.systemDataPath = sys
|
||||||
|
m.tier2Mirror = copyTree
|
||||||
|
m.tier2SSDFits = func(string, int64) bool { return true }
|
||||||
|
m.samePhysicalDevice = oneDrivePerSubtree
|
||||||
|
|
||||||
|
if err := m.RunTier2("app"); err != nil {
|
||||||
|
t.Fatalf("RunTier2 (the code that writes the mirror this task reads): %v", err)
|
||||||
|
}
|
||||||
|
destBase := filepath.Join(dest, "backups", "secondary", "app")
|
||||||
|
if _, sErr := os.Stat(UnitManifestFile(tier2UnitDir(destBase))); sErr != nil {
|
||||||
|
t.Fatalf("Tier-2 did not mirror the unit — the fixture proves nothing: %v", sErr)
|
||||||
|
}
|
||||||
|
|
||||||
|
var vd, dp []string
|
||||||
|
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
|
||||||
|
vd = append(vd, dumpDir)
|
||||||
|
entries, rErr := os.ReadDir(dumpDir)
|
||||||
|
if rErr != nil {
|
||||||
|
return 0, nil
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for _, e := range entries {
|
||||||
|
if strings.HasSuffix(e.Name(), ".tar") {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, nil
|
||||||
|
}
|
||||||
|
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
||||||
|
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
||||||
|
}
|
||||||
|
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||||
|
dp = append(dp, p)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return &r102T2{m: m, fake: fake, liveDrive: live, destDrive: dest, destBase: destBase,
|
||||||
|
primaryUnit: primaryUnit, volDirs: &vd, dbPaths: &dp}
|
||||||
|
}
|
||||||
|
|
||||||
|
// markMirror rewrites the MIRROR's app.yaml so the two copies are distinguishable. It edits the copy,
|
||||||
|
// never the primary, so a restore that read the primary would return the pre-edit value.
|
||||||
|
func (f *r102T2) markMirror(t *testing.T, marker string) {
|
||||||
|
t.Helper()
|
||||||
|
p := filepath.Join(UnitComposeDir(tier2UnitDir(f.destBase)), "app.yaml")
|
||||||
|
b, err := os.ReadFile(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
out := strings.Replace(string(b), "SUBDOMAIN: primary", "SUBDOMAIN: "+marker, 1)
|
||||||
|
if out == string(b) {
|
||||||
|
t.Fatalf("mirror app.yaml did not carry the expected marker; contents:\n%s", b)
|
||||||
|
}
|
||||||
|
mustWrite(t, p, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// B1 — TestR102_Tier2UnitRestoreReadsTheSecondaryMirror.
|
||||||
|
func TestR102_Tier2UnitRestoreReadsTheSecondaryMirror(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
f.markMirror(t, "from-the-mirror")
|
||||||
|
|
||||||
|
res, err := f.m.RestoreTier2Unit("app")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Tier-2 unit restore: %v", err)
|
||||||
|
}
|
||||||
|
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "from-the-mirror" {
|
||||||
|
t.Errorf("config came from the PRIMARY unit: SUBDOMAIN=%q", got)
|
||||||
|
}
|
||||||
|
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
|
||||||
|
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
|
||||||
|
t.Errorf("volume tars read from %v, want the mirror's %q", *f.volDirs, wantVol)
|
||||||
|
}
|
||||||
|
wantDB := UnitDBDumpDir(tier2UnitDir(f.destBase))
|
||||||
|
if len(*f.dbPaths) != 1 || filepath.Dir((*f.dbPaths)[0]) != wantDB {
|
||||||
|
t.Errorf("DB dump read from %v, want a file under the mirror's %q", *f.dbPaths, wantDB)
|
||||||
|
}
|
||||||
|
if res.VolumesReplayed != 1 || res.DBsReplayed != 1 {
|
||||||
|
t.Errorf("nothing came back: volumes=%d dbs=%d", res.VolumesReplayed, res.DBsReplayed)
|
||||||
|
}
|
||||||
|
if res.ManifestVolumes != 1 || res.ManifestDBs != 1 {
|
||||||
|
t.Errorf("the mirror's manifest did not enumerate its dumps: %+v", res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// B2 — TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT.
|
||||||
|
//
|
||||||
|
// THE ACCEPTANCE TEST. Tier-2 exists for the loss of the primary drive, and in that failure the
|
||||||
|
// primary recovery unit is GONE. A test that leaves the primary in place proves the code compiles,
|
||||||
|
// not that the mirror is a route (07-backup-architecture §7.2).
|
||||||
|
//
|
||||||
|
// The primary unit is not merely emptied — the whole `backups/` tree on the live drive is removed and
|
||||||
|
// then made unreadable, so any code that tried to fall back to it would error rather than silently
|
||||||
|
// find nothing.
|
||||||
|
//
|
||||||
|
// Red-proof (recorded in REPORT.md): point RestoreTier2Unit back at the primary unit path → this test
|
||||||
|
// fails with "no readable recovery unit ... falling back to volume-only restore" behaviour and the
|
||||||
|
// mirror's marker never reaching the redeploy.
|
||||||
|
func TestR102_Tier2UnitRestoreWorksWithThePrimaryUnitABSENT(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1))
|
||||||
|
f.markMirror(t, "only-the-mirror-survives")
|
||||||
|
|
||||||
|
// The primary drive's whole backup tree is destroyed, then the parent is made unreadable so a
|
||||||
|
// fallback read cannot quietly return "nothing here".
|
||||||
|
primaryBackups := filepath.Join(f.liveDrive, "backups")
|
||||||
|
if err := os.RemoveAll(primaryBackups); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.MkdirAll(primaryBackups, 0o000); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { _ = os.Chmod(primaryBackups, 0o755) })
|
||||||
|
if _, err := os.Stat(f.primaryUnit); err == nil {
|
||||||
|
t.Fatalf("the primary unit is still readable at %q — the test would prove nothing", f.primaryUnit)
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := f.m.RestoreTier2Unit("app")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("the restore must succeed from the mirror ALONE — this is the whole point of R-102: %v", err)
|
||||||
|
}
|
||||||
|
if got := f.fake.gotEnv["SUBDOMAIN"]; got != "only-the-mirror-survives" {
|
||||||
|
t.Errorf("SUBDOMAIN=%q — the mirror was not the source", got)
|
||||||
|
}
|
||||||
|
if got := f.fake.gotEnv["SECRET_KEY"]; got != "key-primary" {
|
||||||
|
t.Errorf("the data-encrypting key did not come back from the mirrored unit: %q", got)
|
||||||
|
}
|
||||||
|
if res.VolumesReplayed != 2 {
|
||||||
|
t.Errorf("VolumesReplayed=%d, want 2 from the mirror", res.VolumesReplayed)
|
||||||
|
}
|
||||||
|
if res.DBsReplayed != 1 {
|
||||||
|
t.Errorf("DBsReplayed=%d, want 1 from the mirror", res.DBsReplayed)
|
||||||
|
}
|
||||||
|
wantVol := UnitVolumeDumpDir(tier2UnitDir(f.destBase))
|
||||||
|
if len(*f.volDirs) != 1 || (*f.volDirs)[0] != wantVol {
|
||||||
|
t.Errorf("volume source = %v, want %q", *f.volDirs, wantVol)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// B3 — TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag. Every restore in this manager shares one
|
||||||
|
// running flag. The Tier-2 unit route must be inside it, not beside it (R-351b).
|
||||||
|
func TestR102_Tier2UnitRestoreTakesTheSingleWriterFlag(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
||||||
|
|
||||||
|
// A backup/restore is already in flight.
|
||||||
|
if err := f.m.acquireRunning(); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, err := f.m.RestoreTier2Unit("app")
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("a second restore started while one was in flight")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "already in progress") {
|
||||||
|
t.Errorf("refusal = %v, want the single-writer refusal", err)
|
||||||
|
}
|
||||||
|
if f.fake.stopped {
|
||||||
|
t.Error("the app was stopped by a restore that should never have started")
|
||||||
|
}
|
||||||
|
f.m.releaseRunning()
|
||||||
|
|
||||||
|
// And the flag is RELEASED afterwards, or the next restore would be refused forever.
|
||||||
|
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
|
||||||
|
t.Fatalf("restore after the flag cleared: %v", err)
|
||||||
|
}
|
||||||
|
if err := f.m.acquireRunning(); err != nil {
|
||||||
|
t.Errorf("the running flag was not released after the Tier-2 unit restore: %v", err)
|
||||||
|
}
|
||||||
|
f.m.releaseRunning()
|
||||||
|
}
|
||||||
|
|
||||||
|
// B4 — TestR102_NoTier2CopyIsAnHonestRefusal. No recorded copy at all: refuse with the reason the
|
||||||
|
// file restore already uses, and take no outage.
|
||||||
|
func TestR102_NoTier2CopyIsAnHonestRefusal(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
||||||
|
// Forget the recorded copy entirely — the "this app has never had a Tier-2 run" shape.
|
||||||
|
if err := f.m.settings.SetCrossDriveConfig("app", &settings.CrossDriveBackup{}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
_, err := f.m.RestoreTier2Unit("app")
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected a refusal with no recorded copy")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), errNoTier2Copy.Error()) {
|
||||||
|
t.Errorf("refusal = %v, want errNoTier2Copy", err)
|
||||||
|
}
|
||||||
|
if f.fake.stopped {
|
||||||
|
t.Error("the app was stopped despite the refusal")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// B5 — TestR102_CopyAgeIsCarriedToTheSurface. The action overwrites live data with a copy of a
|
||||||
|
// certain age, so the age has to reach the surface. R-101 governs WHICH date: the success anchor,
|
||||||
|
// never the attempt clock, and Tier2CopyDate says which one it returned.
|
||||||
|
func TestR102_CopyAgeIsCarriedToTheSurface(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, "")
|
||||||
|
|
||||||
|
cov, err := f.m.Tier2RestoreCoverage("app")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("coverage: %v", err)
|
||||||
|
}
|
||||||
|
if cov.CopyLastSuccess == "" {
|
||||||
|
t.Fatal("the successful Tier-2 run left no LastSuccess for the surface to name")
|
||||||
|
}
|
||||||
|
date, proven := cov.Tier2CopyDate()
|
||||||
|
if !proven || date != cov.CopyLastSuccess {
|
||||||
|
t.Errorf("Tier2CopyDate() = (%q,%v), want the success anchor %q", date, proven, cov.CopyLastSuccess)
|
||||||
|
}
|
||||||
|
|
||||||
|
// R-101's other half: a later FAILED attempt must not become the date shown. LastRun advances,
|
||||||
|
// LastSuccess does not, and the surface must keep naming the copy that actually exists.
|
||||||
|
older := cov.CopyLastSuccess
|
||||||
|
if err := f.m.settings.UpdateCrossDriveStatus("app", func(c *settings.CrossDriveBackup) {
|
||||||
|
c.LastRun = "2099-01-01T00:00:00Z"
|
||||||
|
c.LastStatus = "error"
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
cov2, err := f.m.Tier2RestoreCoverage("app")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("coverage after a failed attempt: %v", err)
|
||||||
|
}
|
||||||
|
date2, proven2 := cov2.Tier2CopyDate()
|
||||||
|
if !proven2 || date2 != older {
|
||||||
|
t.Errorf("after a failed attempt the surface would name %q (proven=%v); want the last SUCCESS %q", date2, proven2, older)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestR102_Tier2UnitRestoreDoesNotWriteTheMirror — a restore reads its source. If the Tier-2 copy
|
||||||
|
// were mutated, the second drive would stop being a way back the moment it was used once.
|
||||||
|
func TestR102_Tier2UnitRestoreDoesNotWriteTheMirror(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
before := fingerprintTree(t, f.destBase)
|
||||||
|
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
|
||||||
|
t.Fatalf("restore: %v", err)
|
||||||
|
}
|
||||||
|
if after := fingerprintTree(t, f.destBase); after != before {
|
||||||
|
t.Error("the Tier-2 copy was written to by a restore that only reads it")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,369 @@
|
|||||||
|
package backup
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"io"
|
||||||
|
"log"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// R-102 Group A — the recovery-unit restore can be pointed at a unit ANYWHERE.
|
||||||
|
//
|
||||||
|
// The defect: every reader of a recovery unit could only name a path under `backups/primary/`
|
||||||
|
// (appbackup/paths.go joined it literally), while Tier-2 mirrored each app's whole unit to
|
||||||
|
// `<dest>/backups/secondary/<app>/recovery-unit/` on every run. So the mirror was written nightly for
|
||||||
|
// months and read by nothing — and it was unreadable in exactly the failure Tier-2 exists for, where
|
||||||
|
// the primary drive and its unit are gone (07-backup-architecture §6.3, §7.2).
|
||||||
|
//
|
||||||
|
// Every test below asserts WHICH DIRECTORY was read, not merely that a restore succeeded. A test that
|
||||||
|
// only checked `err == nil` would have passed against the pre-fix code, because the pre-fix code read
|
||||||
|
// a real unit — just never the one that survives.
|
||||||
|
|
||||||
|
// r102Unit captures a REAL recovery unit for "app" onto its own fresh drive, with an identifying
|
||||||
|
// SUBDOMAIN, a portable DB password and a portable data key, plus optional volume tars and a DB dump.
|
||||||
|
//
|
||||||
|
// The unit is produced by the production CaptureRecoveryUnit, not hand-written: the capture side and
|
||||||
|
// the restore side must meet at real bytes, so a change to the on-disk shape cannot pass by having a
|
||||||
|
// test agree with itself (the reason captureFixtureUnit is built the same way).
|
||||||
|
func r102Unit(t *testing.T, marker string, volTars []string, dbDump string) (drive, unitDir string) {
|
||||||
|
t.Helper()
|
||||||
|
tmp := t.TempDir()
|
||||||
|
drive = filepath.Join(tmp, "drive")
|
||||||
|
return drive, r102UnitOnDrive(t, drive, marker, volTars, dbDump)
|
||||||
|
}
|
||||||
|
|
||||||
|
// r102UnitOnDrive is r102Unit against a drive the caller already owns — the Tier-2 fixture needs the
|
||||||
|
// unit to sit on the SAME drive the Tier-2 run will mirror FROM.
|
||||||
|
func r102UnitOnDrive(t *testing.T, drive, marker string, volTars []string, dbDump string) (unitDir string) {
|
||||||
|
t.Helper()
|
||||||
|
tmp := t.TempDir()
|
||||||
|
stackDir := filepath.Join(tmp, "stack")
|
||||||
|
if err := os.MkdirAll(stackDir, 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mustWrite(t, filepath.Join(stackDir, "docker-compose.yml"),
|
||||||
|
"services:\n app:\n image: example/app:1\n db:\n image: postgres:16\n")
|
||||||
|
mustWrite(t, filepath.Join(stackDir, ".felhom.yml"), "display_name: App\n")
|
||||||
|
mustWrite(t, filepath.Join(stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: "+marker+"\n")
|
||||||
|
|
||||||
|
info := RecoveryInfo{
|
||||||
|
StackDir: stackDir,
|
||||||
|
DisplayName: "App",
|
||||||
|
ImagePins: []string{"example/app:1"},
|
||||||
|
NonSecretEnv: map[string]string{"SUBDOMAIN": marker},
|
||||||
|
SecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"},
|
||||||
|
DataKeyEnvVars: []string{"SECRET_KEY"},
|
||||||
|
PortableSecretEnvVars: []string{"DB_PASSWORD", "SECRET_KEY"},
|
||||||
|
PortableSecrets: map[string]string{"DB_PASSWORD": "pw-" + marker, "SECRET_KEY": "key-" + marker},
|
||||||
|
}
|
||||||
|
unitDir = RecoveryUnitPath(drive, "app")
|
||||||
|
// The dumps are written BEFORE the capture so the manifest enumerates them — a manifest that
|
||||||
|
// lists nothing is the R-353 "the backup held only settings" shape, which is a different case.
|
||||||
|
for _, v := range volTars {
|
||||||
|
mustWrite(t, filepath.Join(UnitVolumeDumpDir(unitDir), v), "tar:"+v+":"+marker)
|
||||||
|
}
|
||||||
|
if dbDump != "" {
|
||||||
|
mustWrite(t, filepath.Join(UnitDBDumpDir(unitDir), "app-postgres.sql"), dbDump)
|
||||||
|
}
|
||||||
|
|
||||||
|
m := &Manager{
|
||||||
|
logger: log.New(io.Discard, "", 0),
|
||||||
|
systemDataPath: filepath.Join(tmp, "system"),
|
||||||
|
stackProvider: &fakeRecoveryProvider{info: info, hdd: drive},
|
||||||
|
version: "vtest",
|
||||||
|
}
|
||||||
|
if err := m.CaptureRecoveryUnit("app"); err != nil {
|
||||||
|
t.Fatalf("capture fixture %q: %v", marker, err)
|
||||||
|
}
|
||||||
|
return unitDir
|
||||||
|
}
|
||||||
|
|
||||||
|
// r102Manager builds a restore-side Manager whose LIVE drive is liveDrive, with the Docker and DB
|
||||||
|
// seams injected so no daemon is touched. The returned slices record the directories each leg read.
|
||||||
|
func r102Manager(t *testing.T, liveDrive string) (m *Manager, fake *fakeRecoveryProvider, volDirs, dbPaths *[]string) {
|
||||||
|
t.Helper()
|
||||||
|
fake = &fakeRecoveryProvider{hdd: liveDrive, running: true}
|
||||||
|
m = &Manager{
|
||||||
|
logger: log.New(io.Discard, "", 0),
|
||||||
|
systemDataPath: filepath.Join(liveDrive, "..", "sys"),
|
||||||
|
stackProvider: fake,
|
||||||
|
}
|
||||||
|
var vd, dp []string
|
||||||
|
m.volumeReplayFrom = func(_, dumpDir string) (int, error) {
|
||||||
|
vd = append(vd, dumpDir)
|
||||||
|
entries, err := os.ReadDir(dumpDir)
|
||||||
|
if err != nil {
|
||||||
|
return 0, nil
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for _, e := range entries {
|
||||||
|
if strings.HasSuffix(e.Name(), ".tar") {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, nil
|
||||||
|
}
|
||||||
|
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
||||||
|
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
||||||
|
}
|
||||||
|
m.importDBDump = func(_ context.Context, _ DiscoveredDB, p string) error {
|
||||||
|
dp = append(dp, p)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return m, fake, &vd, &dp
|
||||||
|
}
|
||||||
|
|
||||||
|
// A2 — TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir.
|
||||||
|
//
|
||||||
|
// Two complete units exist on disk with DIFFERENT contents. The restore is pointed at the second one
|
||||||
|
// and must read every leg — env, secrets, volume tars, DB dump — out of THAT directory. The first
|
||||||
|
// unit is the app's own primary unit and is present and perfectly readable, which is what makes the
|
||||||
|
// assertion mean something: a restore that ignored unitDir would still succeed, and would silently
|
||||||
|
// return the wrong copy's data.
|
||||||
|
func TestR102_RestoreFromRecoveryUnitAtReadsTheGivenDir(t *testing.T) {
|
||||||
|
primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
_, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar", "vol_b.tar"}, pgDump(2))
|
||||||
|
|
||||||
|
m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive)
|
||||||
|
|
||||||
|
res, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("restore from the named unit: %v", err)
|
||||||
|
}
|
||||||
|
if fake.gotEnv == nil {
|
||||||
|
t.Fatal("recreate was never called — the restore did not reach the redeploy")
|
||||||
|
}
|
||||||
|
if got := fake.gotEnv["SUBDOMAIN"]; got != "mirror" {
|
||||||
|
t.Errorf("config came from the WRONG unit: SUBDOMAIN=%q, want %q", got, "mirror")
|
||||||
|
}
|
||||||
|
if got := fake.gotEnv["SECRET_KEY"]; got != "key-mirror" {
|
||||||
|
t.Errorf("data key came from the WRONG unit: %q", got)
|
||||||
|
}
|
||||||
|
if got := fake.gotEnv["DB_PASSWORD"]; got != "pw-mirror" {
|
||||||
|
t.Errorf("DB password came from the WRONG unit: %q", got)
|
||||||
|
}
|
||||||
|
if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(mirrorUnit) {
|
||||||
|
t.Errorf("volume tars read from %v, want %q", *volDirs, UnitVolumeDumpDir(mirrorUnit))
|
||||||
|
}
|
||||||
|
if res.VolumesReplayed != 2 {
|
||||||
|
t.Errorf("VolumesReplayed=%d, want 2 (the mirror holds two tars; the primary holds one)", res.VolumesReplayed)
|
||||||
|
}
|
||||||
|
if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(mirrorUnit) {
|
||||||
|
t.Errorf("DB dump read from %v, want a file under %q", *dbPaths, UnitDBDumpDir(mirrorUnit))
|
||||||
|
}
|
||||||
|
// The negative control: nothing was read out of the primary unit.
|
||||||
|
for _, d := range *volDirs {
|
||||||
|
if strings.HasPrefix(d, primaryUnit) {
|
||||||
|
t.Errorf("a leg was read from the PRIMARY unit %q despite a mirror being named", d)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A3 — TestR102_PrimaryPathIsUnchanged. The one-argument wrapper must still resolve to the app's own
|
||||||
|
// drive, `backups/primary/<stack>`. Asserted on the DIRECTORIES the legs actually read, not on a
|
||||||
|
// path expression re-derived in the test.
|
||||||
|
func TestR102_PrimaryPathIsUnchanged(t *testing.T) {
|
||||||
|
primaryDrive, primaryUnit := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
m, fake, volDirs, dbPaths := r102Manager(t, primaryDrive)
|
||||||
|
|
||||||
|
if _, err := m.RestoreFromRecoveryUnit("app"); err != nil {
|
||||||
|
t.Fatalf("primary restore: %v", err)
|
||||||
|
}
|
||||||
|
if got := fake.gotEnv["SUBDOMAIN"]; got != "primary" {
|
||||||
|
t.Errorf("SUBDOMAIN=%q, want the primary unit's %q", got, "primary")
|
||||||
|
}
|
||||||
|
if !strings.Contains(primaryUnit, filepath.Join("backups", "primary", "app")) {
|
||||||
|
t.Fatalf("fixture is not where the primary unit belongs: %q", primaryUnit)
|
||||||
|
}
|
||||||
|
if len(*volDirs) != 1 || (*volDirs)[0] != UnitVolumeDumpDir(primaryUnit) {
|
||||||
|
t.Errorf("volume dir = %v, want %q", *volDirs, UnitVolumeDumpDir(primaryUnit))
|
||||||
|
}
|
||||||
|
if len(*dbPaths) != 1 || filepath.Dir((*dbPaths)[0]) != UnitDBDumpDir(primaryUnit) {
|
||||||
|
t.Errorf("db dump dir = %v, want %q", *dbPaths, UnitDBDumpDir(primaryUnit))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A4 — TestR102_LiveDestinationIsUnchanged. THE SOURCE MOVES; THE DESTINATION DOES NOT.
|
||||||
|
//
|
||||||
|
// The restore reads a mirror that lives on a different drive entirely, and must still write the app
|
||||||
|
// back to its own live namespace: the definition through RecreateStackDefinitionFromUnit (whose
|
||||||
|
// compose dir must be the MIRROR's, since that is the definition being restored), and the data into
|
||||||
|
// the LIVE Docker volumes and the LIVE database container. A restore that also relocated the app's
|
||||||
|
// data would be a migration, not a restore.
|
||||||
|
//
|
||||||
|
// The observable for "the destination did not move" is that nothing under the mirror's own drive was
|
||||||
|
// written to: the mirror tree is byte-identical before and after.
|
||||||
|
func TestR102_LiveDestinationIsUnchanged(t *testing.T) {
|
||||||
|
primaryDrive, _ := r102Unit(t, "primary", []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
mirrorDrive, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2))
|
||||||
|
|
||||||
|
before := fingerprintTree(t, mirrorDrive)
|
||||||
|
|
||||||
|
m, fake, _, _ := r102Manager(t, primaryDrive)
|
||||||
|
var recreateComposeDir string
|
||||||
|
m.stackProvider = &r102RecordingProvider{fakeRecoveryProvider: fake, composeDirOut: &recreateComposeDir}
|
||||||
|
|
||||||
|
if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil {
|
||||||
|
t.Fatalf("restore: %v", err)
|
||||||
|
}
|
||||||
|
if recreateComposeDir != UnitComposeDir(mirrorUnit) {
|
||||||
|
t.Errorf("recreate read compose from %q, want the mirror's %q", recreateComposeDir, UnitComposeDir(mirrorUnit))
|
||||||
|
}
|
||||||
|
if after := fingerprintTree(t, mirrorDrive); after != before {
|
||||||
|
t.Error("the SOURCE mirror was written to — a restore reads its source and never writes it")
|
||||||
|
}
|
||||||
|
// The live drive is where the app lives, and the restore must not have relocated it: the app's
|
||||||
|
// own drive path is still the one the manager resolves for it.
|
||||||
|
if got := m.GetAppDrivePath("app"); got != primaryDrive {
|
||||||
|
t.Errorf("the app's live drive moved to %q, want %q", got, primaryDrive)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// r102RecordingProvider captures the compose directory RecreateStackDefinitionFromUnit is handed —
|
||||||
|
// the one place the restore names a source directory to the guest side.
|
||||||
|
type r102RecordingProvider struct {
|
||||||
|
*fakeRecoveryProvider
|
||||||
|
composeDirOut *string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *r102RecordingProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error {
|
||||||
|
*f.composeDirOut = composeDir
|
||||||
|
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A5 — TestR102_MutationOrderIsPreserved. R-47's ordering is pinned and R-102 must not have moved it:
|
||||||
|
// stop → volumes → recreate → DB-only start → replay → full start. The DB-only phase exists because
|
||||||
|
// starting the whole stack first let the application rebuild schema underneath the replay (H4,
|
||||||
|
// DIAG-immich-restore-round2-2026-07-19).
|
||||||
|
//
|
||||||
|
// The state AT REPLAY TIME is asserted, not just the call sequence — H4 looked correctly ordered and
|
||||||
|
// the app was up.
|
||||||
|
//
|
||||||
|
// Red-proof (recorded in REPORT.md): swap the volume replay and the recreate step in
|
||||||
|
// RestoreFromRecoveryUnitAt → this test fails on the sequence assertion.
|
||||||
|
func TestR102_MutationOrderIsPreserved(t *testing.T) {
|
||||||
|
primaryDrive, _ := r102Unit(t, "primary", nil, pgDump(1))
|
||||||
|
_, mirrorUnit := r102Unit(t, "mirror", []string{"vol_a.tar"}, pgDump(2))
|
||||||
|
|
||||||
|
m, fake, _, _ := r102Manager(t, primaryDrive)
|
||||||
|
|
||||||
|
var volumesDoneAtRecreate, fullUpAtReplay bool
|
||||||
|
var volumesReplayed bool
|
||||||
|
m.volumeReplayFrom = func(_, _ string) (int, error) {
|
||||||
|
volumesReplayed = true
|
||||||
|
if fake.gotEnv != nil {
|
||||||
|
t.Error("volumes were replayed AFTER the definition was recreated")
|
||||||
|
}
|
||||||
|
return 1, nil
|
||||||
|
}
|
||||||
|
prov := &r102OrderProvider{fakeRecoveryProvider: fake, volumesReplayed: &volumesReplayed, seen: &volumesDoneAtRecreate}
|
||||||
|
m.stackProvider = prov
|
||||||
|
m.importDBDump = func(context.Context, DiscoveredDB, string) error {
|
||||||
|
fullUpAtReplay = fake.fullStarted
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := m.RestoreFromRecoveryUnitAt("app", mirrorUnit); err != nil {
|
||||||
|
t.Fatalf("restore: %v", err)
|
||||||
|
}
|
||||||
|
if got := strings.Join(fake.calls, ","); got != "stop,recreate,startsvc:db,start" {
|
||||||
|
t.Fatalf("sequence = %q, want stop → recreate → db-only start → (replay) → full start", got)
|
||||||
|
}
|
||||||
|
if !volumesDoneAtRecreate {
|
||||||
|
t.Error("the volume replay had not run when the definition was recreated")
|
||||||
|
}
|
||||||
|
if fullUpAtReplay {
|
||||||
|
t.Error("the FULL stack was already up when the DB replay fired — this is H4 exactly (R-47)")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// r102OrderProvider records whether the volume replay had already happened by the time the definition
|
||||||
|
// was recreated — the ordering fact the call log alone cannot carry.
|
||||||
|
type r102OrderProvider struct {
|
||||||
|
*fakeRecoveryProvider
|
||||||
|
volumesReplayed *bool
|
||||||
|
seen *bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *r102OrderProvider) RecreateStackDefinitionFromUnit(name, composeDir string, fullEnv map[string]string) error {
|
||||||
|
*f.seen = *f.volumesReplayed
|
||||||
|
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, fullEnv)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A6 — TestR102_MissingManifestInMirrorFailsClosed. A DIRECTORY IS NOT A PACKAGE.
|
||||||
|
//
|
||||||
|
// `<dest>/backups/secondary/<app>/recovery-unit/` can exist and be useless: a copy interrupted
|
||||||
|
// mid-run, or a tree whose manifest.json never landed. The Tier-2 unit restore must refuse it and
|
||||||
|
// must leave the app running — a restore armed over an unopenable unit would stop the app, replay
|
||||||
|
// nothing, and rewrite its definition from an empty capture. Same lesson as R-358 one tier over.
|
||||||
|
//
|
||||||
|
// Placed with Group A because it is the fail-closed half of the parameterised unit; the route it
|
||||||
|
// exercises is Part 1.3's.
|
||||||
|
func TestR102_MissingManifestInMirrorFailsClosed(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
setup func(t *testing.T, unitDir string)
|
||||||
|
}{
|
||||||
|
{"manifest absent", func(t *testing.T, unitDir string) {
|
||||||
|
if err := os.Remove(UnitManifestFile(unitDir)); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}},
|
||||||
|
{"manifest unparseable", func(t *testing.T, unitDir string) {
|
||||||
|
mustWrite(t, UnitManifestFile(unitDir), "{ this is not json")
|
||||||
|
}},
|
||||||
|
{"recovery-unit absent entirely", func(t *testing.T, unitDir string) {
|
||||||
|
if err := os.RemoveAll(unitDir); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}},
|
||||||
|
} {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
tc.setup(t, tier2UnitDir(f.destBase))
|
||||||
|
|
||||||
|
cov, covErr := f.m.Tier2RestoreCoverage("app")
|
||||||
|
if covErr != nil {
|
||||||
|
t.Fatalf("coverage: %v", covErr)
|
||||||
|
}
|
||||||
|
if cov.CanRestoreUnit() {
|
||||||
|
t.Error("CanRestoreUnit() said yes over an unopenable mirror — the surface would offer the action")
|
||||||
|
}
|
||||||
|
_, err := f.m.RestoreTier2Unit("app")
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("the restore ran over an unopenable mirror")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), ErrTier2NoUnitInCopy.Error()) {
|
||||||
|
t.Errorf("refusal = %v, want ErrTier2NoUnitInCopy", err)
|
||||||
|
}
|
||||||
|
if f.fake.stopped {
|
||||||
|
t.Error("the app was STOPPED despite the refusal — the outage this gate exists to avoid")
|
||||||
|
}
|
||||||
|
if f.fake.gotEnv != nil {
|
||||||
|
t.Error("the app's definition was rewritten despite the refusal")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestR102_UnopenableMirrorIsStillDisclosedAsUnread is the other side of A6, and the reason
|
||||||
|
// UnitRestorable is a second field rather than a widening of HasUnit: a half-copied mirror cannot be
|
||||||
|
// restored, but it IS captured data the file restore is not looking at, so the disclosure must stand.
|
||||||
|
func TestR102_UnopenableMirrorIsStillDisclosedAsUnread(t *testing.T) {
|
||||||
|
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
|
||||||
|
mustWrite(t, UnitManifestFile(tier2UnitDir(f.destBase)), "{ not json")
|
||||||
|
|
||||||
|
cov, err := f.m.Tier2RestoreCoverage("app")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("coverage: %v", err)
|
||||||
|
}
|
||||||
|
if !cov.HasUnit {
|
||||||
|
t.Error("HasUnit went false for a mirror that exists — the file restore would stop disclosing unread data")
|
||||||
|
}
|
||||||
|
if cov.UnitRestorable {
|
||||||
|
t.Error("UnitRestorable stayed true over an unparseable manifest")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -89,10 +89,25 @@ func (m *Manager) reimportDBDumpsFrom(ctx context.Context, stackName, dumpDir st
|
|||||||
return imported, nil
|
return imported, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dbReimportTimeout bounds a DB replay so a stuck import cannot hang a restore indefinitely. Named
|
||||||
|
// once because both bounded entry points below must agree: two paths that differ in how long they
|
||||||
|
// let a wedged import hold the restore are two different products (R-102 added the second one).
|
||||||
|
const dbReimportTimeout = 35 * time.Minute
|
||||||
|
|
||||||
// reimportDBDumpsCtx is a small helper that runs reimportDBDumps with a bounded context so a stuck DB
|
// reimportDBDumpsCtx is a small helper that runs reimportDBDumps with a bounded context so a stuck DB
|
||||||
// import cannot hang the restore indefinitely.
|
// import cannot hang the restore indefinitely.
|
||||||
func (m *Manager) reimportDBDumpsCtx(stackName, nsRoot string) (int, error) {
|
func (m *Manager) reimportDBDumpsCtx(stackName, nsRoot string) (int, error) {
|
||||||
ctx, cancel := context.WithTimeout(context.Background(), 35*time.Minute)
|
ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout)
|
||||||
defer cancel()
|
defer cancel()
|
||||||
return m.reimportDBDumps(ctx, stackName, nsRoot)
|
return m.reimportDBDumps(ctx, stackName, nsRoot)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// reimportDBDumpsAtCtx is reimportDBDumpsCtx with an EXPLICIT dump directory — the bounded-context
|
||||||
|
// twin of reimportDBDumpsFrom, added for R-102 so the Tier-2 unit restore can replay out of the
|
||||||
|
// secondary mirror. Same discovery/import seams, same failure semantics, same bound; only the source
|
||||||
|
// directory differs.
|
||||||
|
func (m *Manager) reimportDBDumpsAtCtx(stackName, dumpDir string) (int, error) {
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout)
|
||||||
|
defer cancel()
|
||||||
|
return m.reimportDBDumpsFrom(ctx, stackName, dumpDir)
|
||||||
|
}
|
||||||
|
|||||||
@@ -178,7 +178,36 @@ type UnitRestoreResult struct {
|
|||||||
// D5: this no longer needs the guest. A restore with the guest's app.yaml absent succeeds, which is
|
// D5: this no longer needs the guest. A restore with the guest's app.yaml absent succeeds, which is
|
||||||
// pinned by TestRestoreFromRecoveryUnitWithGuestAbsent — the withheld class is regenerated (O4) and
|
// pinned by TestRestoreFromRecoveryUnitWithGuestAbsent — the withheld class is regenerated (O4) and
|
||||||
// only a data key missing from BOTH sources still refuses.
|
// only a data key missing from BOTH sources still refuses.
|
||||||
|
//
|
||||||
|
// R-102: this is now the thin caller. It names the PRIMARY unit — the app's own drive,
|
||||||
|
// backups/primary/<stack> — and hands it to RestoreFromRecoveryUnitAt, which holds the whole body.
|
||||||
|
// An unresolvable drive path is still refused inside …At, in the same place and with the same
|
||||||
|
// message, so the order of the checks a caller can observe is unchanged.
|
||||||
func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, error) {
|
func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, error) {
|
||||||
|
return m.RestoreFromRecoveryUnitAt(stackName, RecoveryUnitPath(m.namespaceRoot(m.GetAppDrivePath(stackName)), stackName))
|
||||||
|
}
|
||||||
|
|
||||||
|
// RestoreFromRecoveryUnitAt is RestoreFromRecoveryUnit with an EXPLICIT recovery-unit directory.
|
||||||
|
//
|
||||||
|
// R-102. ONE implementation, two callers — the same rule restoreDockerVolumesFrom states beside
|
||||||
|
// itself in restore.go, and for the same reason: a second copy of this body is exactly how the local
|
||||||
|
// path and the off-site path drifted apart until nothing compared them.
|
||||||
|
//
|
||||||
|
// The reason it exists: Tier-2 mirrors the app's whole recovery unit to
|
||||||
|
// <dest>/backups/secondary/<stack>/recovery-unit/ on every run, and until now every reader of a unit
|
||||||
|
// could only name a path under backups/primary/. So in the one failure Tier-2 exists for — the
|
||||||
|
// primary drive is lost, taking the primary unit with it — the surviving copy could not be opened by
|
||||||
|
// any action in the product (07-backup-architecture §6.3, §7.2).
|
||||||
|
//
|
||||||
|
// THE SOURCE MOVES; THE DESTINATION DOES NOT. unitDir changes only where the manifest, the compose
|
||||||
|
// capture, the .sql dumps and the volume tars are READ from. The app's data is written back to the
|
||||||
|
// live Docker volumes and the live database container, and its definition to the guest, exactly as
|
||||||
|
// before — a restore that also relocated the app's data would be a migration, not a restore.
|
||||||
|
//
|
||||||
|
// Everything else is pinned and unchanged: the mutation order (stop → volumes → recreate → DB-only
|
||||||
|
// start → replay → start, R-47), the secret reconciliation with unit-over-guest precedence and the
|
||||||
|
// fail-closed data-key gate, and the no-unit fallback to RestoreApp with its CountsUnknown handling.
|
||||||
|
func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRestoreResult, error) {
|
||||||
var res UnitRestoreResult
|
var res UnitRestoreResult
|
||||||
if m.stackProvider == nil {
|
if m.stackProvider == nil {
|
||||||
return res, fmt.Errorf("stack provider not configured")
|
return res, fmt.Errorf("stack provider not configured")
|
||||||
@@ -197,15 +226,18 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult,
|
|||||||
m.mu.Unlock()
|
m.mu.Unlock()
|
||||||
}()
|
}()
|
||||||
|
|
||||||
|
// The DESTINATION side, and it is deliberately still resolved here: RestoreApp (the no-unit
|
||||||
|
// fallback below) needs it, and 07-backup-architecture §6.3 records that the restore destination
|
||||||
|
// is resolved by the same rule as the capture destination. It is no longer used to derive any
|
||||||
|
// SOURCE path — that is what unitDir is for.
|
||||||
drivePath := m.GetAppDrivePath(stackName)
|
drivePath := m.GetAppDrivePath(stackName)
|
||||||
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
||||||
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
||||||
}
|
}
|
||||||
nsRoot := m.namespaceRoot(drivePath)
|
|
||||||
|
|
||||||
manifest := readManifest(RecoveryUnitManifestPath(nsRoot, stackName))
|
manifest := readManifest(UnitManifestFile(unitDir))
|
||||||
if manifest == nil {
|
if manifest == nil {
|
||||||
m.logger.Printf("[WARN] [backup] No recovery unit for %s — falling back to volume-only restore", stackName)
|
m.logger.Printf("[WARN] [backup] No readable recovery unit for %s at %s — falling back to volume-only restore", stackName, unitDir)
|
||||||
m.mu.Lock()
|
m.mu.Lock()
|
||||||
m.running = false // RestoreApp re-acquires the running flag
|
m.running = false // RestoreApp re-acquires the running flag
|
||||||
m.mu.Unlock()
|
m.mu.Unlock()
|
||||||
@@ -220,7 +252,7 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult,
|
|||||||
// was just parsed above, so the claim and the outcome are counted from the same document.
|
// was just parsed above, so the claim and the outcome are counted from the same document.
|
||||||
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
|
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
|
||||||
|
|
||||||
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
|
composeDir := UnitComposeDir(unitDir)
|
||||||
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
|
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
|
||||||
|
|
||||||
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
|
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
|
||||||
@@ -275,8 +307,11 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult,
|
|||||||
stackName, len(unresolved), unresolved)
|
stackName, len(unresolved), unresolved)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit: images=%d, secrets recovered=%d/%d, data_keys=%d",
|
// R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with
|
||||||
stackName, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars))
|
// two answers, and "an absent log line is not evidence" — the drill reads this line to prove the
|
||||||
|
// secondary mirror, not the primary unit, was the source. It is a path, never a secret.
|
||||||
|
m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit %s: images=%d, secrets recovered=%d/%d, data_keys=%d",
|
||||||
|
stackName, unitDir, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars))
|
||||||
|
|
||||||
// R-47: which compose service holds the database, and is there anything to replay? Resolved from
|
// R-47: which compose service holds the database, and is there anything to replay? Resolved from
|
||||||
// the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed
|
// the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed
|
||||||
@@ -286,7 +321,8 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult,
|
|||||||
// "cannot tell" is not "no database" — leave it empty and let the gate decide.
|
// "cannot tell" is not "no database" — leave it empty and let the gate decide.
|
||||||
m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr)
|
m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr)
|
||||||
}
|
}
|
||||||
hasDumps := hasReplayableDump(AppDBDumpPath(nsRoot, stackName))
|
dbDumpDir := UnitDBDumpDir(unitDir)
|
||||||
|
hasDumps := hasReplayableDump(dbDumpDir)
|
||||||
if hasDumps && len(dbServices) == 0 {
|
if hasDumps && len(dbServices) == 0 {
|
||||||
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName)
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName)
|
||||||
return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stackName)
|
return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stackName)
|
||||||
@@ -302,7 +338,17 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult,
|
|||||||
// R-353: the count is captured even when the replay errors — a partial replay is a fact the
|
// R-353: the count is captured even when the replay errors — a partial replay is a fact the
|
||||||
// customer's sentence has to be built from, and discarding it on the error path is how Scenario C
|
// customer's sentence has to be built from, and discarding it on the error path is how Scenario C
|
||||||
// would end up wearing Scenario B's wording.
|
// would end up wearing Scenario B's wording.
|
||||||
replayed, volErr := m.restoreDockerVolumes(stackName, drivePath)
|
// R-354's volume-REPLAY seam, reused here rather than a second one being invented. It defaults to
|
||||||
|
// the real restoreDockerVolumesFrom, so production behaviour is byte-for-byte what it was; what it
|
||||||
|
// buys is that R-102's acceptance test can assert WHICH directory the tars came out of without a
|
||||||
|
// Docker daemon. For 40 of the 53 catalogue apps that archive is the entire dataset, so "the
|
||||||
|
// mirror was the source" has to be provable for the volume leg too, not only for the env and the
|
||||||
|
// database.
|
||||||
|
volReplay := m.volumeReplayFrom
|
||||||
|
if volReplay == nil {
|
||||||
|
volReplay = m.restoreDockerVolumesFrom
|
||||||
|
}
|
||||||
|
replayed, volErr := volReplay(stackName, UnitVolumeDumpDir(unitDir))
|
||||||
res.VolumesReplayed = replayed
|
res.VolumesReplayed = replayed
|
||||||
if volErr != nil {
|
if volErr != nil {
|
||||||
m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, volErr)
|
m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, volErr)
|
||||||
@@ -322,7 +368,7 @@ func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult,
|
|||||||
if dataErr == nil {
|
if dataErr == nil {
|
||||||
dataErr = err
|
dataErr = err
|
||||||
}
|
}
|
||||||
} else if n, err := m.reimportDBDumpsCtx(stackName, nsRoot); err != nil {
|
} else if n, err := m.reimportDBDumpsAtCtx(stackName, dbDumpDir); err != nil {
|
||||||
res.DBsReplayed = n // partial credit: whatever imported before the failure really did import
|
res.DBsReplayed = n // partial credit: whatever imported before the failure really did import
|
||||||
m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err)
|
m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err)
|
||||||
if dataErr == nil {
|
if dataErr == nil {
|
||||||
|
|||||||
@@ -39,6 +39,12 @@ var (
|
|||||||
// are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action
|
// are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action
|
||||||
// that does work, instead of taking an outage and reporting "no missing files".
|
// that does work, instead of taking an outage and reporting "no missing files".
|
||||||
ErrTier2NoRestorableData = errors.New("ennek az alkalmazásnak az adatai nem ebből a másolatból állíthatók vissza")
|
ErrTier2NoRestorableData = errors.New("ennek az alkalmazásnak az adatai nem ebből a másolatból állíthatók vissza")
|
||||||
|
// ErrTier2NoUnitInCopy (R-102) — the recorded Tier-2 copy holds no OPENABLE recovery unit: either
|
||||||
|
// recovery-unit/ is absent, or it is a directory without a readable manifest.json. Exported so the
|
||||||
|
// handler can refuse before beginning any op. FAIL CLOSED is the whole point of the second half:
|
||||||
|
// a directory that exists is not a package, and reading a half-copied mirror as if it were one is
|
||||||
|
// how a restore would overwrite live data with nothing.
|
||||||
|
ErrTier2NoUnitInCopy = errors.New("a másodlagos másolatban nincs megnyitható mentési egység ehhez az alkalmazáshoz")
|
||||||
)
|
)
|
||||||
|
|
||||||
// Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1
|
// Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1
|
||||||
@@ -50,14 +56,64 @@ var (
|
|||||||
// tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in
|
// tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in
|
||||||
// which case the restore is a guaranteed no-op no matter how much data was lost.
|
// which case the restore is a guaranteed no-op no matter how much data was lost.
|
||||||
type Tier2Coverage struct {
|
type Tier2Coverage struct {
|
||||||
Legs []string // subtrees this restore reads and that exist in the copy: "hdd", "userdata"
|
Legs []string // subtrees the FILE restore reads and that exist in the copy: "hdd", "userdata"
|
||||||
HasUnit bool // recovery-unit/ present — captured, but NOT restorable by this path
|
HasUnit bool // recovery-unit/ present as a DIRECTORY — the disclosure fact, see below
|
||||||
|
|
||||||
|
// UnitRestorable (R-102) — the mirror is a real PACKAGE, not merely a directory: recovery-unit/
|
||||||
|
// exists AND carries a manifest.json that parses. This is the gate for the UNIT restore.
|
||||||
|
//
|
||||||
|
// It is a SECOND FIELD and not a widening of HasUnit, and the distinction is load-bearing in both
|
||||||
|
// directions. HasUnit answers "is there captured data this FILE restore is not looking at?" — the
|
||||||
|
// question tier2UnitNotCoveredMsg is appended for, and the honest answer for a half-copied mirror
|
||||||
|
// is still yes. UnitRestorable answers "can the unit restore open this?" — and for that same
|
||||||
|
// half-copied mirror the answer is no. Collapsing them would either silence a true disclosure or
|
||||||
|
// arm a restore over an unopenable package.
|
||||||
|
UnitRestorable bool
|
||||||
|
|
||||||
|
// CopyLastRun / CopyLastSuccess — WHEN the copy this restore would read was written, so the
|
||||||
|
// surface can name the date before it overwrites anything with it (Scenario E). Filled only by
|
||||||
|
// Tier2RestoreCoverage, which is the path that holds the settings; tier2CoverageAt is a pure
|
||||||
|
// filesystem inspection and leaves them empty. They are STRINGS in the recorded RFC3339 form,
|
||||||
|
// carried verbatim — no formatting decision is taken in this package.
|
||||||
|
//
|
||||||
|
// R-101 applies here exactly as it does on the backup card: CopyLastRun is the ATTEMPT clock and
|
||||||
|
// CopyLastSuccess is the only evidence a copy was actually made. A surface that shows one must
|
||||||
|
// not present it as the other.
|
||||||
|
CopyLastRun string
|
||||||
|
CopyLastSuccess string
|
||||||
}
|
}
|
||||||
|
|
||||||
// CanRestore reports whether the restore has any subtree to read at all.
|
// CanRestore reports whether the FILE restore has any subtree to read at all.
|
||||||
|
//
|
||||||
|
// R-102/R-103: this answers exactly one question and must keep answering only that one. Widening it
|
||||||
|
// to include the unit is R-356 arriving a second time — there, ONE predicate meant both "has this app
|
||||||
|
// a drive?" and "is this app installed?", and it refused 40 running apps for months while telling
|
||||||
|
// their owners to reinstall them somewhere those apps never offer. Two questions, two predicates.
|
||||||
func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 }
|
func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 }
|
||||||
|
|
||||||
// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat — no side effects.
|
// CanRestoreUnit reports whether the UNIT restore can run from this copy — the second predicate.
|
||||||
|
func (c Tier2Coverage) CanRestoreUnit() bool { return c.UnitRestorable }
|
||||||
|
|
||||||
|
// tier2UnitDir returns the recovery-unit directory inside a resolved Tier-2 copy. ONE expression of
|
||||||
|
// where Tier-2 puts the mirror; tier2.go writes it at the same relative name ("Unit leg (always)").
|
||||||
|
func tier2UnitDir(destBase string) string {
|
||||||
|
return filepath.Join(destBase, "recovery-unit")
|
||||||
|
}
|
||||||
|
|
||||||
|
// tier2UnitIsOpenable reports whether a mirrored unit directory is a PACKAGE and not just a
|
||||||
|
// directory. Fail-closed by construction: the manifest must be present AND parse (readManifest
|
||||||
|
// returns nil for both a missing file and malformed JSON), because an unopenable unit that armed a
|
||||||
|
// restore would stop the app, replay nothing, and rewrite its definition from an empty capture.
|
||||||
|
//
|
||||||
|
// This is the R-358 lesson one tier over — the off-site scratch marker — arriving on Tier-2.
|
||||||
|
func tier2UnitIsOpenable(unitDir string) bool {
|
||||||
|
if fi, err := os.Stat(unitDir); err != nil || !fi.IsDir() {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return readManifest(UnitManifestFile(unitDir)) != nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat/read — no side effects.
|
||||||
func tier2CoverageAt(destBase string) Tier2Coverage {
|
func tier2CoverageAt(destBase string) Tier2Coverage {
|
||||||
var c Tier2Coverage
|
var c Tier2Coverage
|
||||||
for _, leg := range []string{"hdd", "userdata"} {
|
for _, leg := range []string{"hdd", "userdata"} {
|
||||||
@@ -65,9 +121,11 @@ func tier2CoverageAt(destBase string) Tier2Coverage {
|
|||||||
c.Legs = append(c.Legs, leg)
|
c.Legs = append(c.Legs, leg)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if fi, err := os.Stat(filepath.Join(destBase, "recovery-unit")); err == nil && fi.IsDir() {
|
unitDir := tier2UnitDir(destBase)
|
||||||
|
if fi, err := os.Stat(unitDir); err == nil && fi.IsDir() {
|
||||||
c.HasUnit = true
|
c.HasUnit = true
|
||||||
}
|
}
|
||||||
|
c.UnitRestorable = tier2UnitIsOpenable(unitDir)
|
||||||
return c
|
return c
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -79,7 +137,61 @@ func (m *Manager) Tier2RestoreCoverage(stackName string) (Tier2Coverage, error)
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return Tier2Coverage{}, err
|
return Tier2Coverage{}, err
|
||||||
}
|
}
|
||||||
return tier2CoverageAt(destBase), nil
|
cov := tier2CoverageAt(destBase)
|
||||||
|
// R-102: carry WHEN the copy was written, so the surface can name the date on an action that
|
||||||
|
// overwrites live data with it. Read from the same recorded config tier2RecordedCopyDir just
|
||||||
|
// resolved the path from, so the date and the directory cannot describe different runs.
|
||||||
|
if m.settings != nil {
|
||||||
|
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil {
|
||||||
|
cov.CopyLastRun, cov.CopyLastSuccess = cfg.LastRun, cfg.LastSuccess
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return cov, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// RestoreTier2Unit runs the FULL recovery-unit restore from the app's Tier-2 copy on the SECOND
|
||||||
|
// DRIVE — R-102, and the reason this task exists.
|
||||||
|
//
|
||||||
|
// Tier-2 has mirrored each app's whole recovery unit to
|
||||||
|
// <dest>/backups/secondary/<app>/recovery-unit/ on every run for months, and no code path read it:
|
||||||
|
// every reader of a unit could only name a path under backups/primary/. So in the exact failure
|
||||||
|
// Tier-2 exists for — the primary drive is lost, and the primary unit with it — the surviving copy
|
||||||
|
// was unopenable by any customer action (07-backup-architecture §6.3, §7.2).
|
||||||
|
//
|
||||||
|
// It is NOT the additive file restore beside it. This one OVERWRITES: named volumes are recreated
|
||||||
|
// from the mirror's tars and the database is replayed from the mirror's dump. The surface must carry
|
||||||
|
// that difference; see tier2UnitConfirm in the web package.
|
||||||
|
//
|
||||||
|
// THE SINGLE-WRITER FLAG IS TAKEN INSIDE RestoreFromRecoveryUnitAt, exactly as on the primary path —
|
||||||
|
// do NOT add an acquireRunning() here. A second acquire would refuse the restore it is guarding.
|
||||||
|
func (m *Manager) RestoreTier2Unit(stackName string) (UnitRestoreResult, error) {
|
||||||
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
||||||
|
if err != nil {
|
||||||
|
return UnitRestoreResult{}, err
|
||||||
|
}
|
||||||
|
unitDir := tier2UnitDir(destBase)
|
||||||
|
if !tier2UnitIsOpenable(unitDir) {
|
||||||
|
m.logger.Printf("[WARN] [backup] Tier-2 unit restore refused for %s: no openable recovery unit in the recorded copy — the app was NOT stopped", stackName)
|
||||||
|
return UnitRestoreResult{}, ErrTier2NoUnitInCopy
|
||||||
|
}
|
||||||
|
// WARN, not INFO: this is the destructive one of the two Tier-2 restores. The unit directory is a
|
||||||
|
// path and never a secret, and naming it is what makes "the SECONDARY mirror was the source" a
|
||||||
|
// positive observable in the log rather than an absence to be argued from.
|
||||||
|
m.logger.Printf("[WARN] [backup] Tier-2 UNIT restore for %s from the secondary mirror %s — this OVERWRITES live app data", stackName, unitDir)
|
||||||
|
return m.RestoreFromRecoveryUnitAt(stackName, unitDir)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Tier2CopyDate returns the date the surface should name for this app's Tier-2 copy, preferring the
|
||||||
|
// last SUCCESS over the last ATTEMPT (R-101: a timestamp that records "we tried" cannot answer "did
|
||||||
|
// it work"), and reports whether the returned value is a proven success.
|
||||||
|
//
|
||||||
|
// It exists so the confirm text and the outcome sentence cannot disagree about which copy is being
|
||||||
|
// restored: one resolver, two readers.
|
||||||
|
func (c Tier2Coverage) Tier2CopyDate() (date string, proven bool) {
|
||||||
|
if c.CopyLastSuccess != "" {
|
||||||
|
return c.CopyLastSuccess, true
|
||||||
|
}
|
||||||
|
return c.CopyLastRun, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every
|
// tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every
|
||||||
|
|||||||
Reference in New Issue
Block a user