Files
felhom-controller/controller/internal/backup/r403_rehydrate_test.go
T
admin 2358e561b7
gates / gates (push) Successful in 11s
R-403: a poorer copy must never delete a richer one
MEASURED FIRST, then fixed. On the shipped v0.229.0, on demo-hp, an app's Tier-2 copy went from
120 082 104 B (4 database dumps + 3 named-volume tars) to 7 036 B (none of either) in ONE nightly
run, and the run recorded itself a success: 'Tier 2 copied docmost -> ... (14.9 KB, 0 leg(s), 0s)'.
Evidence: felhom.eu/documentation/audits/DRILL-r403-tier2-delete-2026-08-31/.

The mechanism was three individually-correct lines: RunTier2 guards the unit leg with os.Stat only
(does the folder exist), rsyncMirror is rsync -a --delete, and nothing between them compared source
to destination. An EMPTY unit is a folder that exists.

THE GUARD. One predicate, unitCarriesData/unitIsHollow (r403_hollow.go), asking the MANIFEST and
never the byte size - a big compose tree with no dumps is dangerous, a tiny unit for a tiny app is
fine. Fail closed on an absent or unparseable manifest. RunTier2 skips the unit leg when the source
is hollow AND the destination is not; the other legs still run, the run is not failed, and the skip
is recorded for the SURFACE (CrossDriveBackup.UnitLegSkipped + UnitPackageDate) as well as logged.

--delete STAYS and shrinking stays legal. 07 section 8 row 5's derived-copy rule is unchanged; the
fence is exactly one shape. TestR403_DataLegShrinkIsUnaffected is the guard on the guard.

THE HONESTY. A preserved package is older than the run that preserved it, so the card carries a
notice and the unit-restore confirm names the PACKAGE's date - read from the mirrored manifest's own
created_at, not from the status record - plus a clause saying why it is older.

THE CAUSE. RestoreTier2Unit now refills a hollow or absent primary unit from the mirror it just
restored from, INSIDE the call before returning. The hollow manifest was written two seconds after
a restore by the 5-minute capture job; any follow-up job races it. The capture itself is NOT guarded:
a capture describing an empty drive as empty is correct, and with the primary refilled there is no
hollow state left to describe. Never over a complete primary, never after a failed restore.

recordTier2Success and tier2UnitConfirmMsg keep their old signatures as thin callers, so no existing
test needed editing. New seam unitRehydrate, separate from tier2Mirror on purpose.

22 new Go tests. Red-proofs run and reverted: A6 (predicate -> size threshold), B1 (guard removed ->
the copy's 3 files are DELETED and the seam is called), B6 (a general never-shrink rule -> the shrink
case fails), C2 (only-when-hollow dropped -> the complete primary is overwritten).
2026-08-31 14:02:13 +02:00

161 lines
6.4 KiB
Go

package backup
import (
"errors"
"os"
"path/filepath"
"testing"
)
// R-403 Group C — the rehydrate: after a Tier-2 unit restore, the app's own drive gets its package
// back, INSIDE the call.
//
// The cause half. On 2026-08-31 the hollow primary manifest was written TWO SECONDS after a restore
// of exactly this shape, by the 5-minute `backup-cache` job. Anything that runs after the call
// returns races that job; only doing it before returning cannot lose.
// C1 — a hollow primary is refilled from the mirror that was just restored from.
func TestR403_HollowPrimaryIsRefilledFromTheMirror(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar", "vol_b.tar"}, pgDump(1))
primaryUnit := RecoveryUnitPath(f.liveDrive, "app")
// The R-102 scenario: the app's own package is gone.
if err := os.RemoveAll(primaryUnit); err != nil {
t.Fatal(err)
}
if unitCarriesData(primaryUnit) {
t.Fatal("fixture wrong: the primary still carries data")
}
var copied [][2]string
f.m.unitRehydrate = func(src, dst string) error {
copied = append(copied, [2]string{src, dst})
return copyTree(src, dst)
}
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
t.Fatalf("restore: %v", err)
}
if len(copied) != 1 {
t.Fatalf("rehydrate ran %d time(s), want 1: %v", len(copied), copied)
}
mirrorUnit := tier2UnitDir(f.destBase)
if copied[0][0] != mirrorUnit || copied[0][1] != primaryUnit {
t.Errorf("rehydrate copied %v → %v, want %v → %v", copied[0][0], copied[0][1], mirrorUnit, primaryUnit)
}
// THE CONSEQUENCE: the primary is a real package again, so the next capture has nothing hollow to
// describe and the next Tier-2 run has nothing poorer to mirror.
if !unitCarriesData(primaryUnit) {
t.Error("the primary unit is still hollow after the restore — R-403's cause is not closed")
}
for _, rel := range []string{"volume-dumps/vol_a.tar", "volume-dumps/vol_b.tar", "db-dumps/app-postgres.sql"} {
if _, err := os.Stat(filepath.Join(primaryUnit, rel)); err != nil {
t.Errorf("%s did not come back to the app's own drive: %v", rel, err)
}
}
}
// C2 — TestR403_CompletePrimaryIsLeftByteIdentical. Scenario F, and it is R-403 pointed the other
// way: a primary that already carries data may be NEWER than the mirror, and overwriting it with an
// older copy is the same defect this task exists to remove.
//
// Red-proof (recorded in REPORT.md): drop the `unitCarriesData(primaryUnit)` condition and this
// fails on the fingerprint.
func TestR403_CompletePrimaryIsLeftByteIdentical(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
primaryUnit := RecoveryUnitPath(f.liveDrive, "app")
if !unitCarriesData(primaryUnit) {
t.Fatal("fixture wrong: the primary must start complete")
}
// Make the primary distinguishable from the mirror, so an overwrite would show.
mustWrite(t, filepath.Join(primaryUnit, "volume-dumps", "only-on-the-primary.tar"), "NEWER")
before := fingerprintTree(t, primaryUnit)
var ran int
f.m.unitRehydrate = func(src, dst string) error { ran++; return copyTree(src, dst) }
if _, err := f.m.RestoreTier2Unit("app"); err != nil {
t.Fatalf("restore: %v", err)
}
if ran != 0 {
t.Errorf("the rehydrate ran %d time(s) over a COMPLETE primary — it would overwrite newer material", ran)
}
if after := fingerprintTree(t, primaryUnit); after != before {
t.Error("the complete primary unit was modified; it must be byte-identical")
}
if _, err := os.Stat(filepath.Join(primaryUnit, "volume-dumps", "only-on-the-primary.tar")); err != nil {
t.Error("the primary-only tar was destroyed by the rehydrate")
}
}
// C3 — a FAILED restore writes no package. A package written from a run that did not succeed would
// look like a backup and describe data that never landed.
func TestR403_FailedRestoreDoesNotWriteAPackage(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
primaryUnit := RecoveryUnitPath(f.liveDrive, "app")
if err := os.RemoveAll(primaryUnit); err != nil {
t.Fatal(err)
}
// Make the restore itself fail, after it has started.
f.fake.startSvcErr = errors.New("injected: the database service would not start")
f.m.volumeReplayFrom = func(string, string) (int, error) {
return 0, errors.New("injected: the volume replay failed")
}
var ran int
f.m.unitRehydrate = func(src, dst string) error { ran++; return copyTree(src, dst) }
if _, err := f.m.RestoreTier2Unit("app"); err == nil {
t.Fatal("the fixture did not make the restore fail")
}
if ran != 0 {
t.Errorf("the rehydrate ran %d time(s) after a FAILED restore", ran)
}
if unitCarriesData(primaryUnit) {
t.Error("a package was written on the app's drive by a restore that failed")
}
}
// C4 — TestR403_RehydrateHappensBeforeTheCallReturns. Asserted as ORDERING, never as a timer.
//
// The whole requirement is that the 5-minute capture job cannot observe the hollow state. A test
// that slept and then looked would pass on a racing implementation on a fast machine.
func TestR403_RehydrateHappensBeforeTheCallReturns(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
primaryUnit := RecoveryUnitPath(f.liveDrive, "app")
if err := os.RemoveAll(primaryUnit); err != nil {
t.Fatal(err)
}
var rehydrated bool
f.m.unitRehydrate = func(src, dst string) error { rehydrated = true; return copyTree(src, dst) }
// The observation is taken on the line AFTER the call returns, with nothing waited for. If the
// implementation deferred the work to a goroutine or a scheduled job, this reads false.
_, err := f.m.RestoreTier2Unit("app")
if err != nil {
t.Fatalf("restore: %v", err)
}
if !rehydrated {
t.Fatal("the rehydrate had NOT run when the call returned — a follow-up job races the capture (R-403)")
}
if !unitCarriesData(primaryUnit) {
t.Error("the primary was still hollow at the instant the call returned")
}
}
// A rehydrate failure must not turn a successful restore into a reported failure — the app is back.
func TestR403_RehydrateFailureDoesNotFailTheRestore(t *testing.T) {
f := r102Tier2Fixture(t, []string{"vol_a.tar"}, pgDump(1))
if err := os.RemoveAll(RecoveryUnitPath(f.liveDrive, "app")); err != nil {
t.Fatal(err)
}
f.m.unitRehydrate = func(string, string) error { return errors.New("injected: disk full") }
res, err := f.m.RestoreTier2Unit("app")
if err != nil {
t.Fatalf("a failed rehydrate must not fail the restore: %v", err)
}
if res.VolumesReplayed != 1 {
t.Errorf("the restore's own result was disturbed: %+v", res)
}
}