Files
felhom-controller/controller/internal/backup/r403_mirror_guard_test.go
T
admin 2358e561b7
gates / gates (push) Successful in 11s
R-403: a poorer copy must never delete a richer one
MEASURED FIRST, then fixed. On the shipped v0.229.0, on demo-hp, an app's Tier-2 copy went from
120 082 104 B (4 database dumps + 3 named-volume tars) to 7 036 B (none of either) in ONE nightly
run, and the run recorded itself a success: 'Tier 2 copied docmost -> ... (14.9 KB, 0 leg(s), 0s)'.
Evidence: felhom.eu/documentation/audits/DRILL-r403-tier2-delete-2026-08-31/.

The mechanism was three individually-correct lines: RunTier2 guards the unit leg with os.Stat only
(does the folder exist), rsyncMirror is rsync -a --delete, and nothing between them compared source
to destination. An EMPTY unit is a folder that exists.

THE GUARD. One predicate, unitCarriesData/unitIsHollow (r403_hollow.go), asking the MANIFEST and
never the byte size - a big compose tree with no dumps is dangerous, a tiny unit for a tiny app is
fine. Fail closed on an absent or unparseable manifest. RunTier2 skips the unit leg when the source
is hollow AND the destination is not; the other legs still run, the run is not failed, and the skip
is recorded for the SURFACE (CrossDriveBackup.UnitLegSkipped + UnitPackageDate) as well as logged.

--delete STAYS and shrinking stays legal. 07 section 8 row 5's derived-copy rule is unchanged; the
fence is exactly one shape. TestR403_DataLegShrinkIsUnaffected is the guard on the guard.

THE HONESTY. A preserved package is older than the run that preserved it, so the card carries a
notice and the unit-restore confirm names the PACKAGE's date - read from the mirrored manifest's own
created_at, not from the status record - plus a clause saying why it is older.

THE CAUSE. RestoreTier2Unit now refills a hollow or absent primary unit from the mirror it just
restored from, INSIDE the call before returning. The hollow manifest was written two seconds after
a restore by the 5-minute capture job; any follow-up job races it. The capture itself is NOT guarded:
a capture describing an empty drive as empty is correct, and with the primary refilled there is no
hollow state left to describe. Never over a complete primary, never after a failed restore.

recordTier2Success and tier2UnitConfirmMsg keep their old signatures as thin callers, so no existing
test needed editing. New seam unitRehydrate, separate from tier2Mirror on purpose.

22 new Go tests. Red-proofs run and reverted: A6 (predicate -> size threshold), B1 (guard removed ->
the copy's 3 files are DELETED and the seam is called), B6 (a general never-shrink rule -> the shrink
case fails), C2 (only-when-hollow dropped -> the complete primary is overwritten).
2026-08-31 14:02:13 +02:00

271 lines
11 KiB
Go

package backup
import (
"os"
"path/filepath"
"strings"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
)
// R-403 Group B — the mirror guard, driven through the REAL RunTier2.
//
// The `tier2Mirror` seam records every (src,dst) the run asks for, so a skip is provable as an
// ABSENCE OF A CALL rather than inferred from a log line — and the destination's own bytes are
// compared before and after, which is the consequence the customer actually cares about.
// r403Recorder is a mirror seam that records its calls and then performs a REAL MIRROR — the
// destination is emptied first, so `rsync -a --delete`'s defining behaviour is present in the test.
//
// This matters more than it looks: `copyTree` (the additive helper next door) would let every
// assertion here pass while proving nothing, because the deletion IS the defect. A seam that is
// gentler than the thing it stands in for is a test that cannot see the bug.
type r403Recorder struct{ calls []string }
func (r *r403Recorder) mirror(src, dst string) error {
r.calls = append(r.calls, dst)
if err := os.RemoveAll(dst); err != nil {
return err
}
return copyTree(src, dst)
}
func (r *r403Recorder) calledFor(sub string) bool {
for _, c := range r.calls {
if strings.Contains(c, sub) {
return true
}
}
return false
}
// r403Fixture builds a Manager with a source drive and a registered off-drive target, seeds the
// SOURCE unit and (optionally) a pre-existing DESTINATION unit, and returns the recorder.
func r403Fixture(t *testing.T, srcDB, srcVols []string, destDB, destVols []string, destExists bool) (
m *Manager, rec *r403Recorder, src, destBase string) {
t.Helper()
m, src, target, prov := newTier2V2(t, "app")
rec = &r403Recorder{}
m.tier2Mirror = rec.mirror
prov.has["app"] = false // legacy path, no classified binds → the unit leg is the only leg
destBase = filepath.Join(target, "backups", "secondary", "app")
// SOURCE unit — newTier2V2 already wrote a `{}` manifest; replace it with a real one.
srcUnit := RecoveryUnitPath(src, "app")
man := &RecoveryManifest{SchemaVersion: 2, AppName: "app", DBDumps: srcDB, VolumeDumps: srcVols}
if err := writeManifest(UnitManifestFile(srcUnit), man); err != nil {
t.Fatal(err)
}
for _, d := range srcDB {
mustWrite(t, filepath.Join(UnitDBDumpDir(srcUnit), d), "SOURCE-"+d)
}
for _, v := range srcVols {
mustWrite(t, filepath.Join(UnitVolumeDumpDir(srcUnit), v), "SOURCE-"+v)
}
// DESTINATION unit — the copy that must survive.
if destExists {
destUnit := filepath.Join(destBase, "recovery-unit")
// atomicWrite needs the parent to exist — the real capture creates it before writing.
if err := os.MkdirAll(destUnit, 0o755); err != nil {
t.Fatal(err)
}
dman := &RecoveryManifest{SchemaVersion: 2, AppName: "app", DBDumps: destDB, VolumeDumps: destVols}
if err := writeManifest(UnitManifestFile(destUnit), dman); err != nil {
t.Fatal(err)
}
for _, d := range destDB {
mustWrite(t, filepath.Join(UnitDBDumpDir(destUnit), d), "DEST-"+d)
}
for _, v := range destVols {
mustWrite(t, filepath.Join(UnitVolumeDumpDir(destUnit), v), "DEST-"+v)
}
mustWrite(t, filepath.Join(destBase, tier2LayoutMarker), tier2LayoutVersion)
}
return m, rec, src, destBase
}
// B1 — TestR403_HollowSourceOverCompleteDestIsSkipped. THE ACCEPTANCE TEST.
//
// This is the state measured live on demo-hp 2026-08-31: a hollow primary unit and a complete
// secondary copy. On the shipped v0.229.0 the run deleted 4 database dumps and 3 volume tars —
// 120 082 104 B → 7 036 B — and reported success.
//
// Red-proof (recorded in REPORT.md): remove the guard and this fails on BOTH assertions — the seam
// is called for the unit leg, and the destination's fingerprint changes.
func TestR403_HollowSourceOverCompleteDestIsSkipped(t *testing.T) {
m, rec, _, destBase := r403Fixture(t, nil, nil,
[]string{"app-postgres.sql"}, []string{"app_data.tar", "app_cache.tar"}, true)
destUnit := filepath.Join(destBase, "recovery-unit")
before := fingerprintTree(t, destUnit)
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2 must still succeed — the other legs are not held hostage: %v", err)
}
// THE CONSEQUENCE, first: the customer's package is byte-identical.
if after := fingerprintTree(t, destUnit); after != before {
t.Error("the destination unit CHANGED — a hollow source was mirrored over a complete copy (R-403)")
}
for _, f := range []string{"db-dumps/app-postgres.sql", "volume-dumps/app_data.tar", "volume-dumps/app_cache.tar"} {
if _, err := os.Stat(filepath.Join(destUnit, f)); err != nil {
t.Errorf("%s was DELETED from the copy: %v", f, err)
}
}
// And the mechanism: the mirror was never asked to touch the unit leg.
if rec.calledFor("recovery-unit") {
t.Errorf("the mirror seam WAS called for the unit leg: %v", rec.calls)
}
}
// B2 — a complete source still mirrors. The guard must not become a general refusal.
func TestR403_CompleteSourceStillMirrors(t *testing.T) {
m, rec, _, destBase := r403Fixture(t, []string{"app-postgres.sql"}, []string{"app_data.tar"},
[]string{"old.sql"}, []string{"old.tar"}, true)
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2: %v", err)
}
if !rec.calledFor("recovery-unit") {
t.Fatalf("the unit leg did not run for a complete source: %v", rec.calls)
}
destUnit := filepath.Join(destBase, "recovery-unit")
if _, err := os.Stat(filepath.Join(destUnit, "volume-dumps/app_data.tar")); err != nil {
t.Errorf("the source's tar did not reach the copy: %v", err)
}
// --delete still works: the stale destination file is gone. The mirror is still a MIRROR.
if _, err := os.Stat(filepath.Join(destUnit, "volume-dumps/old.tar")); err == nil {
t.Error("the stale destination tar survived — --delete was neutered, which is NOT the fix")
}
}
// B3 — hollow over hollow still mirrors. An app that genuinely has no database and no volumes is not
// a defect, and both sides agree, so nothing is at risk.
func TestR403_HollowOverHollowStillMirrors(t *testing.T) {
m, rec, _, _ := r403Fixture(t, nil, nil, nil, nil, true)
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2: %v", err)
}
if !rec.calledFor("recovery-unit") {
t.Errorf("hollow→hollow was skipped; it must mirror: %v", rec.calls)
}
}
// B3b — a first copy (no destination unit at all) still mirrors, hollow source or not.
func TestR403_FirstCopyStillMirrors(t *testing.T) {
m, rec, _, _ := r403Fixture(t, nil, nil, nil, nil, false)
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2: %v", err)
}
if !rec.calledFor("recovery-unit") {
t.Errorf("the first copy was skipped: %v", rec.calls)
}
}
// B4 — the other legs still run when the unit leg is skipped. A preserved package must not cost the
// customer their file legs.
func TestR403_OtherLegsStillRunWhenTheUnitLegIsSkipped(t *testing.T) {
m, rec, src, _ := r403Fixture(t, nil, nil, []string{"a.sql"}, []string{"a.tar"}, true)
// Give the app a classified file leg with real content.
prov := m.stackProvider.(*t2v2Provider)
prov.has["app"] = true
prov.binds["app"] = []ClassifiedBind{mHDD("appdata/app")}
mustWrite(t, filepath.Join(src, "appdata", "app", "user.txt"), "USERDATA")
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2: %v", err)
}
if rec.calledFor("recovery-unit") {
t.Error("the unit leg ran despite a hollow source over a complete destination")
}
if !rec.calledFor(filepath.Join("hdd", "appdata", "app")) {
t.Errorf("the FILE leg was held hostage by the skipped unit leg: %v", rec.calls)
}
}
// B5 — TestR403_SkipIsRecordedForTheSurface. Not only logged.
//
// A refusal that lives only in a container log is a refusal the customer cannot see, and the whole
// point of preserving the copy is lost if the page then calls it fresh.
func TestR403_SkipIsRecordedForTheSurface(t *testing.T) {
m, _, _, destBase := r403Fixture(t, nil, nil, []string{"a.sql"}, []string{"a.tar"}, true)
// Stamp the destination manifest with a date so the surface has a package date to name.
destUnit := filepath.Join(destBase, "recovery-unit")
dman := &RecoveryManifest{SchemaVersion: 2, AppName: "app", CreatedAt: "2026-08-25T03:30:00Z",
DBDumps: []string{"a.sql"}, VolumeDumps: []string{"a.tar"}}
if err := os.MkdirAll(destUnit, 0o755); err != nil {
t.Fatal(err)
}
if err := writeManifest(UnitManifestFile(destUnit), dman); err != nil {
t.Fatal(err)
}
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2: %v", err)
}
cd := m.settings.GetCrossDriveConfig("app")
if cd == nil {
t.Fatal("no cross-drive record was written at all")
}
if !cd.UnitLegSkipped {
t.Error("UnitLegSkipped is false — the surface cannot tell the package was preserved")
}
if cd.UnitPackageDate != "2026-08-25T03:30:00Z" {
t.Errorf("UnitPackageDate = %q, want the DESTINATION manifest's own created_at", cd.UnitPackageDate)
}
if !strings.Contains(cd.LastWarning, "hi") { // ASCII fragment of "hiányos" (R-364)
t.Errorf("LastWarning does not carry the preserved-copy notice: %q", cd.LastWarning)
}
if cd.LastWarning != tier2UnitPreservedWarning &&
!strings.Contains(cd.LastWarning, tier2UnitPreservedWarning) {
t.Errorf("LastWarning is not the named constant: %q", cd.LastWarning)
}
// And the coverage the restore surface reads agrees, from the ARTIFACT rather than the record.
cov, err := m.Tier2RestoreCoverage("app")
if err != nil {
t.Fatalf("coverage: %v", err)
}
date, older := cov.UnitRestoreDate()
if date != "2026-08-25T03:30:00Z" || !older {
t.Errorf("UnitRestoreDate() = (%q,%v), want the package date and older-than-the-run", date, older)
}
}
// B6 — TestR403_DataLegShrinkIsUnaffected. The guard is on the UNIT leg only.
//
// `07-backup-architecture.md` §8 row 5 records that the secondary is a derived copy, rebuilt on the
// next run, and tier2.go's header records that a classified app's copy legitimately shrinks as
// `export` drops out of its class set. Fencing that would be calling a decision a defect.
//
// Red-proof (recorded in REPORT.md): widen the guard to the data legs and this fails — the stale file
// survives in the copy.
func TestR403_DataLegShrinkIsUnaffected(t *testing.T) {
m, _, src, destBase := r403Fixture(t, []string{"a.sql"}, []string{"a.tar"}, nil, nil, true)
prov := m.stackProvider.(*t2v2Provider)
prov.has["app"] = true
prov.binds["app"] = []ClassifiedBind{mHDD("appdata/app")}
mustWrite(t, filepath.Join(src, "appdata", "app", "kept.txt"), "KEPT")
// A file that exists ONLY in the destination — the shrink case.
mustWrite(t, filepath.Join(destBase, "hdd", "appdata", "app", "dropped.txt"), "SHOULD-GO")
if err := m.RunTier2("app"); err != nil {
t.Fatalf("RunTier2: %v", err)
}
if _, err := os.Stat(filepath.Join(destBase, "hdd", "appdata", "app", "kept.txt")); err != nil {
t.Errorf("the live file did not reach the copy: %v", err)
}
if _, err := os.Stat(filepath.Join(destBase, "hdd", "appdata", "app", "dropped.txt")); err == nil {
t.Error("a data leg stopped shrinking — the guard is TOO WIDE and is fencing a design decision")
}
}
// A guard that quietly widened would also be caught by the class constant staying where it belongs.
func TestR403_GuardUsesTheSharedPredicate(t *testing.T) {
// The predicate answers the same for the same directory whichever caller asks — one definition.
dir := r403Unit(t, nil, []string{"v.tar"}, nil)
if unitIsHollow(dir) != !unitCarriesData(dir) {
t.Error("unitIsHollow is not the negation of unitCarriesData")
}
_ = appbackup.UnitManifestFile // the unit-dir-relative helper is what both callers resolve through
}