Files
felhom-controller/controller/internal/backup/r355_safetydump_test.go
T
admin 5ce3a44645
gates / gates (push) Successful in 11s
v0.218.0: attribute a DB container by its compose project, and replay volumes on the off-site restore
R-355 (first, because it is the only one where data can be lost for good). paperless-ngx's
PostgreSQL was dumped into backups/primary/paperless/db-dumps/ — a directory for a stack that
does not exist, on the system drive — while the app's own unit recorded db_dumps: null. The
same misattribution reached writeSafetyDump, so a destructive restore of that app took NO undo
copy and the fail-closed refusal was never reached. Fixed by reading the compose project label,
which is the stack name by construction (compose runs with cmd.Dir set to the stack dir and no
-p). The old derivation stays as the fallback and an unresolvable attribution is now loud.
Catalogue sweep, proven able to convict: one affected app of 53. The fix is in the controller,
not the catalogue.

R-354. ReconstituteFromOffsite skipped every unit placement and the volume archives live inside
the unit, so the off-site restore had no volume leg at all — proven live with planted files:
calibre-web's 1,422,848-byte config archive was in the unit, the snapshot and the checking
folder, and the restore reported success without it. For the 40 of 53 apps that declare no data
drive that archive is the whole dataset. restoreDockerVolumesFrom is the local path's own replay
with an explicit directory: ONE implementation, two callers. Volumes replay before the database
and inside the stopped window. VolumesReplayed reaches the message.

The comment beside the skip was half false and is corrected; the half that still holds — the
live unit is the local path's source — is named, and scenario D fingerprints the whole live unit
across the operation.

Seven red-proofs, each asserted applied and reverted. Two found defects in the tests, not the
code: scenario D passed with the unit guard removed because the fingerprint had been narrowed
and was blind to the unit root.
2026-08-22 09:43:22 +02:00

126 lines
5.2 KiB
Go

package backup
import (
"context"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
)
// R-355's second half. The naming defect does not stop at the backup: writeSafetyDump filters the
// discovered databases with `db.StackName == stackName`, so an app whose database is attributed to the
// wrong stack has NO database as far as the destructive restore is concerned. It therefore takes no
// undo copy, and the fail-closed refusal that protects every other app cannot fire — the guard is not
// bypassed, it is never reached.
//
// Measured live on 2026-08-21: a destructive restore of `paperless-ngx` ran to completion over a live
// 72-table PostgreSQL with `find /mnt -name "pre-restore-*"` empty both before and after.
func newSafetyTestManager() *Manager {
return &Manager{logger: log.New(io.Discard, "", 0)}
}
// TestR355_SafetyDumpIsTakenForTheCorrectlyAttributedApp is scenario B: the undo copy exists.
func TestR355_SafetyDumpIsTakenForTheCorrectlyAttributedApp(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
// Post-fix attribution: the compose project label resolves this container to `paperless-ngx`.
m.discoverDBs = func(ctx context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{
{StackName: "paperless-ngx", DBType: DBTypePostgres, ContainerName: "paperless-postgres", ContainerID: "cid"},
}, nil
}
m.safetyDumpFn = func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
p := filepath.Join(dumpDir, string(db.StackName)+"-"+string(db.DBType)+".sql")
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(p, []byte("-- 72 tables\n"), 0o644); err != nil {
t.Fatal(err)
}
return DumpResult{DB: db, FilePath: p, Size: 13}
}
safety, err := m.writeSafetyDump(context.Background(), "paperless-ngx", nsRoot)
if err != nil {
t.Fatalf("writeSafetyDump: %v", err)
}
if safety == "" {
t.Fatal("no safety dump was taken for an app that HAS a database — the restore would proceed with no undo")
}
if _, err := os.Stat(safety); err != nil {
t.Fatalf("the safety dump path %q is not on disk: %v", safety, err)
}
if !strings.Contains(filepath.Base(safety), "pre-restore-") {
t.Errorf("the undo copy must carry the pre-restore prefix so it can never be replayed as a source; got %q", filepath.Base(safety))
}
// The consequence that matters: it lives inside THIS app's unit, not a phantom's.
if got, want := filepath.Dir(safety), AppDBDumpPath(nsRoot, "paperless-ngx"); got != want {
t.Errorf("undo copy written to %q, want %q", got, want)
}
}
// TestR355_MisattributedAppGetsNoUndoCopy demonstrates the WRONG OUTCOME — the state the fix removes.
// It models the pre-fix attribution (`paperless`) against a restore of `paperless-ngx` and asserts the
// undo silently does not happen. This is the shape that made a destructive restore unrecoverable.
func TestR355_MisattributedAppGetsNoUndoCopy(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
// PRE-FIX attribution: deriveStackName gave `paperless` for container `paperless-postgres`.
m.discoverDBs = func(ctx context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{
{StackName: "paperless", DBType: DBTypePostgres, ContainerName: "paperless-postgres", ContainerID: "cid"},
}, nil
}
called := false
m.safetyDumpFn = func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
called = true
return DumpResult{DB: db}
}
safety, err := m.writeSafetyDump(context.Background(), "paperless-ngx", nsRoot)
if err != nil {
t.Fatalf("writeSafetyDump: %v", err)
}
if safety != "" || called {
t.Fatalf("precondition lost: the misattributed shape now takes an undo copy (safety=%q called=%v) — "+
"this test documents the defect and must keep failing to find one", safety, called)
}
// And this is precisely why it was invisible: no error, no dump, and the caller reads
// `hasDB == false` — indistinguishable from an app that genuinely has no database.
}
// TestR355_RestoreRefusesWhenTheUndoCannotBeTaken is scenario C, the fail-closed direction. The
// invariant already existed and was proven working live on 2026-08-21 for `romm`; this pins it for the
// app that could not reach it before, so the two cannot drift apart.
func TestR355_RestoreRefusesWhenTheUndoCannotBeTaken(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
m.discoverDBs = func(ctx context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{
{StackName: "paperless-ngx", DBType: DBTypePostgres, ContainerName: "paperless-postgres", ContainerID: "cid"},
}, nil
}
m.safetyDumpFn = func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
return DumpResult{DB: db, Error: os.ErrPermission}
}
safety, err := m.writeSafetyDump(context.Background(), "paperless-ngx", nsRoot)
if err == nil {
t.Fatal("a database that cannot be dumped must be a hard error — the undo would not exist")
}
if safety != "" {
t.Errorf("a failed undo must return no path, got %q", safety)
}
// The message must say the restore did not start, because that is the customer's only signal.
if !strings.Contains(err.Error(), "nem indult el") {
t.Errorf("the refusal must state that the restore did not start; got %q", err.Error())
}
}