Files
felhom-controller/controller/internal/backup/r355_safetydump_test.go
T
admin 2c724c9283
gates / gates (push) Successful in 11s
R-379/R-380: put the customer's undo copy back when a database restore fails
R-379 and R-380 were one failure. Both ended with a half-restored database; the
only difference was whether it looked broken. Postgres emptied and crash-looped;
MariaDB applied part of the dump and reported health=healthy with a zero-row
schema-version table. Measured live on demo-hp 2026-08-22.

The undo copy was already taken and already good - proven by hand that day on
both engines. Nothing in the product could apply it. Now it does, with the same
ImportDump call, before any restart and inside the DB-only window.

The WHOLE undo set, matched on this run's stamp. writeSafetyDump returned one
path for an app with two databases; a rollback on that would restore one and
leave the other half-written.

When the rollback also fails the app is HELD STOPPED (operator ruling): a running
app on a half-written database lets the customer make the damage permanent. Every
start path refuses it - customer button, appstop Recover, boot sweep - via the
shared driveStartGate, checked ABOVE its driveless early return because these
apps have no drive. The marker is ended so nothing auto-restarts it. The row goes
red. Cleared with --clear-restore-hold, an operator CLI route.

--single-transaction is a belt on Postgres only; MariaDB DDL is not transactional
and that is why the rollback is the fix.

R-381: the engine's stderr stops reaching the customer (615 bytes on MariaDB, its
middle rows out of their own database) and starts reaching the operator log,
which never had it.
R-382: the summary log prints the volume count it already held.
Undo copies resolve to their own app, are marked IsUndo, and are capped at 3 per
app, pruned from the capture side. The reported render-as-an-app symptom did NOT
reproduce - the live page was read first and had zero occurrences.

Tests 1468 -> 1483. Eight red-proofs; ONE PASSED and is reported: the R-381
behavioural test injected below ImportDump. A guard at that layer now convicts.
2026-08-22 18:03:18 +02:00

138 lines
5.9 KiB
Go

package backup
import (
"context"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
)
// R-355's second half. The naming defect does not stop at the backup: writeSafetyDump filters the
// discovered databases with `db.StackName == stackName`, so an app whose database is attributed to the
// wrong stack has NO database as far as the destructive restore is concerned. It therefore takes no
// undo copy, and the fail-closed refusal that protects every other app cannot fire — the guard is not
// bypassed, it is never reached.
//
// Measured live on 2026-08-21: a destructive restore of `paperless-ngx` ran to completion over a live
// 72-table PostgreSQL with `find /mnt -name "pre-restore-*"` empty both before and after.
func newSafetyTestManager() *Manager {
return &Manager{logger: log.New(io.Discard, "", 0)}
}
// TestR355_SafetyDumpIsTakenForTheCorrectlyAttributedApp is scenario B: the undo copy exists.
func TestR355_SafetyDumpIsTakenForTheCorrectlyAttributedApp(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
// Post-fix attribution: the compose project label resolves this container to `paperless-ngx`.
m.discoverDBs = func(ctx context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{
{StackName: "paperless-ngx", DBType: DBTypePostgres, ContainerName: "paperless-postgres", ContainerID: "cid"},
}, nil
}
m.safetyDumpFn = func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
p := filepath.Join(dumpDir, string(db.StackName)+"-"+string(db.DBType)+".sql")
if err := os.MkdirAll(dumpDir, 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(p, []byte("-- 72 tables\n"), 0o644); err != nil {
t.Fatal(err)
}
return DumpResult{DB: db, FilePath: p, Size: 13}
}
// v0.220.0 (R-379): writeSafetyDump returns the SET it wrote, so a rollback can re-apply EVERY
// database's undo. `.First()` is the value this signature returned before; these assertions are
// unchanged in meaning.
set, err := m.writeSafetyDump(context.Background(), "paperless-ngx", nsRoot)
safety := set.First()
if err != nil {
t.Fatalf("writeSafetyDump: %v", err)
}
if safety == "" {
t.Fatal("no safety dump was taken for an app that HAS a database — the restore would proceed with no undo")
}
if _, err := os.Stat(safety); err != nil {
t.Fatalf("the safety dump path %q is not on disk: %v", safety, err)
}
if !strings.Contains(filepath.Base(safety), "pre-restore-") {
t.Errorf("the undo copy must carry the pre-restore prefix so it can never be replayed as a source; got %q", filepath.Base(safety))
}
// The consequence that matters: it lives inside THIS app's unit, not a phantom's.
if got, want := filepath.Dir(safety), AppDBDumpPath(nsRoot, "paperless-ngx"); got != want {
t.Errorf("undo copy written to %q, want %q", got, want)
}
}
// TestR355_MisattributedAppGetsNoUndoCopy demonstrates the WRONG OUTCOME — the state the fix removes.
// It models the pre-fix attribution (`paperless`) against a restore of `paperless-ngx` and asserts the
// undo silently does not happen. This is the shape that made a destructive restore unrecoverable.
func TestR355_MisattributedAppGetsNoUndoCopy(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
// PRE-FIX attribution: deriveStackName gave `paperless` for container `paperless-postgres`.
m.discoverDBs = func(ctx context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{
{StackName: "paperless", DBType: DBTypePostgres, ContainerName: "paperless-postgres", ContainerID: "cid"},
}, nil
}
called := false
m.safetyDumpFn = func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
called = true
return DumpResult{DB: db}
}
// v0.220.0 (R-379): writeSafetyDump returns the SET it wrote, so a rollback can re-apply EVERY
// database's undo. `.First()` is the value this signature returned before; these assertions are
// unchanged in meaning.
set, err := m.writeSafetyDump(context.Background(), "paperless-ngx", nsRoot)
safety := set.First()
if err != nil {
t.Fatalf("writeSafetyDump: %v", err)
}
if safety != "" || called {
t.Fatalf("precondition lost: the misattributed shape now takes an undo copy (safety=%q called=%v) — "+
"this test documents the defect and must keep failing to find one", safety, called)
}
// And this is precisely why it was invisible: no error, no dump, and the caller reads
// `hasDB == false` — indistinguishable from an app that genuinely has no database.
}
// TestR355_RestoreRefusesWhenTheUndoCannotBeTaken is scenario C, the fail-closed direction. The
// invariant already existed and was proven working live on 2026-08-21 for `romm`; this pins it for the
// app that could not reach it before, so the two cannot drift apart.
func TestR355_RestoreRefusesWhenTheUndoCannotBeTaken(t *testing.T) {
nsRoot := t.TempDir()
m := newSafetyTestManager()
m.discoverDBs = func(ctx context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{
{StackName: "paperless-ngx", DBType: DBTypePostgres, ContainerName: "paperless-postgres", ContainerID: "cid"},
}, nil
}
m.safetyDumpFn = func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
return DumpResult{DB: db, Error: os.ErrPermission}
}
// v0.220.0 (R-379): writeSafetyDump returns the SET it wrote, so a rollback can re-apply EVERY
// database's undo. `.First()` is the value this signature returned before; these assertions are
// unchanged in meaning.
set, err := m.writeSafetyDump(context.Background(), "paperless-ngx", nsRoot)
safety := set.First()
if err == nil {
t.Fatal("a database that cannot be dumped must be a hard error — the undo would not exist")
}
if safety != "" {
t.Errorf("a failed undo must return no path, got %q", safety)
}
// The message must say the restore did not start, because that is the customer's only signal.
if !strings.Contains(err.Error(), "nem indult el") {
t.Errorf("the refusal must state that the restore did not start; got %q", err.Error())
}
}