2c724c9283
gates / gates (push) Successful in 11s
R-379 and R-380 were one failure. Both ended with a half-restored database; the only difference was whether it looked broken. Postgres emptied and crash-looped; MariaDB applied part of the dump and reported health=healthy with a zero-row schema-version table. Measured live on demo-hp 2026-08-22. The undo copy was already taken and already good - proven by hand that day on both engines. Nothing in the product could apply it. Now it does, with the same ImportDump call, before any restart and inside the DB-only window. The WHOLE undo set, matched on this run's stamp. writeSafetyDump returned one path for an app with two databases; a rollback on that would restore one and leave the other half-written. When the rollback also fails the app is HELD STOPPED (operator ruling): a running app on a half-written database lets the customer make the damage permanent. Every start path refuses it - customer button, appstop Recover, boot sweep - via the shared driveStartGate, checked ABOVE its driveless early return because these apps have no drive. The marker is ended so nothing auto-restarts it. The row goes red. Cleared with --clear-restore-hold, an operator CLI route. --single-transaction is a belt on Postgres only; MariaDB DDL is not transactional and that is why the rollback is the fix. R-381: the engine's stderr stops reaching the customer (615 bytes on MariaDB, its middle rows out of their own database) and starts reaching the operator log, which never had it. R-382: the summary log prints the volume count it already held. Undo copies resolve to their own app, are marked IsUndo, and are capped at 3 per app, pruned from the capture side. The reported render-as-an-app symptom did NOT reproduce - the live page was read first and had zero occurrences. Tests 1468 -> 1483. Eight red-proofs; ONE PASSED and is reported: the R-381 behavioural test injected below ImportDump. A guard at that layer now convicts.
952 lines
33 KiB
Go
952 lines
33 KiB
Go
package appbackup
|
|
|
|
import (
|
|
"bufio"
|
|
"compress/gzip"
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"log"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
|
)
|
|
|
|
// DBType represents a database engine type.
|
|
type DBType string
|
|
|
|
const (
|
|
DBTypePostgres DBType = "postgres"
|
|
DBTypeMariaDB DBType = "mariadb"
|
|
)
|
|
|
|
// DiscoveredDB holds metadata about a running database container.
|
|
type DiscoveredDB struct {
|
|
ContainerName string
|
|
ContainerID string
|
|
DBType DBType
|
|
DBUser string
|
|
DBName string
|
|
StackName string
|
|
}
|
|
|
|
// DumpResult holds the outcome of a single database dump.
|
|
type DumpResult struct {
|
|
DB DiscoveredDB
|
|
FilePath string
|
|
Size int64
|
|
Duration time.Duration
|
|
Error error
|
|
Validation DumpValidation
|
|
}
|
|
|
|
// DumpValidation holds the result of a dump file structural check.
|
|
type DumpValidation struct {
|
|
Valid bool
|
|
TableCount int
|
|
Error string
|
|
FileSize int64
|
|
ModTime time.Time
|
|
// R-44 (v0.148.0) content sniff — a WARN-LEVEL signal, never a gate.
|
|
//
|
|
// Structural validity says nothing about whether a dump holds the customer's data. The immich
|
|
// dump of 2026-07-19 was 52MB, had a valid header and 60+ CREATE TABLEs, and contained zero
|
|
// users and zero assets: its whole bulk was the geodata reference tables immich ships. Size and
|
|
// table count are therefore both useless as emptiness heuristics — but an accounts table with
|
|
// no rows is a strong, cheap, app-agnostic hint that a dump predates the customer entirely.
|
|
//
|
|
// Deliberately NOT a refusal: plenty of legitimate apps have no users table (UserTableFound
|
|
// false → inconclusive → silent), and a false positive that blocked a restore would be far
|
|
// worse than the skew it guards against. The restore confirm shows it as one extra line.
|
|
UserTableFound bool
|
|
UserRows int
|
|
LooksEmpty bool // UserTableFound && UserRows == 0
|
|
}
|
|
|
|
// DumpFileInfo holds info about a dump file on disk.
|
|
type DumpFileInfo struct {
|
|
FileName string
|
|
StackName string
|
|
DBType DBType
|
|
Size int64
|
|
ModTime time.Time
|
|
Validation DumpValidation
|
|
// IsUndo (R-379, v0.220.0) marks a `pre-restore-` safety copy: the undo taken immediately before
|
|
// a reconstitution, not the app's own backup. Its VISIBILITY is a recorded decision — see the
|
|
// comment on preRestoreDumpPrefix, "an undo the customer cannot see is not much of one" — so this
|
|
// flag names it rather than hiding it.
|
|
//
|
|
// It also fixes a real, if latent, naming defect: the filename parse below used to derive
|
|
// StackName by trimming only the engine suffix, so
|
|
// `pre-restore-20260822T140924Z-docmost-postgres.sql` yielded the phantom stack
|
|
// `pre-restore-20260822T140924Z-docmost`. That name reaches the `dbStacks` lookup in
|
|
// web.buildAppBackupRows as a KEY. It renders nothing today because that function iterates
|
|
// DEPLOYED apps and only reads the map by key — verified on the live page 2026-08-22, which
|
|
// contained zero `pre-restore` strings while four such files sat on disk — but any future code
|
|
// that RANGES over that map would surface a stack that does not exist.
|
|
IsUndo bool
|
|
// UndoAt is when the undo copy was taken, parsed from the stamp in its own filename. Zero for a
|
|
// normal dump.
|
|
UndoAt time.Time
|
|
}
|
|
|
|
// UndoDumpPrefix is the marker on a pre-restore safety copy. It is duplicated from
|
|
// backup.preRestoreDumpPrefix on purpose — appbackup must not import backup (that direction is the
|
|
// dependency, not this one) — and the two are pinned equal by a test.
|
|
const UndoDumpPrefix = "pre-restore-"
|
|
|
|
// trimUndoPrefix splits `pre-restore-<stamp>-<rest>` into (<rest>, <stamp>, true). Returns
|
|
// (base, "", false) for anything that is not an undo copy, so a normal dump flows unchanged.
|
|
func trimUndoPrefix(base string) (string, string, bool) {
|
|
if !strings.HasPrefix(base, UndoDumpPrefix) {
|
|
return base, "", false
|
|
}
|
|
after := strings.TrimPrefix(base, UndoDumpPrefix)
|
|
i := strings.Index(after, "-")
|
|
if i <= 0 {
|
|
return base, "", false // a prefix with no stamp is not the shape writeSafetyDump writes
|
|
}
|
|
return after[i+1:], after[:i], true
|
|
}
|
|
|
|
// DiscoverDatabases finds running database containers via docker ps.
|
|
//
|
|
// knownStacks is the set of actually-deployed stack names; it is used to attribute each DB container to
|
|
// the correct stack (M19). Pass nil/empty for the legacy suffix-strip behaviour.
|
|
func DiscoverDatabases(ctx context.Context, logger *log.Logger, debug bool, knownStacks []string) ([]DiscoveredDB, error) {
|
|
known := make(map[string]bool, len(knownStacks))
|
|
for _, s := range knownStacks {
|
|
if s != "" {
|
|
known[s] = true
|
|
}
|
|
}
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: running docker ps to find database containers")
|
|
}
|
|
// R-355: the compose PROJECT label is asked for, because it is the stack name BY CONSTRUCTION and
|
|
// the container name is only a guess at it. The controller runs compose with `cmd.Dir` set to
|
|
// `/opt/docker/stacks/<stack>` and never passes `-p` (stacks/manager.go:1218-1226), so compose
|
|
// derives the project from that directory — i.e. from the stack name itself.
|
|
//
|
|
// The label sits in the MIDDLE of the format on purpose: `strings.TrimSpace` is applied to the
|
|
// whole `docker ps` output, so a trailing empty field on the LAST line would be eaten and that
|
|
// row would arrive one column short. Image is never empty, so ending on it keeps every row the
|
|
// same width.
|
|
cmd := exec.CommandContext(ctx, "docker", "ps", "--format",
|
|
"{{.ID}}\t{{.Names}}\t{{.Label \"com.docker.compose.project\"}}\t{{.Image}}", "--filter", "status=running")
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
return nil, fmt.Errorf("docker ps failed: %w", err)
|
|
}
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: docker ps output: %s", util.TruncateStr(strings.TrimSpace(string(out)), 500))
|
|
}
|
|
|
|
var dbs []DiscoveredDB
|
|
var skipped int
|
|
|
|
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
|
|
if line == "" {
|
|
continue
|
|
}
|
|
parts := strings.SplitN(line, "\t", 4)
|
|
if len(parts) < 4 {
|
|
continue
|
|
}
|
|
|
|
id, name, project, image := parts[0], parts[1], strings.TrimSpace(parts[2]), strings.ToLower(parts[3])
|
|
|
|
// R-47: the same predicate that DBServiceNames applies to compose `image:` values, so a dump
|
|
// that exists is always attributable to a startable service (see dbservices.go).
|
|
dbType, isDB := dbTypeForImage(image)
|
|
if !isDB {
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: skipping container %s (image=%s, not a database)", name, image)
|
|
}
|
|
skipped++
|
|
continue
|
|
}
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: found %s container: %s (id=%s)", dbType, name, id[:12])
|
|
}
|
|
|
|
db := DiscoveredDB{
|
|
ContainerID: id,
|
|
ContainerName: name,
|
|
DBType: dbType,
|
|
StackName: resolveStackName(name, project, known, logger, debug),
|
|
}
|
|
|
|
// Get env vars from container
|
|
if err := populateDBEnv(ctx, &db); err != nil {
|
|
logger.Printf("[WARN] [backup] Could not read env vars for %s: %v", name, err)
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: skipping %s — env read failed", name)
|
|
}
|
|
continue
|
|
}
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: %s → stack=%s, dbUser=%s, dbName=%s", name, db.StackName, db.DBUser, db.DBName)
|
|
}
|
|
|
|
dbs = append(dbs, db)
|
|
}
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: found %d database(s), skipped %d non-DB container(s)", len(dbs), skipped)
|
|
}
|
|
|
|
logger.Printf("[INFO] [backup] Discovered %d databases", len(dbs))
|
|
return dbs, nil
|
|
}
|
|
|
|
// DumpAll dumps all discovered databases.
|
|
func DumpAll(ctx context.Context, dbs []DiscoveredDB, dumpDir string, logger *log.Logger, debug bool) []DumpResult {
|
|
// Clean up old .tmp files (older than 1 hour)
|
|
cleanupTmpFiles(dumpDir, logger)
|
|
|
|
logger.Printf("[INFO] [backup] Starting DB dump for %d databases", len(dbs))
|
|
var results []DumpResult
|
|
var failed int
|
|
for _, db := range dbs {
|
|
result := DumpOne(ctx, db, dumpDir, logger, debug)
|
|
results = append(results, result)
|
|
if result.Error != nil {
|
|
failed++
|
|
}
|
|
}
|
|
logger.Printf("[INFO] [backup] DB dump complete: %d succeeded, %d failed", len(results)-failed, failed)
|
|
return results
|
|
}
|
|
|
|
// DumpOne dumps a single database.
|
|
func DumpOne(ctx context.Context, db DiscoveredDB, dumpDir string, logger *log.Logger, debug bool) DumpResult {
|
|
start := time.Now()
|
|
result := DumpResult{DB: db}
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: starting dump for container=%s, stack=%s, dbType=%s, dumpDir=%s",
|
|
db.ContainerName, db.StackName, db.DBType, dumpDir)
|
|
}
|
|
|
|
// Ensure dump directory exists
|
|
if err := os.MkdirAll(dumpDir, 0755); err != nil {
|
|
result.Error = fmt.Errorf("creating dump dir: %w", err)
|
|
result.Duration = time.Since(start)
|
|
return result
|
|
}
|
|
|
|
filename := fmt.Sprintf("%s-%s.sql", db.StackName, db.DBType)
|
|
tmpPath := filepath.Join(dumpDir, filename+".tmp")
|
|
finalPath := filepath.Join(dumpDir, filename)
|
|
|
|
// 5-minute timeout per dump
|
|
dumpCtx, cancel := context.WithTimeout(ctx, 5*time.Minute)
|
|
defer cancel()
|
|
|
|
// Verify container is still running
|
|
checkCmd := exec.CommandContext(dumpCtx, "docker", "inspect", "--format", "{{.State.Running}}", db.ContainerID)
|
|
checkOut, err := checkCmd.Output()
|
|
if err != nil || strings.TrimSpace(string(checkOut)) != "true" {
|
|
result.Error = fmt.Errorf("container %s no longer running", db.ContainerName)
|
|
result.Duration = time.Since(start)
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: container %s is no longer running — skipping", db.ContainerName)
|
|
}
|
|
return result
|
|
}
|
|
|
|
// Build dump command
|
|
var cmd *exec.Cmd
|
|
switch db.DBType {
|
|
case DBTypePostgres:
|
|
cmd = exec.CommandContext(dumpCtx, "docker", "exec", db.ContainerID,
|
|
"pg_dump", "-U", db.DBUser, "-d", db.DBName,
|
|
"--clean", "--if-exists", "--no-owner", "--no-privileges")
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: pg_dump command: docker exec %s pg_dump -U %s -d %s --clean --if-exists --no-owner --no-privileges",
|
|
db.ContainerID[:12], db.DBUser, db.DBName)
|
|
}
|
|
case DBTypeMariaDB:
|
|
// Get root password from container env
|
|
password := getMariaDBPassword(dumpCtx, db.ContainerID)
|
|
if password == "" {
|
|
result.Error = fmt.Errorf("could not determine MariaDB root password for %s", db.ContainerName)
|
|
result.Duration = time.Since(start)
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: MariaDB root password not found for %s — skipping", db.ContainerName)
|
|
}
|
|
return result
|
|
}
|
|
cmd = exec.CommandContext(dumpCtx, "docker", "exec", db.ContainerID,
|
|
"mariadb-dump", "-u", "root", "-p***",
|
|
"--single-transaction", "--routines", "--triggers", db.DBName)
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: mariadb-dump command: docker exec %s mariadb-dump -u root -p*** --single-transaction --routines --triggers %s",
|
|
db.ContainerID[:12], db.DBName)
|
|
}
|
|
// Actual command with real password (not logged)
|
|
cmd = exec.CommandContext(dumpCtx, "docker", "exec", db.ContainerID,
|
|
"mariadb-dump", "-u", "root", "-p"+password,
|
|
"--single-transaction", "--routines", "--triggers", db.DBName)
|
|
default:
|
|
result.Error = fmt.Errorf("unsupported DB type: %s", db.DBType)
|
|
result.Duration = time.Since(start)
|
|
return result
|
|
}
|
|
|
|
// Write output to tmp file
|
|
tmpFile, err := os.Create(tmpPath)
|
|
if err != nil {
|
|
result.Error = fmt.Errorf("creating tmp file: %w", err)
|
|
result.Duration = time.Since(start)
|
|
return result
|
|
}
|
|
defer tmpFile.Close()
|
|
|
|
cmd.Stdout = tmpFile
|
|
var stderr strings.Builder
|
|
cmd.Stderr = &stderr
|
|
|
|
err = cmd.Run()
|
|
|
|
if err != nil {
|
|
os.Remove(tmpPath)
|
|
errMsg := stderr.String()
|
|
if len(errMsg) > 200 {
|
|
errMsg = errMsg[:200]
|
|
}
|
|
result.Error = fmt.Errorf("dump failed: %v — %s", err, errMsg)
|
|
result.Duration = time.Since(start)
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: dump command failed for %s: %v", db.ContainerName, result.Error)
|
|
}
|
|
return result
|
|
}
|
|
|
|
// Close and sync tmpFile before rename to ensure data is flushed to disk (H8 fix).
|
|
if err := tmpFile.Sync(); err != nil {
|
|
os.Remove(tmpPath)
|
|
result.Error = fmt.Errorf("syncing dump file: %w", err)
|
|
result.Duration = time.Since(start)
|
|
return result
|
|
}
|
|
if err := tmpFile.Close(); err != nil {
|
|
os.Remove(tmpPath)
|
|
result.Error = fmt.Errorf("closing dump file: %w", err)
|
|
result.Duration = time.Since(start)
|
|
return result
|
|
}
|
|
|
|
// Check file size
|
|
stat, err := os.Stat(tmpPath)
|
|
if err != nil || stat.Size() == 0 {
|
|
os.Remove(tmpPath)
|
|
result.Error = fmt.Errorf("dump produced empty file for %s", db.ContainerName)
|
|
result.Duration = time.Since(start)
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: dump produced empty file for %s", db.ContainerName)
|
|
}
|
|
return result
|
|
}
|
|
|
|
// Rename tmp to final
|
|
if err := os.Rename(tmpPath, finalPath); err != nil {
|
|
os.Remove(tmpPath)
|
|
result.Error = fmt.Errorf("renaming dump file: %w", err)
|
|
result.Duration = time.Since(start)
|
|
return result
|
|
}
|
|
|
|
result.FilePath = finalPath
|
|
result.Size = stat.Size()
|
|
result.Duration = time.Since(start)
|
|
|
|
// Run validation on the dump file
|
|
result.Validation = ValidateDump(finalPath, db.DBType)
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] DumpOne: completed %s → %s (size=%s, valid=%v, tables=%d, duration=%s)",
|
|
db.ContainerName, filename, humanizeBytes(stat.Size()),
|
|
result.Validation.Valid, result.Validation.TableCount, result.Duration.Round(time.Millisecond))
|
|
}
|
|
|
|
logger.Printf("[INFO] [backup] DB dump: %s → %s (%s, %s, %d tables)", db.ContainerName, filename,
|
|
humanizeBytes(stat.Size()), result.Duration.Round(time.Millisecond), result.Validation.TableCount)
|
|
|
|
return result
|
|
}
|
|
|
|
// ValidateDump checks a SQL dump file for basic structural integrity.
|
|
func ValidateDump(filePath string, dbType DBType) DumpValidation {
|
|
stat, err := os.Stat(filePath)
|
|
if err != nil {
|
|
return DumpValidation{Error: fmt.Sprintf("stat failed: %v", err)}
|
|
}
|
|
|
|
v := DumpValidation{
|
|
FileSize: stat.Size(),
|
|
ModTime: stat.ModTime(),
|
|
}
|
|
|
|
if stat.Size() < 100 {
|
|
v.Error = "dump file too small (< 100 bytes)"
|
|
log.Printf("[WARN] [backup] ValidateDump FAIL: %s — %s", filePath, v.Error)
|
|
return v
|
|
}
|
|
|
|
// H1: Use bufio.Scanner to read line-by-line instead of loading entire file into memory.
|
|
// Large dumps (500MB+) would cause massive allocations on every 5-min cache refresh.
|
|
f, err := os.Open(filePath)
|
|
if err != nil {
|
|
v.Error = fmt.Sprintf("read failed: %v", err)
|
|
log.Printf("[WARN] [backup] ValidateDump FAIL: %s — %s", filePath, v.Error)
|
|
return v
|
|
}
|
|
defer f.Close()
|
|
|
|
// Use bufio.Reader instead of Scanner: ReadLine gracefully handles lines
|
|
// longer than the buffer (isPrefix=true) so we can skip them. Only short
|
|
// lines matter (headers, CREATE TABLE). Long COPY/INSERT data lines
|
|
// (e.g., Immich's binary-encoded image data) are skipped without allocating.
|
|
reader := bufio.NewReaderSize(f, 256*1024)
|
|
|
|
lineNum := 0
|
|
headerFound := false
|
|
tableCount := 0
|
|
// R-44 sniff state. inUserCopy tracks a postgres `COPY … FROM stdin;` block for an accounts
|
|
// table; rows are counted until the `\.` terminator.
|
|
inUserCopy := false
|
|
for {
|
|
lineBytes, isPrefix, err := reader.ReadLine()
|
|
if err != nil {
|
|
if err != io.EOF {
|
|
v.Error = fmt.Sprintf("hiba az olvasás közben: %v", err)
|
|
log.Printf("[WARN] [backup] ValidateDump FAIL: %s — read error: %v", filePath, err)
|
|
return v
|
|
}
|
|
break // EOF
|
|
}
|
|
if isPrefix {
|
|
// Line exceeds buffer — skip remainder (COPY data, large INSERTs).
|
|
// A long line inside a user COPY block is still a ROW: count it before discarding it,
|
|
// or a table whose rows happen to be wide would sniff as empty and raise a false alarm.
|
|
if inUserCopy {
|
|
v.UserRows++
|
|
}
|
|
for isPrefix && err == nil {
|
|
_, isPrefix, err = reader.ReadLine()
|
|
}
|
|
continue
|
|
}
|
|
|
|
line := string(lineBytes)
|
|
lineNum++
|
|
|
|
// R-44 content sniff (warn-level; see DumpValidation).
|
|
if inUserCopy {
|
|
if line == `\.` {
|
|
inUserCopy = false
|
|
} else {
|
|
v.UserRows++
|
|
}
|
|
} else if isUserCopyStart(line, dbType) {
|
|
inUserCopy = true
|
|
v.UserTableFound = true
|
|
} else if dbType == DBTypeMariaDB && isUserInsert(line) {
|
|
// mysqldump writes multi-row `INSERT INTO \`users\` VALUES (…),(…);` — the row count is
|
|
// not worth parsing out of it, and presence alone answers the only question asked here.
|
|
v.UserTableFound = true
|
|
v.UserRows++
|
|
}
|
|
|
|
// Header check — scan first 10 lines for expected dump header
|
|
// MariaDB 11.4+ prepends a sandbox comment before the header line
|
|
if lineNum <= 10 && !headerFound {
|
|
switch dbType {
|
|
case DBTypeMariaDB:
|
|
if strings.HasPrefix(line, "-- MariaDB dump") ||
|
|
strings.HasPrefix(line, "-- MySQL dump") ||
|
|
strings.HasPrefix(line, "-- mysqldump") {
|
|
headerFound = true
|
|
}
|
|
case DBTypePostgres:
|
|
if strings.HasPrefix(line, "-- PostgreSQL database dump") {
|
|
headerFound = true
|
|
}
|
|
}
|
|
}
|
|
|
|
// Count CREATE TABLE statements
|
|
upper := strings.ToUpper(strings.TrimSpace(line))
|
|
if strings.HasPrefix(upper, "CREATE TABLE") {
|
|
tableCount++
|
|
}
|
|
}
|
|
|
|
v.TableCount = tableCount
|
|
|
|
if !headerFound {
|
|
switch dbType {
|
|
case DBTypeMariaDB:
|
|
v.Error = "MariaDB dump missing comment header"
|
|
case DBTypePostgres:
|
|
v.Error = "PostgreSQL dump missing comment header"
|
|
}
|
|
log.Printf("[WARN] [backup] ValidateDump FAIL: %s — %s", filePath, v.Error)
|
|
return v
|
|
}
|
|
|
|
if tableCount == 0 {
|
|
v.Error = "no CREATE TABLE statements found"
|
|
log.Printf("[WARN] [backup] ValidateDump FAIL: %s — %s (header was found, scanned %d lines)", filePath, v.Error, lineNum)
|
|
return v
|
|
}
|
|
|
|
v.LooksEmpty = v.UserTableFound && v.UserRows == 0
|
|
if v.LooksEmpty {
|
|
log.Printf("[WARN] [backup] ValidateDump: %s is structurally valid (%d tables) but its accounts table has NO rows — the dump may predate the customer's data", filePath, tableCount)
|
|
}
|
|
|
|
v.Valid = true
|
|
return v
|
|
}
|
|
|
|
// userTableNames are the table names treated as "the accounts table" by the R-44 sniff. Kept
|
|
// deliberately short: a wider net (anything containing "user") would match join/audit tables like
|
|
// `user_metadata` or `album_user`, which are legitimately empty on a healthy single-user install
|
|
// and would produce exactly the false alarm this signal must not raise.
|
|
var userTableNames = []string{"user", "users", "account", "accounts"}
|
|
|
|
// isUserCopyStart reports whether a line opens a postgres `COPY <accounts-table> … FROM stdin;`
|
|
// block. pg_dump writes the table qualified and optionally quoted — `COPY public."user" (…)`,
|
|
// `COPY public.users (…)` — so both forms are matched.
|
|
func isUserCopyStart(line string, dbType DBType) bool {
|
|
if dbType != DBTypePostgres || !strings.HasPrefix(line, "COPY ") {
|
|
return false
|
|
}
|
|
rest := strings.TrimPrefix(line, "COPY ")
|
|
sp := strings.IndexByte(rest, ' ')
|
|
if sp < 0 {
|
|
return false
|
|
}
|
|
return matchesUserTable(rest[:sp])
|
|
}
|
|
|
|
// isUserInsert reports whether a line is a mysqldump INSERT into an accounts table.
|
|
func isUserInsert(line string) bool {
|
|
const pfx = "INSERT INTO "
|
|
if !strings.HasPrefix(line, pfx) {
|
|
return false
|
|
}
|
|
rest := strings.TrimPrefix(line, pfx)
|
|
sp := strings.IndexByte(rest, ' ')
|
|
if sp < 0 {
|
|
return false
|
|
}
|
|
return matchesUserTable(rest[:sp])
|
|
}
|
|
|
|
// matchesUserTable strips schema qualification and quoting from a dumped table reference and
|
|
// reports whether the bare name is an accounts table.
|
|
func matchesUserTable(ref string) bool {
|
|
if dot := strings.LastIndexByte(ref, '.'); dot >= 0 {
|
|
ref = ref[dot+1:]
|
|
}
|
|
ref = strings.Trim(ref, "\"`")
|
|
for _, n := range userTableNames {
|
|
if strings.EqualFold(ref, n) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// ListDumpFiles returns info about SQL dump files on disk.
|
|
//
|
|
// M18: ValidateDump scans the dump line-by-line; on a customer with hundreds-of-MB dumps that is wasted
|
|
// disk I/O + CPU on every ~5-min scheduler cycle. `cached` is an optional lookup that returns a
|
|
// previously-computed DumpValidation for a file whose (name, size, modtime) match — when it returns ok,
|
|
// the expensive ValidateDump is skipped. Pass nil to always validate (legacy fast path / other callers).
|
|
func ListDumpFiles(dumpDir string, cached func(name string, size int64, mod time.Time) (DumpValidation, bool)) ([]DumpFileInfo, error) {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
if os.IsNotExist(err) {
|
|
return nil, nil
|
|
}
|
|
return nil, fmt.Errorf("reading dump dir: %w", err)
|
|
}
|
|
|
|
var files []DumpFileInfo
|
|
for _, e := range entries {
|
|
// M2: Check .tmp before .sql to correctly skip ".sql.tmp" temp files (was dead code before).
|
|
if e.IsDir() || strings.HasSuffix(e.Name(), ".tmp") {
|
|
continue
|
|
}
|
|
if !strings.HasSuffix(e.Name(), ".sql") {
|
|
continue
|
|
}
|
|
|
|
info, err := e.Info()
|
|
if err != nil {
|
|
continue
|
|
}
|
|
|
|
f := DumpFileInfo{
|
|
FileName: e.Name(),
|
|
Size: info.Size(),
|
|
ModTime: info.ModTime(),
|
|
}
|
|
|
|
// Parse stack name and DB type from filename: "paperless-ngx-postgres.sql"
|
|
base := strings.TrimSuffix(e.Name(), ".sql")
|
|
// R-379: strip the `pre-restore-<stamp>-` head FIRST, so an undo copy resolves to the app it
|
|
// belongs to instead of a phantom stack named after its own timestamp.
|
|
if rest, stamp, ok := trimUndoPrefix(base); ok {
|
|
base = rest
|
|
f.IsUndo = true
|
|
if t, err := time.Parse("20060102T150405Z", stamp); err == nil {
|
|
f.UndoAt = t
|
|
}
|
|
}
|
|
if strings.HasSuffix(base, "-postgres") {
|
|
f.StackName = strings.TrimSuffix(base, "-postgres")
|
|
f.DBType = DBTypePostgres
|
|
} else if strings.HasSuffix(base, "-mariadb") {
|
|
f.StackName = strings.TrimSuffix(base, "-mariadb")
|
|
f.DBType = DBTypeMariaDB
|
|
} else {
|
|
f.StackName = base
|
|
}
|
|
|
|
// M18: reuse a cached validation when the file is unchanged (name+size+modtime), else validate.
|
|
if cached != nil {
|
|
if v, ok := cached(e.Name(), info.Size(), info.ModTime()); ok {
|
|
f.Validation = v
|
|
files = append(files, f)
|
|
continue
|
|
}
|
|
}
|
|
fullPath := filepath.Join(dumpDir, e.Name())
|
|
f.Validation = ValidateDump(fullPath, f.DBType)
|
|
|
|
files = append(files, f)
|
|
}
|
|
|
|
return files, nil
|
|
}
|
|
|
|
func populateDBEnv(ctx context.Context, db *DiscoveredDB) error {
|
|
cmd := exec.CommandContext(ctx, "docker", "inspect", db.ContainerID,
|
|
"--format", "{{range .Config.Env}}{{println .}}{{end}}")
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
env := make(map[string]string)
|
|
for _, line := range strings.Split(string(out), "\n") {
|
|
if idx := strings.IndexByte(line, '='); idx > 0 {
|
|
env[line[:idx]] = line[idx+1:]
|
|
}
|
|
}
|
|
|
|
switch db.DBType {
|
|
case DBTypePostgres:
|
|
db.DBUser = env["POSTGRES_USER"]
|
|
if db.DBUser == "" {
|
|
db.DBUser = "postgres"
|
|
}
|
|
db.DBName = env["POSTGRES_DB"]
|
|
if db.DBName == "" {
|
|
db.DBName = db.DBUser
|
|
}
|
|
case DBTypeMariaDB:
|
|
db.DBName = env["MYSQL_DATABASE"]
|
|
if db.DBName == "" {
|
|
db.DBName = env["MARIADB_DATABASE"]
|
|
}
|
|
if db.DBName == "" {
|
|
db.DBName = "mysql" // fallback to dump all
|
|
}
|
|
db.DBUser = "root"
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// ImportDump replays a (possibly gzipped) SQL dump into a RUNNING database container — the read-side
|
|
// counterpart to DumpOne (F17). It reuses the per-engine clients (psql / mariadb) and the DiscoveredDB's
|
|
// OWN credentials (discovered from the live container env), so the caller needs no external env map. The
|
|
// container must already be running (the restore flow brings the stack up first); ImportDump briefly
|
|
// waits for the engine to accept connections, then pipes the dump in. The backup dumps are produced with
|
|
// DROP/CREATE (pg_dump --clean --if-exists; mariadb-dump's default --add-drop-table), so a replay fully
|
|
// reconstructs the captured logical state.
|
|
func ImportDump(ctx context.Context, db DiscoveredDB, dumpPath string, logger *log.Logger, debug bool) error {
|
|
if err := waitDBReady(ctx, db, 30*time.Second); err != nil {
|
|
return fmt.Errorf("waiting for %s (%s) readiness: %w", db.ContainerName, db.DBType, err)
|
|
}
|
|
|
|
f, err := os.Open(dumpPath)
|
|
if err != nil {
|
|
return fmt.Errorf("opening dump %s: %w", dumpPath, err)
|
|
}
|
|
defer f.Close()
|
|
|
|
var reader io.Reader = f
|
|
if strings.HasSuffix(dumpPath, ".gz") {
|
|
gr, err := gzip.NewReader(f)
|
|
if err != nil {
|
|
return fmt.Errorf("opening gzip %s: %w", dumpPath, err)
|
|
}
|
|
defer gr.Close()
|
|
reader = gr
|
|
}
|
|
|
|
impCtx, cancel := context.WithTimeout(ctx, 30*time.Minute)
|
|
defer cancel()
|
|
|
|
var cmd *exec.Cmd
|
|
switch db.DBType {
|
|
case DBTypePostgres:
|
|
user := db.DBUser
|
|
if user == "" {
|
|
user = "postgres"
|
|
}
|
|
dbName := db.DBName
|
|
if dbName == "" {
|
|
dbName = user
|
|
}
|
|
// ON_ERROR_STOP=1: a real import error must FAIL (and surface), not silently half-apply.
|
|
//
|
|
// --single-transaction (R-380, v0.220.0) is a BELT, not the fix. It makes a Postgres replay
|
|
// all-or-nothing at the engine, so the emptied-and-crash-looping state measured on `docmost`
|
|
// on 2026-08-22 stops being reachable on this engine. It does NOT make MariaDB atomic —
|
|
// MariaDB's DDL is not transactional, so a partial apply there is unavoidable at the engine
|
|
// and is why the rollback in offbox_reconstitute.go exists and is the actual fix. Do not
|
|
// read this flag as making the rollback optional.
|
|
cmd = exec.CommandContext(impCtx, "docker", "exec", "-i", db.ContainerID,
|
|
"psql", "-v", "ON_ERROR_STOP=1", "--single-transaction", "-U", user, "-d", dbName)
|
|
case DBTypeMariaDB:
|
|
password := getMariaDBPassword(impCtx, db.ContainerID)
|
|
if password == "" {
|
|
return fmt.Errorf("could not determine MariaDB root password for %s", db.ContainerName)
|
|
}
|
|
cmd = exec.CommandContext(impCtx, "docker", "exec", "-i", db.ContainerID,
|
|
"mariadb", "-u", "root", "-p"+password, db.DBName)
|
|
default:
|
|
return fmt.Errorf("unsupported DB type: %s", db.DBType)
|
|
}
|
|
|
|
cmd.Stdin = reader
|
|
var stderr strings.Builder
|
|
cmd.Stderr = &stderr
|
|
if debug && logger != nil {
|
|
logger.Printf("[DEBUG] [backup] ImportDump: importing %s into %s (%s)", dumpPath, db.ContainerName, db.DBType)
|
|
}
|
|
if err := cmd.Run(); err != nil {
|
|
// R-381: the engine's stderr goes to the OPERATOR LOG, in full. It used to be truncated to
|
|
// 300 chars and folded into the returned error, which the restore surface then rendered to
|
|
// the customer — measured 2026-08-22 at 407 bytes (Postgres, a caret diagram and
|
|
// `exit status 3`) and 615 bytes (MariaDB, whose middle was an `INSERT INTO migrations
|
|
// VALUES (...)` listing, i.e. ROWS OUT OF THE CUSTOMER'S OWN DATABASE, HTML-escaped, on
|
|
// their dashboard).
|
|
//
|
|
// The diagnostic is ADDED here, not removed: previously nothing logged the full text at all,
|
|
// so this is strictly more for the operator and strictly less for the customer.
|
|
full := strings.TrimSpace(stderr.String())
|
|
if logger != nil {
|
|
logger.Printf("[ERROR] [backup] %s import into %s FAILED: %v — engine output follows:\n%s",
|
|
db.DBType, db.ContainerName, err, full)
|
|
}
|
|
return fmt.Errorf("%s import into %s failed: %w", db.DBType, db.ContainerName, err)
|
|
}
|
|
if logger != nil {
|
|
logger.Printf("[INFO] [backup] Imported DB dump %s into %s (%s)", filepath.Base(dumpPath), db.ContainerName, db.DBType)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// waitDBReady polls until the database accepts connections (pg_isready / mariadb-admin ping).
|
|
func waitDBReady(ctx context.Context, db DiscoveredDB, timeout time.Duration) error {
|
|
deadline := time.Now().Add(timeout)
|
|
for {
|
|
c, cancel := context.WithTimeout(ctx, 5*time.Second)
|
|
var cmd *exec.Cmd
|
|
switch db.DBType {
|
|
case DBTypePostgres:
|
|
user := db.DBUser
|
|
if user == "" {
|
|
user = "postgres"
|
|
}
|
|
cmd = exec.CommandContext(c, "docker", "exec", db.ContainerID, "pg_isready", "-U", user)
|
|
case DBTypeMariaDB:
|
|
pw := getMariaDBPassword(c, db.ContainerID)
|
|
cmd = exec.CommandContext(c, "docker", "exec", db.ContainerID, "mariadb-admin", "ping", "-u", "root", "-p"+pw)
|
|
default:
|
|
cancel()
|
|
return fmt.Errorf("unsupported DB type: %s", db.DBType)
|
|
}
|
|
err := cmd.Run()
|
|
cancel()
|
|
if err == nil {
|
|
return nil
|
|
}
|
|
if time.Now().After(deadline) {
|
|
return fmt.Errorf("timeout after %s", timeout)
|
|
}
|
|
time.Sleep(2 * time.Second)
|
|
}
|
|
}
|
|
|
|
func getMariaDBPassword(ctx context.Context, containerID string) string {
|
|
cmd := exec.CommandContext(ctx, "docker", "inspect", containerID,
|
|
"--format", "{{range .Config.Env}}{{println .}}{{end}}")
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
|
|
for _, line := range strings.Split(string(out), "\n") {
|
|
if strings.HasPrefix(line, "MYSQL_ROOT_PASSWORD=") {
|
|
return strings.TrimPrefix(line, "MYSQL_ROOT_PASSWORD=")
|
|
}
|
|
if strings.HasPrefix(line, "MARIADB_ROOT_PASSWORD=") {
|
|
return strings.TrimPrefix(line, "MARIADB_ROOT_PASSWORD=")
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// deriveStackName maps a DB container name to its owning stack name.
|
|
//
|
|
// M19: the old logic pure-suffix-stripped on `-` (postgres/db/mariadb/mysql/database/redis/cache),
|
|
// which misattributes a stack whose real slug ENDS in a role token (e.g. a stack literally named
|
|
// `my-cache` → stripped to `my`). When the set of actually-deployed stack names is known, cross-reference
|
|
// it so the result is a real stack:
|
|
// - candidate := suffix-strip result.
|
|
// - known[candidate] → candidate (a real DB-role suffix of a real stack, e.g. romm-postgres→romm).
|
|
// - else known[containerName] → containerName (the container name IS the stack — don't strip, e.g. my-cache).
|
|
// - else longest known prefix → handles <stack>_postgres / <stack>-1 / compose-suffixed names.
|
|
// - else → candidate (fall back to today's suffix-strip; preserves behaviour when
|
|
// the stack list is empty/unavailable, so nothing regresses).
|
|
//
|
|
// A nil/empty `known` map = the legacy fast path (pure suffix-strip).
|
|
// resolveStackName attributes a running DB container to the stack that owns it.
|
|
//
|
|
// R-355, and the reason this exists rather than more string-munging: the container name is a GUESS at
|
|
// the stack name and the compose project label is the ANSWER. `paperless-ngx` runs a container called
|
|
// `paperless-postgres`; no amount of suffix-stripping or prefix-matching can bridge that, and
|
|
// deriveStackName's last line returns its unresolved candidate as though it had resolved it. The dump
|
|
// then went to `backups/primary/paperless/db-dumps/` — a directory for a stack that does not exist, on
|
|
// the system drive — while the app's own recovery unit recorded `db_dumps: null`. Nothing collected it,
|
|
// nothing off-sited it, nothing restored it, and because writeSafetyDump filters on this same value the
|
|
// destructive restore took NO undo copy and told the customer the app had no database. Measured live
|
|
// 2026-08-21: 284 617 bytes, 72 tables, valid, and unreachable.
|
|
//
|
|
// The label is authoritative BY CONSTRUCTION — see the note on the `docker ps` format above — so it is
|
|
// preferred whenever it names a stack we know. It is not trusted blindly: a project label naming an
|
|
// unknown stack would write into an unknown app's directory, which is the very fault being fixed.
|
|
//
|
|
// The fallback is unchanged, so every container whose name already resolves keeps its exact previous
|
|
// attribution (pinned by TestResolveStackName_UnaffectedAppsAreUnchanged).
|
|
//
|
|
// When NEITHER route resolves to a known stack we still return the old candidate — refusing here would
|
|
// change behaviour for any container legitimately relying on the legacy path — but we say so loudly,
|
|
// because the silent version of this line is what hid R-355 for the life of the feature.
|
|
func resolveStackName(containerName, composeProject string, known map[string]bool, logger *log.Logger, debug bool) string {
|
|
if composeProject != "" && (len(known) == 0 || known[composeProject]) {
|
|
if debug && composeProject != deriveStackName(containerName, known) {
|
|
logger.Printf("[DEBUG] DiscoverDatabases: %s → stack %q from the compose project label (the container name would have given %q)",
|
|
containerName, composeProject, deriveStackName(containerName, known))
|
|
}
|
|
return composeProject
|
|
}
|
|
fallback := deriveStackName(containerName, known)
|
|
if len(known) > 0 && !known[fallback] {
|
|
logger.Printf("[WARN] [backup] DB container %q could not be attributed to any deployed stack (compose project %q, name gives %q) — its dump will be written under %q, which no recovery unit reads. This is the R-355 shape; investigate before trusting that app's backup.",
|
|
containerName, composeProject, fallback, fallback)
|
|
}
|
|
return fallback
|
|
}
|
|
|
|
func deriveStackName(containerName string, known map[string]bool) string {
|
|
candidate := suffixStripStackName(containerName)
|
|
|
|
if len(known) == 0 {
|
|
return candidate
|
|
}
|
|
if known[candidate] {
|
|
return candidate
|
|
}
|
|
if known[containerName] {
|
|
return containerName
|
|
}
|
|
// Longest known stack name that is a prefix of the container name (tie-break: longest wins).
|
|
best := ""
|
|
for name := range known {
|
|
if name == "" || len(name) >= len(containerName) {
|
|
continue
|
|
}
|
|
// boundary char so "rom" doesn't match "romm-..."; compose/role separators are - or _.
|
|
sep := containerName[len(name)]
|
|
if strings.HasPrefix(containerName, name) && (sep == '-' || sep == '_') && len(name) > len(best) {
|
|
best = name
|
|
}
|
|
}
|
|
if best != "" {
|
|
return best
|
|
}
|
|
return candidate
|
|
}
|
|
|
|
// suffixStripStackName is the legacy pure suffix-strip (the M19 fallback when no stack list is known).
|
|
func suffixStripStackName(containerName string) string {
|
|
knownSuffixes := []string{"postgres", "db", "mariadb", "mysql", "database", "redis", "cache"}
|
|
|
|
parts := strings.Split(containerName, "-")
|
|
if len(parts) <= 1 {
|
|
return containerName
|
|
}
|
|
|
|
last := strings.ToLower(parts[len(parts)-1])
|
|
for _, suffix := range knownSuffixes {
|
|
if last == suffix {
|
|
return strings.Join(parts[:len(parts)-1], "-")
|
|
}
|
|
}
|
|
return containerName
|
|
}
|
|
|
|
func cleanupTmpFiles(dumpDir string, logger *log.Logger) {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
return
|
|
}
|
|
|
|
cutoff := time.Now().Add(-1 * time.Hour)
|
|
for _, e := range entries {
|
|
if !strings.HasSuffix(e.Name(), ".tmp") {
|
|
continue
|
|
}
|
|
info, err := e.Info()
|
|
if err != nil {
|
|
continue
|
|
}
|
|
if info.ModTime().Before(cutoff) {
|
|
path := filepath.Join(dumpDir, e.Name())
|
|
os.Remove(path)
|
|
logger.Printf("[INFO] [backup] Cleaned up stale tmp file: %s", e.Name())
|
|
}
|
|
}
|
|
}
|
|
|
|
// M1: formatBytes removed — use humanizeBytes() from appdata.go (same package, no duplication).
|