Files
felhom-controller/controller/internal/backup/restore_db.go
T
admin 80e6ad8c47
gates / gates (push) Successful in 26s
controller v0.267.0: tests off DooPlex's Docker, cut-off copies refused, two pages true
R-650: internal/dockerexec — every docker exec routed through it; under
go test a real docker is refused (opt-in FELHOM_TEST_REAL_DOCKER=1; a stub
under the temp dir is allowed). api/stacks/web tests run under a silent
stub (TestMain). TestR650_NoBareDockerExec pins it repo-wide.
R-640: a dump without its engine's completion marker is refused before
the first mutation (unit + off-site restore) and again before any load.
R-499: the Tier-2 page's system-disk sentence has four true branches.
R-518: the backup button states the measured ~8 min stop.
R-626: measured on 9202, not reproduced.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-23 20:25:28 +02:00

155 lines
6.3 KiB
Go

package backup
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
)
// reimportDBDumps replays the captured per-app .sql dumps back into the app's now-running database
// container(s) — the F17 fix. The per-app backup captures a logical SQL dump (DumpOne →
// <stack>-<dbtype>.sql) but the legacy restore only repopulated Docker volume tars and NEVER replayed
// the dump, so DB-resident data (e.g. rows in a DB whose data dir is a bind mount, not a named volume)
// did not come back. This runs AFTER volume restore + stack bring-up, so the dump WINS over any
// volume-tar copy of the DB (the operator-chosen precedence: the consistent logical dump is authoritative).
//
// It uses the live container's OWN discovered credentials (DiscoveredDB), so no env threading is needed.
// A dump whose DB container is not found is logged and skipped; an actual import FAILURE is returned
// (surfaced, not swallowed) so a failed data restore cannot read as success.
func (m *Manager) reimportDBDumps(ctx context.Context, stackName, nsRoot string) (int, error) {
return m.reimportDBDumpsFrom(ctx, stackName, AppDBDumpPath(nsRoot, stackName))
}
// reimportDBDumpsFrom is reimportDBDumps with an EXPLICIT dump directory. The offsite
// reconstitution path (R-43) replays out of the restored SCRATCH unit rather than the live one:
// the local unit is deliberately never overwritten by a placement, so the dump that belongs to the
// chosen snapshot exists only under the scratch. Same discovery/import seams, same failure
// semantics — only the source directory differs.
func (m *Manager) reimportDBDumpsFrom(ctx context.Context, stackName, dumpDir string) (int, error) {
entries, err := os.ReadDir(dumpDir)
if err != nil {
if os.IsNotExist(err) {
return 0, nil // no DB dumps for this app
}
return 0, fmt.Errorf("reading db-dump dir: %w", err)
}
hasDump := false
for _, e := range entries {
if !e.IsDir() && filepath.Ext(e.Name()) == ".sql" {
hasDump = true
break
}
}
if !hasDump {
return 0, nil
}
discover := m.discoverDBs
if discover == nil {
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
}
}
imp := m.importDBDump
if imp == nil {
imp = func(ctx context.Context, db DiscoveredDB, dumpPath string) error {
return ImportDump(ctx, db, dumpPath, m.logger, m.isDebug())
}
}
dbs, err := discover(ctx)
if err != nil {
return 0, fmt.Errorf("discovering DB containers for %s: %w", stackName, err)
}
var imported int
for _, db := range dbs {
if db.StackName != stackName {
continue
}
// The dump for this DB is named "<stack>-<dbtype>.sql" (see DumpOne).
dumpPath := filepath.Join(dumpDir, fmt.Sprintf("%s-%s.sql", stackName, db.DBType))
if _, statErr := os.Stat(dumpPath); statErr != nil {
continue // no dump for this particular DB engine
}
// R-640: a cut-off copy loads as SUCCESS into an empty PostgreSQL database. Checked here, on
// the one path every replay takes, whatever `imp` is — the seam must not be able to skip it.
if err := appbackup.CheckDumpComplete(dumpPath, db.DBType); err != nil {
m.logger.Printf("[ERROR] [backup] Restore %s: NOT replaying %s — %v", stackName, filepath.Base(dumpPath), err)
return imported, util.MsgError("err.backup.adatbazis_masolat_csonka_nem_toltve", stackName)
}
m.logger.Printf("[INFO] [backup] Restore %s: replaying DB dump into %s (%s)", stackName, db.ContainerName, db.DBType)
if err := imp(ctx, db, dumpPath); err != nil {
return imported, fmt.Errorf("importing %s dump for %s: %w", db.DBType, stackName, err)
}
imported++
}
if imported == 0 {
m.logger.Printf("[WARN] [backup] Restore %s: a .sql dump exists but no matching running DB container was found — DB content NOT restored", stackName)
} else {
m.logger.Printf("[INFO] [backup] Restore %s: replayed %d DB dump(s)", stackName, imported)
}
return imported, nil
}
// dbReimportTimeout bounds a DB replay so a stuck import cannot hang a restore indefinitely. Named
// once because both bounded entry points below must agree: two paths that differ in how long they
// let a wedged import hold the restore are two different products (R-102 added the second one).
const dbReimportTimeout = 35 * time.Minute
// reimportDBDumpsCtx is a small helper that runs reimportDBDumps with a bounded context so a stuck DB
// import cannot hang the restore indefinitely.
func (m *Manager) reimportDBDumpsCtx(stackName, nsRoot string) (int, error) {
ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout)
defer cancel()
return m.reimportDBDumps(ctx, stackName, nsRoot)
}
// reimportDBDumpsAtCtx is reimportDBDumpsCtx with an EXPLICIT dump directory — the bounded-context
// twin of reimportDBDumpsFrom, added for R-102 so the Tier-2 unit restore can replay out of the
// secondary mirror. Same discovery/import seams, same failure semantics, same bound; only the source
// directory differs.
func (m *Manager) reimportDBDumpsAtCtx(stackName, dumpDir string) (int, error) {
ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout)
defer cancel()
return m.reimportDBDumpsFrom(ctx, stackName, dumpDir)
}
// incompleteDumps names the replayable dumps in dumpDir that do not end with their engine's
// completion marker (R-640). Only the files a replay would load are checked: `<stack>-<engine>.sql`,
// never the pre-restore safety copies. A directory that cannot be read yields nothing here — the
// replay's own ReadDir reports that.
func incompleteDumps(dumpDir string) []string {
entries, err := os.ReadDir(dumpDir)
if err != nil {
return nil
}
var bad []string
for _, e := range entries {
name := e.Name()
if e.IsDir() || filepath.Ext(name) != ".sql" || strings.HasPrefix(name, preRestoreDumpPrefix) {
continue
}
var t DBType
switch {
case strings.HasSuffix(name, "-"+string(appbackup.DBTypePostgres)+".sql"):
t = appbackup.DBTypePostgres
case strings.HasSuffix(name, "-"+string(appbackup.DBTypeMariaDB)+".sql"):
t = appbackup.DBTypeMariaDB
default:
continue // not a file the replay loads
}
if err := appbackup.CheckDumpComplete(filepath.Join(dumpDir, name), t); err != nil {
bad = append(bad, name)
}
}
return bad
}