0f9b796615
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable).
114 lines
4.7 KiB
Go
114 lines
4.7 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"time"
|
|
)
|
|
|
|
// reimportDBDumps replays the captured per-app .sql dumps back into the app's now-running database
|
|
// container(s) — the F17 fix. The per-app backup captures a logical SQL dump (DumpOne →
|
|
// <stack>-<dbtype>.sql) but the legacy restore only repopulated Docker volume tars and NEVER replayed
|
|
// the dump, so DB-resident data (e.g. rows in a DB whose data dir is a bind mount, not a named volume)
|
|
// did not come back. This runs AFTER volume restore + stack bring-up, so the dump WINS over any
|
|
// volume-tar copy of the DB (the operator-chosen precedence: the consistent logical dump is authoritative).
|
|
//
|
|
// It uses the live container's OWN discovered credentials (DiscoveredDB), so no env threading is needed.
|
|
// A dump whose DB container is not found is logged and skipped; an actual import FAILURE is returned
|
|
// (surfaced, not swallowed) so a failed data restore cannot read as success.
|
|
func (m *Manager) reimportDBDumps(ctx context.Context, stackName, nsRoot string) (int, error) {
|
|
return m.reimportDBDumpsFrom(ctx, stackName, AppDBDumpPath(nsRoot, stackName))
|
|
}
|
|
|
|
// reimportDBDumpsFrom is reimportDBDumps with an EXPLICIT dump directory. The offsite
|
|
// reconstitution path (R-43) replays out of the restored SCRATCH unit rather than the live one:
|
|
// the local unit is deliberately never overwritten by a placement, so the dump that belongs to the
|
|
// chosen snapshot exists only under the scratch. Same discovery/import seams, same failure
|
|
// semantics — only the source directory differs.
|
|
func (m *Manager) reimportDBDumpsFrom(ctx context.Context, stackName, dumpDir string) (int, error) {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
if os.IsNotExist(err) {
|
|
return 0, nil // no DB dumps for this app
|
|
}
|
|
return 0, fmt.Errorf("reading db-dump dir: %w", err)
|
|
}
|
|
hasDump := false
|
|
for _, e := range entries {
|
|
if !e.IsDir() && filepath.Ext(e.Name()) == ".sql" {
|
|
hasDump = true
|
|
break
|
|
}
|
|
}
|
|
if !hasDump {
|
|
return 0, nil
|
|
}
|
|
|
|
discover := m.discoverDBs
|
|
if discover == nil {
|
|
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
|
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
|
}
|
|
}
|
|
imp := m.importDBDump
|
|
if imp == nil {
|
|
imp = func(ctx context.Context, db DiscoveredDB, dumpPath string) error {
|
|
return ImportDump(ctx, db, dumpPath, m.logger, m.isDebug())
|
|
}
|
|
}
|
|
|
|
dbs, err := discover(ctx)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("discovering DB containers for %s: %w", stackName, err)
|
|
}
|
|
|
|
var imported int
|
|
for _, db := range dbs {
|
|
if db.StackName != stackName {
|
|
continue
|
|
}
|
|
// The dump for this DB is named "<stack>-<dbtype>.sql" (see DumpOne).
|
|
dumpPath := filepath.Join(dumpDir, fmt.Sprintf("%s-%s.sql", stackName, db.DBType))
|
|
if _, statErr := os.Stat(dumpPath); statErr != nil {
|
|
continue // no dump for this particular DB engine
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Restore %s: replaying DB dump into %s (%s)", stackName, db.ContainerName, db.DBType)
|
|
if err := imp(ctx, db, dumpPath); err != nil {
|
|
return imported, fmt.Errorf("importing %s dump for %s: %w", db.DBType, stackName, err)
|
|
}
|
|
imported++
|
|
}
|
|
|
|
if imported == 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: a .sql dump exists but no matching running DB container was found — DB content NOT restored", stackName)
|
|
} else {
|
|
m.logger.Printf("[INFO] [backup] Restore %s: replayed %d DB dump(s)", stackName, imported)
|
|
}
|
|
return imported, nil
|
|
}
|
|
|
|
// dbReimportTimeout bounds a DB replay so a stuck import cannot hang a restore indefinitely. Named
|
|
// once because both bounded entry points below must agree: two paths that differ in how long they
|
|
// let a wedged import hold the restore are two different products (R-102 added the second one).
|
|
const dbReimportTimeout = 35 * time.Minute
|
|
|
|
// reimportDBDumpsCtx is a small helper that runs reimportDBDumps with a bounded context so a stuck DB
|
|
// import cannot hang the restore indefinitely.
|
|
func (m *Manager) reimportDBDumpsCtx(stackName, nsRoot string) (int, error) {
|
|
ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout)
|
|
defer cancel()
|
|
return m.reimportDBDumps(ctx, stackName, nsRoot)
|
|
}
|
|
|
|
// reimportDBDumpsAtCtx is reimportDBDumpsCtx with an EXPLICIT dump directory — the bounded-context
|
|
// twin of reimportDBDumpsFrom, added for R-102 so the Tier-2 unit restore can replay out of the
|
|
// secondary mirror. Same discovery/import seams, same failure semantics, same bound; only the source
|
|
// directory differs.
|
|
func (m *Manager) reimportDBDumpsAtCtx(stackName, dumpDir string) (int, error) {
|
|
ctx, cancel := context.WithTimeout(context.Background(), dbReimportTimeout)
|
|
defer cancel()
|
|
return m.reimportDBDumpsFrom(ctx, stackName, dumpDir)
|
|
}
|