0f9b796615
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable).
395 lines
20 KiB
Go
395 lines
20 KiB
Go
package backup
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// reconcileRestoreSecrets merges the recovery unit's non-secret env with the secrets recovered from
|
|
// the unit itself (D5) and from the guest's own app.yaml, and applies the FAIL-CLOSED data-key gate.
|
|
// It is the safety-critical heart of Phase 2b and is deliberately a pure function (no I/O) so it can
|
|
// be exhaustively unit-tested — the D5 source arrives as an ARGUMENT, not as a read.
|
|
//
|
|
// Policy:
|
|
// - Regenerate NOTHING here. Secrets come from the unit (portable class) or the guest (the rest).
|
|
// - A missing DATA-ENCRYPTING key (`dataKeyNames`) is FATAL: regenerating it would render the
|
|
// restored data unreadable, so we refuse and tell the operator to do a PBS whole-guest restore.
|
|
// D5 means the key is normally IN the unit — but "normally" is not a reason to soften the gate.
|
|
// - A missing resettable secret is NON-fatal: returned in `missing` so the caller can warn or
|
|
// regenerate it (O4). No data is lost.
|
|
//
|
|
// PRECEDENCE — the UNIT WINS over the guest when both hold a value for the same name.
|
|
//
|
|
// This is not arbitrary and it is not "newest wins". The unit's secrets are captured in the SAME run
|
|
// as the dumps beside them (runVolumeDumps → captureAllRecoveryUnits, backup.go), so the unit's value
|
|
// is the one that MATCHES THE DATA ABOUT TO BE RESTORED, whereas the guest's value is merely the most
|
|
// recent. Where they disagree the guest's has been rotated since the capture, and preferring it is
|
|
// precisely the data-loss bug:
|
|
// - a rotated data-encrypting key does not decrypt data encrypted with the old one;
|
|
// - a rotated DB password does not match the scram/mysql hash inside the restored data directory
|
|
// (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so the app cannot reach its own rows.
|
|
//
|
|
// The restore persists fullEnv back to the guest's app.yaml (RecreateStackDefinitionFromUnit), so
|
|
// unit-wins also leaves the guest consistent with the data now on disk.
|
|
func reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets map[string]string, secretNames, dataKeyNames []string) (fullEnv map[string]string, missing []string, err error) {
|
|
fullEnv = make(map[string]string, len(nonSecretEnv)+len(secretNames))
|
|
for k, v := range nonSecretEnv {
|
|
fullEnv[k] = v
|
|
}
|
|
// resolve applies the precedence: unit first, guest only as a fallback.
|
|
resolve := func(n string) (string, bool) {
|
|
if v, ok := unitSecrets[n]; ok && v != "" {
|
|
return v, true
|
|
}
|
|
if v, ok := guestSecrets[n]; ok && v != "" {
|
|
return v, true
|
|
}
|
|
return "", false
|
|
}
|
|
have := func(n string) bool {
|
|
_, ok := resolve(n)
|
|
return ok
|
|
}
|
|
for _, n := range secretNames {
|
|
if v, ok := resolve(n); ok {
|
|
fullEnv[n] = v
|
|
} else {
|
|
missing = append(missing, n)
|
|
}
|
|
}
|
|
// Fail-closed: any unrecoverable data-encrypting key aborts the restore.
|
|
var missingDataKeys []string
|
|
for _, dk := range dataKeyNames {
|
|
if !have(dk) {
|
|
missingDataKeys = append(missingDataKeys, dk)
|
|
}
|
|
}
|
|
if len(missingDataKeys) > 0 {
|
|
return nil, missing, fmt.Errorf(
|
|
"refusing to restore: data-encrypting key(s) %v are in NEITHER the recovery unit nor the guest's app.yaml — "+
|
|
"a PBS whole-guest restore is required first (regenerating the key would render stored data unreadable)",
|
|
missingDataKeys)
|
|
}
|
|
return fullEnv, missing, nil
|
|
}
|
|
|
|
// readUnitEnv parses a recovery unit's app.yaml and SPLITS it into the plain config env and the
|
|
// secrets the unit carries (D5), using the manifest's portable-secret names as the discriminator.
|
|
//
|
|
// The split is driven by the MANIFEST, not by guessing from key names: the manifest and the app.yaml
|
|
// are captured together and checksummed together, so they cannot disagree about which entries are
|
|
// secrets. A schema-1 unit has no portable names, so everything lands in nonSecret — exactly the
|
|
// pre-D5 behaviour, which is what makes an old unit still restorable.
|
|
func readUnitEnv(path string, portableNames []string) (nonSecret, unitSecrets map[string]string) {
|
|
nonSecret, unitSecrets = map[string]string{}, map[string]string{}
|
|
data, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nonSecret, unitSecrets
|
|
}
|
|
var s strippedAppYaml
|
|
if yaml.Unmarshal(data, &s) != nil || s.Env == nil {
|
|
return nonSecret, unitSecrets
|
|
}
|
|
isPortable := make(map[string]bool, len(portableNames))
|
|
for _, n := range portableNames {
|
|
isPortable[n] = true
|
|
}
|
|
for k, v := range s.Env {
|
|
if isPortable[k] {
|
|
unitSecrets[k] = v
|
|
continue
|
|
}
|
|
nonSecret[k] = v
|
|
}
|
|
return nonSecret, unitSecrets
|
|
}
|
|
|
|
// hasReplayableDump reports whether dumpDir holds a .sql dump that the replay could actually use.
|
|
// The `pre-restore-` safety dumps are EXCLUDED: they live in the same directory (deliberately — an
|
|
// undo the customer cannot see is not much of one) but are never a replay source, so counting them
|
|
// would arm the DB-only phase, and its fail-closed gate, for an app that has nothing to replay.
|
|
func hasReplayableDump(dumpDir string) bool {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
return false
|
|
}
|
|
for _, e := range entries {
|
|
if e.IsDir() || filepath.Ext(e.Name()) != ".sql" {
|
|
continue
|
|
}
|
|
if !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// UnitRestoreResult is what a local recovery-unit restore actually did, so the surface can STATE it
|
|
// rather than report a bare completion.
|
|
//
|
|
// It exists for the same reason OffsiteReconstituteResult does, and it is the same lesson arriving on
|
|
// the other path: on 2026-08-21 an opengist restore reported "Restore-from-unit completed" over a unit
|
|
// holding manifest.json and compose/ and nothing else, and no screen could have told the customer that
|
|
// no data had been returned (R-353).
|
|
//
|
|
// The Manifest* counts are carried BECAUSE zero-replayed has two causes and they are not the same
|
|
// fact. A unit that lists no dumps means THE BACKUP held no data. A unit that lists dumps none of which
|
|
// replayed means something is wrong and the customer's live data was left untouched. R-355 is the
|
|
// standing rule this obeys: a claim about the APP must never be inferred from a counter — and here it
|
|
// is not merely unproven but unprovable, because 07-backup-architecture §6.3 records that an app's
|
|
// canonical .sql could be absent from the unit for reasons that have nothing to do with whether the app
|
|
// has a database (R-361 destroyed exactly that file for four months).
|
|
type UnitRestoreResult struct {
|
|
// VolumesReplayed is how many named-volume tars were unpacked into live Docker volumes.
|
|
VolumesReplayed int
|
|
// DBsReplayed is how many .sql dumps were imported. Never inferred from the presence of a database
|
|
// service — only a completed import increments it.
|
|
DBsReplayed int
|
|
// ManifestVolumes is len(manifest.VolumeDumps): what the unit CLAIMS it captured. The gap between
|
|
// this and VolumesReplayed is the whole of Scenario C.
|
|
ManifestVolumes int
|
|
// ManifestDBs is len(manifest.DBDumps): the same claim for the database leg.
|
|
ManifestDBs int
|
|
// CountsUnknown marks a run whose counts could not be established AT ALL — today the one case is
|
|
// the no-unit fallback to RestoreApp, which returns only an error and whose signature is
|
|
// deliberately out of scope.
|
|
//
|
|
// IT EXISTS BECAUSE THE ZERO VALUE WOULD OTHERWISE LIE. Without it a fallback restore that really
|
|
// replayed three volumes reports VolumesReplayed=0 / ManifestVolumes=0 — the shape the surface
|
|
// reads as „ez a mentés csak a beállításokat tartalmazta, adatot nem". That is a confident false
|
|
// statement, and precisely the R-88 failure direction (degrade to NO DATA rather than to UNKNOWN)
|
|
// this whole change exists to remove. An unknown must be carried, never drawn as a zero.
|
|
CountsUnknown bool
|
|
}
|
|
|
|
// RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit.
|
|
//
|
|
// It reads the unit manifest, takes the portable secrets from the UNIT and the rest from the guest's
|
|
// live app.yaml (unit wins — see reconcileRestoreSecrets), applies the fail-closed data-key gate,
|
|
// restores the named-volume data from the unit's tars, then restores the app's definition from the unit
|
|
// and redeploys it with the reconstructed env (re-pulling the pinned image). If no unit exists it falls
|
|
// back to the legacy volume-only RestoreApp.
|
|
//
|
|
// D5: this no longer needs the guest. A restore with the guest's app.yaml absent succeeds, which is
|
|
// pinned by TestRestoreFromRecoveryUnitWithGuestAbsent — the withheld class is regenerated (O4) and
|
|
// only a data key missing from BOTH sources still refuses.
|
|
//
|
|
// R-102: this is now the thin caller. It names the PRIMARY unit — the app's own drive,
|
|
// backups/primary/<stack> — and hands it to RestoreFromRecoveryUnitAt, which holds the whole body.
|
|
// An unresolvable drive path is still refused inside …At, in the same place and with the same
|
|
// message, so the order of the checks a caller can observe is unchanged.
|
|
func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, error) {
|
|
return m.RestoreFromRecoveryUnitAt(stackName, RecoveryUnitPath(m.namespaceRoot(m.GetAppDrivePath(stackName)), stackName))
|
|
}
|
|
|
|
// RestoreFromRecoveryUnitAt is RestoreFromRecoveryUnit with an EXPLICIT recovery-unit directory.
|
|
//
|
|
// R-102. ONE implementation, two callers — the same rule restoreDockerVolumesFrom states beside
|
|
// itself in restore.go, and for the same reason: a second copy of this body is exactly how the local
|
|
// path and the off-site path drifted apart until nothing compared them.
|
|
//
|
|
// The reason it exists: Tier-2 mirrors the app's whole recovery unit to
|
|
// <dest>/backups/secondary/<stack>/recovery-unit/ on every run, and until now every reader of a unit
|
|
// could only name a path under backups/primary/. So in the one failure Tier-2 exists for — the
|
|
// primary drive is lost, taking the primary unit with it — the surviving copy could not be opened by
|
|
// any action in the product (07-backup-architecture §6.3, §7.2).
|
|
//
|
|
// THE SOURCE MOVES; THE DESTINATION DOES NOT. unitDir changes only where the manifest, the compose
|
|
// capture, the .sql dumps and the volume tars are READ from. The app's data is written back to the
|
|
// live Docker volumes and the live database container, and its definition to the guest, exactly as
|
|
// before — a restore that also relocated the app's data would be a migration, not a restore.
|
|
//
|
|
// Everything else is pinned and unchanged: the mutation order (stop → volumes → recreate → DB-only
|
|
// start → replay → start, R-47), the secret reconciliation with unit-over-guest precedence and the
|
|
// fail-closed data-key gate, and the no-unit fallback to RestoreApp with its CountsUnknown handling.
|
|
func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRestoreResult, error) {
|
|
var res UnitRestoreResult
|
|
if m.stackProvider == nil {
|
|
return res, fmt.Errorf("stack provider not configured")
|
|
}
|
|
|
|
m.mu.Lock()
|
|
if m.running {
|
|
m.mu.Unlock()
|
|
return res, fmt.Errorf("backup or restore already in progress")
|
|
}
|
|
m.running = true
|
|
m.mu.Unlock()
|
|
defer func() {
|
|
m.mu.Lock()
|
|
m.running = false
|
|
m.mu.Unlock()
|
|
}()
|
|
|
|
// The DESTINATION side, and it is deliberately still resolved here: RestoreApp (the no-unit
|
|
// fallback below) needs it, and 07-backup-architecture §6.3 records that the restore destination
|
|
// is resolved by the same rule as the capture destination. It is no longer used to derive any
|
|
// SOURCE path — that is what unitDir is for.
|
|
drivePath := m.GetAppDrivePath(stackName)
|
|
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
|
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
|
|
manifest := readManifest(UnitManifestFile(unitDir))
|
|
if manifest == nil {
|
|
m.logger.Printf("[WARN] [backup] No readable recovery unit for %s at %s — falling back to volume-only restore", stackName, unitDir)
|
|
m.mu.Lock()
|
|
m.running = false // RestoreApp re-acquires the running flag
|
|
m.mu.Unlock()
|
|
// The fallback has no unit and no manifest, and `RestoreApp` returns only an error — so the
|
|
// counts here are genuinely UNKNOWN, not zero. Reporting zero would tell a customer whose
|
|
// volumes were just restored that their backup held no data.
|
|
res.CountsUnknown = true
|
|
return res, m.RestoreApp(stackName, "")
|
|
}
|
|
|
|
// R-353: what the unit CLAIMS it holds, recorded before any mutation. Read from the manifest that
|
|
// was just parsed above, so the claim and the outcome are counted from the same document.
|
|
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
|
|
|
|
composeDir := UnitComposeDir(unitDir)
|
|
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
|
|
|
|
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
|
|
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
|
|
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
|
|
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
|
|
// reconcileRestoreSecrets), then the fail-closed gate.
|
|
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
|
|
fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
|
|
return res, err
|
|
}
|
|
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
|
|
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
|
|
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
|
|
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
|
|
// never generated: the fail-closed gate above already refused if one was missing, and the
|
|
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
|
|
//
|
|
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
|
|
// generator run means the secret was empty at capture AND absent from the guest.
|
|
//
|
|
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
|
|
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
|
|
// regenerated value leaves the app unable to authenticate against its own restored rows while the
|
|
// dump replay, which uses the container's local trust socket, still reports success. The old wording
|
|
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
|
|
// DB/root-password fields and is now scoped to what is actually true.
|
|
if len(missing) > 0 {
|
|
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
|
|
for _, dk := range manifest.DataKeyEnvVars {
|
|
dataKeySet[dk] = true
|
|
}
|
|
var generated, unresolved []string
|
|
for _, name := range missing {
|
|
if !dataKeySet[name] && m.generateSecret != nil {
|
|
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
|
|
fullEnv[name] = v
|
|
generated = append(generated, name)
|
|
continue
|
|
}
|
|
}
|
|
unresolved = append(unresolved, name)
|
|
}
|
|
if len(generated) > 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
|
|
stackName, generated)
|
|
}
|
|
if len(unresolved) > 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
|
|
stackName, len(unresolved), unresolved)
|
|
}
|
|
}
|
|
// R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with
|
|
// two answers, and "an absent log line is not evidence" — the drill reads this line to prove the
|
|
// secondary mirror, not the primary unit, was the source. It is a path, never a secret.
|
|
m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit %s: images=%d, secrets recovered=%d/%d, data_keys=%d",
|
|
stackName, unitDir, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars))
|
|
|
|
// R-47: which compose service holds the database, and is there anything to replay? Resolved from
|
|
// the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed
|
|
// BEFORE the first mutation, so the refusal below leaves the live app completely untouched.
|
|
dbServices, dsErr := DBServiceNames(filepath.Join(composeDir, "docker-compose.yml"))
|
|
if dsErr != nil {
|
|
// "cannot tell" is not "no database" — leave it empty and let the gate decide.
|
|
m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr)
|
|
}
|
|
dbDumpDir := UnitDBDumpDir(unitDir)
|
|
hasDumps := hasReplayableDump(dbDumpDir)
|
|
if hasDumps && len(dbServices) == 0 {
|
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName)
|
|
return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stackName)
|
|
}
|
|
|
|
// Stop, restore named-volume data, recreate the definition, replay the DB with ONLY the database
|
|
// service running, and only then start the whole stack.
|
|
// F17: surface a data-restore failure instead of swallowing it (we still bring the app back up).
|
|
var dataErr error
|
|
if err := m.stackProvider.StopStack(stackName); err != nil {
|
|
m.logger.Printf("[WARN] [backup] could not stop %s before restore: %v (continuing)", stackName, err)
|
|
}
|
|
// R-353: the count is captured even when the replay errors — a partial replay is a fact the
|
|
// customer's sentence has to be built from, and discarding it on the error path is how Scenario C
|
|
// would end up wearing Scenario B's wording.
|
|
// R-354's volume-REPLAY seam, reused here rather than a second one being invented. It defaults to
|
|
// the real restoreDockerVolumesFrom, so production behaviour is byte-for-byte what it was; what it
|
|
// buys is that R-102's acceptance test can assert WHICH directory the tars came out of without a
|
|
// Docker daemon. For 40 of the 53 catalogue apps that archive is the entire dataset, so "the
|
|
// mirror was the source" has to be provable for the volume leg too, not only for the env and the
|
|
// database.
|
|
volReplay := m.volumeReplayFrom
|
|
if volReplay == nil {
|
|
volReplay = m.restoreDockerVolumesFrom
|
|
}
|
|
replayed, volErr := volReplay(stackName, UnitVolumeDumpDir(unitDir))
|
|
res.VolumesReplayed = replayed
|
|
if volErr != nil {
|
|
m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, volErr)
|
|
dataErr = volErr
|
|
}
|
|
if err := m.stackProvider.RecreateStackDefinitionFromUnit(stackName, composeDir, fullEnv); err != nil {
|
|
return res, fmt.Errorf("recreating %s from unit: %w", stackName, err)
|
|
}
|
|
// F17: the captured .sql dump is the authoritative logical DB state — replay it AFTER the volume
|
|
// restore, so the dump WINS over any volume-tar copy of the database.
|
|
// R-47: the replay happens with ONLY the database service up. This used to run after
|
|
// RecreateStackFromUnit had already brought the WHOLE stack up, letting the application rebuild
|
|
// schema objects underneath the replay (H4, DIAG-immich-restore-round2-2026-07-19).
|
|
if hasDumps {
|
|
if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
} else if n, err := m.reimportDBDumpsAtCtx(stackName, dbDumpDir); err != nil {
|
|
res.DBsReplayed = n // partial credit: whatever imported before the failure really did import
|
|
m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
} else {
|
|
res.DBsReplayed = n
|
|
}
|
|
}
|
|
if err := m.stackProvider.StartStack(stackName); err != nil {
|
|
return res, fmt.Errorf("starting %s after restore from unit: %w", stackName, err)
|
|
}
|
|
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
|
|
m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err)
|
|
}
|
|
|
|
if dataErr != nil {
|
|
return res, fmt.Errorf("restore of %s from unit completed with data errors: %w", stackName, dataErr)
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s — %d volume(s) of %d listed, %d database(s) of %d listed",
|
|
stackName, res.VolumesReplayed, res.ManifestVolumes, res.DBsReplayed, res.ManifestDBs)
|
|
return res, nil
|
|
}
|