78ff991f1c
Closes R-47. No new agent coupling — MinAgent stays 0.90.0. The replay needs a running DB container, so both restore paths started the WHOLE stack first, giving the application a window to rebuild the very schema objects the dump was about to create. Measured live on 2026-07-19 (H4, DIAG-immich-restore-round2): immich-server rebuilt clip_index two seconds before the dump's CREATE INDEX, the replay aborted "already exists" under ON_ERROR_STOP=1, and immich reported schema drift. The data survived only because pg_dump emits COPY before CREATE INDEX. Both paths now open a DB-ONLY window: only the stack's database service(s) come up, the dump is replayed with the app still down, and the full start runs only after the replay exits 0. Fail-closed: a dump with no identifiable DB service refuses BEFORE the first mutation. Every exit from the window still does a best-effort full start, so a failed restore never leaves a box with a database and no application. New: appbackup.DBServiceNames (yaml.v3 services-map parse — never a line scan; immich's top-level volume keys are the decoy) sharing dbTypeForImage with DiscoverDatabases; stacks.Manager.StartStackServices (refuses an empty list — argument-less `up -d` is a full start); RedeployFromEnv split into PersistUnitRedeployConfig + its unchanged tail. StackDataProvider's RecreateStackFromUnit becomes RecreateStackDefinitionFromUnit — the hidden `up -d` inside the old name is what carried the defect on the local path. 19 new tests (ordering plus state-at-replay-time, zero-mutation fail-closed effects, replay-failure bring-up, parser decoys, empty-list refusal); three companion red-proofs run and reverted. 23/23 packages green. Not yet live-validated: STOP-1 supervised reconstitute, golden 0.153.0.
231 lines
9.9 KiB
Go
231 lines
9.9 KiB
Go
package backup
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// reconcileRestoreSecrets merges the recovery unit's non-secret env with the secrets recovered from
|
|
// the guest's own app.yaml, and applies the FAIL-CLOSED data-key gate. It is the safety-critical heart
|
|
// of Phase 2b and is deliberately a pure function (no I/O) so it can be exhaustively unit-tested.
|
|
//
|
|
// Policy (per the Phase 2 design — see REPORT/CHANGELOG):
|
|
// - Regenerate NOTHING. Every secret comes from the guest (live rootfs, or PBS whole-guest restore).
|
|
// - A missing DATA-ENCRYPTING key (`dataKeyNames`) is FATAL: regenerating it would render the
|
|
// restored data unreadable, so we refuse and tell the operator to do a PBS whole-guest restore.
|
|
// - A missing resettable secret (DB password, admin password) is NON-fatal: it's returned in
|
|
// `missing` so the caller can warn; the app may simply need a credential reset, no data is lost.
|
|
func reconcileRestoreSecrets(nonSecretEnv, recoveredSecrets map[string]string, secretNames, dataKeyNames []string) (fullEnv map[string]string, missing []string, err error) {
|
|
fullEnv = make(map[string]string, len(nonSecretEnv)+len(secretNames))
|
|
for k, v := range nonSecretEnv {
|
|
fullEnv[k] = v
|
|
}
|
|
have := func(n string) bool {
|
|
v, ok := recoveredSecrets[n]
|
|
return ok && v != ""
|
|
}
|
|
for _, n := range secretNames {
|
|
if have(n) {
|
|
fullEnv[n] = recoveredSecrets[n]
|
|
} else {
|
|
missing = append(missing, n)
|
|
}
|
|
}
|
|
// Fail-closed: any unrecoverable data-encrypting key aborts the restore.
|
|
var missingDataKeys []string
|
|
for _, dk := range dataKeyNames {
|
|
if !have(dk) {
|
|
missingDataKeys = append(missingDataKeys, dk)
|
|
}
|
|
}
|
|
if len(missingDataKeys) > 0 {
|
|
return nil, missing, fmt.Errorf(
|
|
"refusing to restore: data-encrypting key(s) %v could not be recovered from the guest's app.yaml — "+
|
|
"a PBS whole-guest restore is required first (regenerating the key would render stored data unreadable)",
|
|
missingDataKeys)
|
|
}
|
|
return fullEnv, missing, nil
|
|
}
|
|
|
|
// readStrippedEnv parses the non-secret env from a recovery unit's secret-stripped app.yaml.
|
|
func readStrippedEnv(path string) map[string]string {
|
|
data, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return map[string]string{}
|
|
}
|
|
var s strippedAppYaml
|
|
if yaml.Unmarshal(data, &s) != nil || s.Env == nil {
|
|
return map[string]string{}
|
|
}
|
|
return s.Env
|
|
}
|
|
|
|
// hasReplayableDump reports whether dumpDir holds a .sql dump that the replay could actually use.
|
|
// The `pre-restore-` safety dumps are EXCLUDED: they live in the same directory (deliberately — an
|
|
// undo the customer cannot see is not much of one) but are never a replay source, so counting them
|
|
// would arm the DB-only phase, and its fail-closed gate, for an app that has nothing to replay.
|
|
func hasReplayableDump(dumpDir string) bool {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
return false
|
|
}
|
|
for _, e := range entries {
|
|
if e.IsDir() || filepath.Ext(e.Name()) != ".sql" {
|
|
continue
|
|
}
|
|
if !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit + the guest's own secrets.
|
|
//
|
|
// It reads the unit manifest, recovers the secret values from the guest's live app.yaml, applies the
|
|
// fail-closed data-key gate, restores the named-volume data from the unit's tars, then restores the
|
|
// app's definition from the unit and redeploys it with the reconstructed env (re-pulling the pinned
|
|
// image). No secret is ever regenerated, and no secret is read from the unit. If no unit exists it
|
|
// falls back to the legacy volume-only RestoreApp.
|
|
func (m *Manager) RestoreFromRecoveryUnit(stackName string) error {
|
|
if m.stackProvider == nil {
|
|
return fmt.Errorf("stack provider not configured")
|
|
}
|
|
|
|
m.mu.Lock()
|
|
if m.running {
|
|
m.mu.Unlock()
|
|
return fmt.Errorf("backup or restore already in progress")
|
|
}
|
|
m.running = true
|
|
m.mu.Unlock()
|
|
defer func() {
|
|
m.mu.Lock()
|
|
m.running = false
|
|
m.mu.Unlock()
|
|
}()
|
|
|
|
drivePath := m.GetAppDrivePath(stackName)
|
|
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
|
return fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
nsRoot := m.namespaceRoot(drivePath)
|
|
|
|
manifest := readManifest(RecoveryUnitManifestPath(nsRoot, stackName))
|
|
if manifest == nil {
|
|
m.logger.Printf("[WARN] [backup] No recovery unit for %s — falling back to volume-only restore", stackName)
|
|
m.mu.Lock()
|
|
m.running = false // RestoreApp re-acquires the running flag
|
|
m.mu.Unlock()
|
|
return m.RestoreApp(stackName, "")
|
|
}
|
|
|
|
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
|
|
nonSecretEnv := readStrippedEnv(filepath.Join(composeDir, "app.yaml"))
|
|
|
|
// Recover secrets from the GUEST (never the unit), then apply the fail-closed gate.
|
|
recovered := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
|
|
fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, recovered, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
|
|
return err
|
|
}
|
|
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
|
|
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
|
|
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
|
|
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
|
|
// never generated: the fail-closed gate above already refused if one was missing, and the
|
|
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
|
|
if len(missing) > 0 {
|
|
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
|
|
for _, dk := range manifest.DataKeyEnvVars {
|
|
dataKeySet[dk] = true
|
|
}
|
|
var generated, unresolved []string
|
|
for _, name := range missing {
|
|
if !dataKeySet[name] && m.generateSecret != nil {
|
|
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
|
|
fullEnv[name] = v
|
|
generated = append(generated, name)
|
|
continue
|
|
}
|
|
}
|
|
unresolved = append(unresolved, name)
|
|
}
|
|
if len(generated) > 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); stored data is unaffected (no data-key involved)",
|
|
stackName, generated)
|
|
}
|
|
if len(unresolved) > 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
|
|
stackName, len(unresolved), unresolved)
|
|
}
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit: images=%d, secrets recovered=%d/%d, data_keys=%d",
|
|
stackName, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars))
|
|
|
|
// R-47: which compose service holds the database, and is there anything to replay? Resolved from
|
|
// the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed
|
|
// BEFORE the first mutation, so the refusal below leaves the live app completely untouched.
|
|
dbServices, dsErr := DBServiceNames(filepath.Join(composeDir, "docker-compose.yml"))
|
|
if dsErr != nil {
|
|
// "cannot tell" is not "no database" — leave it empty and let the gate decide.
|
|
m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr)
|
|
}
|
|
hasDumps := hasReplayableDump(AppDBDumpPath(nsRoot, stackName))
|
|
if hasDumps && len(dbServices) == 0 {
|
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName)
|
|
return fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stackName)
|
|
}
|
|
|
|
// Stop, restore named-volume data, recreate the definition, replay the DB with ONLY the database
|
|
// service running, and only then start the whole stack.
|
|
// F17: surface a data-restore failure instead of swallowing it (we still bring the app back up).
|
|
var dataErr error
|
|
if err := m.stackProvider.StopStack(stackName); err != nil {
|
|
m.logger.Printf("[WARN] [backup] could not stop %s before restore: %v (continuing)", stackName, err)
|
|
}
|
|
if err := m.restoreDockerVolumes(stackName, drivePath); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, err)
|
|
dataErr = err
|
|
}
|
|
if err := m.stackProvider.RecreateStackDefinitionFromUnit(stackName, composeDir, fullEnv); err != nil {
|
|
return fmt.Errorf("recreating %s from unit: %w", stackName, err)
|
|
}
|
|
// F17: the captured .sql dump is the authoritative logical DB state — replay it AFTER the volume
|
|
// restore, so the dump WINS over any volume-tar copy of the database.
|
|
// R-47: the replay happens with ONLY the database service up. This used to run after
|
|
// RecreateStackFromUnit had already brought the WHOLE stack up, letting the application rebuild
|
|
// schema objects underneath the replay (H4, DIAG-immich-restore-round2-2026-07-19).
|
|
if hasDumps {
|
|
if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
} else if _, err := m.reimportDBDumpsCtx(stackName, nsRoot); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
}
|
|
}
|
|
if err := m.stackProvider.StartStack(stackName); err != nil {
|
|
return fmt.Errorf("starting %s after restore from unit: %w", stackName, err)
|
|
}
|
|
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
|
|
m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err)
|
|
}
|
|
|
|
if dataErr != nil {
|
|
return fmt.Errorf("restore of %s from unit completed with data errors: %w", stackName, dataErr)
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s", stackName)
|
|
return nil
|
|
}
|