package backup import ( "fmt" "os" "path/filepath" "strings" "time" "gopkg.in/yaml.v3" ) // reconcileRestoreSecrets merges the recovery unit's non-secret env with the secrets recovered from // the guest's own app.yaml, and applies the FAIL-CLOSED data-key gate. It is the safety-critical heart // of Phase 2b and is deliberately a pure function (no I/O) so it can be exhaustively unit-tested. // // Policy (per the Phase 2 design — see REPORT/CHANGELOG): // - Regenerate NOTHING. Every secret comes from the guest (live rootfs, or PBS whole-guest restore). // - A missing DATA-ENCRYPTING key (`dataKeyNames`) is FATAL: regenerating it would render the // restored data unreadable, so we refuse and tell the operator to do a PBS whole-guest restore. // - A missing resettable secret (DB password, admin password) is NON-fatal: it's returned in // `missing` so the caller can warn; the app may simply need a credential reset, no data is lost. func reconcileRestoreSecrets(nonSecretEnv, recoveredSecrets map[string]string, secretNames, dataKeyNames []string) (fullEnv map[string]string, missing []string, err error) { fullEnv = make(map[string]string, len(nonSecretEnv)+len(secretNames)) for k, v := range nonSecretEnv { fullEnv[k] = v } have := func(n string) bool { v, ok := recoveredSecrets[n] return ok && v != "" } for _, n := range secretNames { if have(n) { fullEnv[n] = recoveredSecrets[n] } else { missing = append(missing, n) } } // Fail-closed: any unrecoverable data-encrypting key aborts the restore. var missingDataKeys []string for _, dk := range dataKeyNames { if !have(dk) { missingDataKeys = append(missingDataKeys, dk) } } if len(missingDataKeys) > 0 { return nil, missing, fmt.Errorf( "refusing to restore: data-encrypting key(s) %v could not be recovered from the guest's app.yaml — "+ "a PBS whole-guest restore is required first (regenerating the key would render stored data unreadable)", missingDataKeys) } return fullEnv, missing, nil } // readStrippedEnv parses the non-secret env from a recovery unit's secret-stripped app.yaml. func readStrippedEnv(path string) map[string]string { data, err := os.ReadFile(path) if err != nil { return map[string]string{} } var s strippedAppYaml if yaml.Unmarshal(data, &s) != nil || s.Env == nil { return map[string]string{} } return s.Env } // hasReplayableDump reports whether dumpDir holds a .sql dump that the replay could actually use. // The `pre-restore-` safety dumps are EXCLUDED: they live in the same directory (deliberately — an // undo the customer cannot see is not much of one) but are never a replay source, so counting them // would arm the DB-only phase, and its fail-closed gate, for an app that has nothing to replay. func hasReplayableDump(dumpDir string) bool { entries, err := os.ReadDir(dumpDir) if err != nil { return false } for _, e := range entries { if e.IsDir() || filepath.Ext(e.Name()) != ".sql" { continue } if !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) { return true } } return false } // RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit + the guest's own secrets. // // It reads the unit manifest, recovers the secret values from the guest's live app.yaml, applies the // fail-closed data-key gate, restores the named-volume data from the unit's tars, then restores the // app's definition from the unit and redeploys it with the reconstructed env (re-pulling the pinned // image). No secret is ever regenerated, and no secret is read from the unit. If no unit exists it // falls back to the legacy volume-only RestoreApp. func (m *Manager) RestoreFromRecoveryUnit(stackName string) error { if m.stackProvider == nil { return fmt.Errorf("stack provider not configured") } m.mu.Lock() if m.running { m.mu.Unlock() return fmt.Errorf("backup or restore already in progress") } m.running = true m.mu.Unlock() defer func() { m.mu.Lock() m.running = false m.mu.Unlock() }() drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { return fmt.Errorf("cannot determine drive path for %s", stackName) } nsRoot := m.namespaceRoot(drivePath) manifest := readManifest(RecoveryUnitManifestPath(nsRoot, stackName)) if manifest == nil { m.logger.Printf("[WARN] [backup] No recovery unit for %s — falling back to volume-only restore", stackName) m.mu.Lock() m.running = false // RestoreApp re-acquires the running flag m.mu.Unlock() return m.RestoreApp(stackName, "") } composeDir := RecoveryUnitComposePath(nsRoot, stackName) nonSecretEnv := readStrippedEnv(filepath.Join(composeDir, "app.yaml")) // Recover secrets from the GUEST (never the unit), then apply the fail-closed gate. recovered := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars) fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, recovered, manifest.SecretEnvVars, manifest.DataKeyEnvVars) if err != nil { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err) return err } // O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank // string" → exit 1). Generate a replacement via the deploy flow's generator instead — // RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands // encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are // never generated: the fail-closed gate above already refused if one was missing, and the // generator itself refuses data-key fields (defense-in-depth). Values are never logged. if len(missing) > 0 { dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars)) for _, dk := range manifest.DataKeyEnvVars { dataKeySet[dk] = true } var generated, unresolved []string for _, name := range missing { if !dataKeySet[name] && m.generateSecret != nil { if v, ok := m.generateSecret(stackName, name); ok && v != "" { fullEnv[name] = v generated = append(generated, name) continue } } unresolved = append(unresolved, name) } if len(generated) > 0 { m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); stored data is unaffected (no data-key involved)", stackName, generated) } if len(unresolved) > 0 { m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually", stackName, len(unresolved), unresolved) } } m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit: images=%d, secrets recovered=%d/%d, data_keys=%d", stackName, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars)) // R-47: which compose service holds the database, and is there anything to replay? Resolved from // the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed // BEFORE the first mutation, so the refusal below leaves the live app completely untouched. dbServices, dsErr := DBServiceNames(filepath.Join(composeDir, "docker-compose.yml")) if dsErr != nil { // "cannot tell" is not "no database" — leave it empty and let the gate decide. m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr) } hasDumps := hasReplayableDump(AppDBDumpPath(nsRoot, stackName)) if hasDumps && len(dbServices) == 0 { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName) return fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stackName) } // Stop, restore named-volume data, recreate the definition, replay the DB with ONLY the database // service running, and only then start the whole stack. // F17: surface a data-restore failure instead of swallowing it (we still bring the app back up). var dataErr error if err := m.stackProvider.StopStack(stackName); err != nil { m.logger.Printf("[WARN] [backup] could not stop %s before restore: %v (continuing)", stackName, err) } if err := m.restoreDockerVolumes(stackName, drivePath); err != nil { m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, err) dataErr = err } if err := m.stackProvider.RecreateStackDefinitionFromUnit(stackName, composeDir, fullEnv); err != nil { return fmt.Errorf("recreating %s from unit: %w", stackName, err) } // F17: the captured .sql dump is the authoritative logical DB state — replay it AFTER the volume // restore, so the dump WINS over any volume-tar copy of the database. // R-47: the replay happens with ONLY the database service up. This used to run after // RecreateStackFromUnit had already brought the WHOLE stack up, letting the application rebuild // schema objects underneath the replay (H4, DIAG-immich-restore-round2-2026-07-19). if hasDumps { if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil { m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err) if dataErr == nil { dataErr = err } } else if _, err := m.reimportDBDumpsCtx(stackName, nsRoot); err != nil { m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err) if dataErr == nil { dataErr = err } } } if err := m.stackProvider.StartStack(stackName); err != nil { return fmt.Errorf("starting %s after restore from unit: %w", stackName, err) } if err := m.waitForHealthy(stackName, 90*time.Second); err != nil { m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err) } if dataErr != nil { return fmt.Errorf("restore of %s from unit completed with data errors: %w", stackName, dataErr) } m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s", stackName) return nil }