package backup import ( "errors" "fmt" "gitea.dooplex.hu/admin/felhom-controller/internal/util" "os" "path/filepath" "strings" "time" "gopkg.in/yaml.v3" ) // reconcileRestoreSecrets merges the recovery unit's non-secret env with the secrets recovered from // the unit itself (D5) and from the guest's own app.yaml, and applies the FAIL-CLOSED data-key gate. // It is the safety-critical heart of Phase 2b and is deliberately a pure function (no I/O) so it can // be exhaustively unit-tested — the D5 source arrives as an ARGUMENT, not as a read. // // Policy: // - Regenerate NOTHING here. Secrets come from the unit (portable class) or the guest (the rest). // - A missing DATA-ENCRYPTING key (`dataKeyNames`) is FATAL: regenerating it would render the // restored data unreadable, so we refuse and tell the operator to do a PBS whole-guest restore. // D5 means the key is normally IN the unit — but "normally" is not a reason to soften the gate. // - A missing resettable secret is NON-fatal: returned in `missing` so the caller can warn or // regenerate it (O4). No data is lost. // // PRECEDENCE — the UNIT WINS over the guest when both hold a value for the same name. // // This is not arbitrary and it is not "newest wins". The unit's secrets are captured in the SAME run // as the dumps beside them (runVolumeDumps → captureAllRecoveryUnits, backup.go), so the unit's value // is the one that MATCHES THE DATA ABOUT TO BE RESTORED, whereas the guest's value is merely the most // recent. Where they disagree the guest's has been rotated since the capture, and preferring it is // precisely the data-loss bug: // - a rotated data-encrypting key does not decrypt data encrypted with the old one; // - a rotated DB password does not match the scram/mysql hash inside the restored data directory // (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so the app cannot reach its own rows. // // The restore persists fullEnv back to the guest's app.yaml (RecreateStackDefinitionFromUnit), so // unit-wins also leaves the guest consistent with the data now on disk. func reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets map[string]string, secretNames, dataKeyNames []string) (fullEnv map[string]string, missing []string, err error) { fullEnv = make(map[string]string, len(nonSecretEnv)+len(secretNames)) for k, v := range nonSecretEnv { fullEnv[k] = v } // resolve applies the precedence: unit first, guest only as a fallback. resolve := func(n string) (string, bool) { if v, ok := unitSecrets[n]; ok && v != "" { return v, true } if v, ok := guestSecrets[n]; ok && v != "" { return v, true } return "", false } have := func(n string) bool { _, ok := resolve(n) return ok } for _, n := range secretNames { if v, ok := resolve(n); ok { fullEnv[n] = v } else { missing = append(missing, n) } } // Fail-closed: any unrecoverable data-encrypting key aborts the restore. var missingDataKeys []string for _, dk := range dataKeyNames { if !have(dk) { missingDataKeys = append(missingDataKeys, dk) } } if len(missingDataKeys) > 0 { return nil, missing, fmt.Errorf( "refusing to restore: data-encrypting key(s) %v are in NEITHER the recovery unit nor the guest's app.yaml — "+ "a PBS whole-guest restore is required first (regenerating the key would render stored data unreadable)", missingDataKeys) } return fullEnv, missing, nil } // readUnitEnv parses a recovery unit's app.yaml and SPLITS it into the plain config env and the // secrets the unit carries (D5), using the manifest's portable-secret names as the discriminator. // // The split is driven by the MANIFEST, not by guessing from key names: the manifest and the app.yaml // are captured together and checksummed together, so they cannot disagree about which entries are // secrets. A schema-1 unit has no portable names, so everything lands in nonSecret — exactly the // pre-D5 behaviour, which is what makes an old unit still restorable. func readUnitEnv(path string, portableNames []string) (nonSecret, unitSecrets map[string]string) { nonSecret, unitSecrets = map[string]string{}, map[string]string{} data, err := os.ReadFile(path) if err != nil { return nonSecret, unitSecrets } var s strippedAppYaml if yaml.Unmarshal(data, &s) != nil || s.Env == nil { return nonSecret, unitSecrets } isPortable := make(map[string]bool, len(portableNames)) for _, n := range portableNames { isPortable[n] = true } for k, v := range s.Env { if isPortable[k] { unitSecrets[k] = v continue } nonSecret[k] = v } return nonSecret, unitSecrets } // hasReplayableDump reports whether dumpDir holds a .sql dump that the replay could actually use. // The `pre-restore-` safety dumps are EXCLUDED: they live in the same directory (deliberately — an // undo the customer cannot see is not much of one) but are never a replay source, so counting them // would arm the DB-only phase, and its fail-closed gate, for an app that has nothing to replay. func hasReplayableDump(dumpDir string) bool { entries, err := os.ReadDir(dumpDir) if err != nil { return false } for _, e := range entries { if e.IsDir() || filepath.Ext(e.Name()) != ".sql" { continue } if !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) { return true } } return false } // UnitRestoreResult is what a local recovery-unit restore actually did, so the surface can STATE it // rather than report a bare completion. // // It exists for the same reason OffsiteReconstituteResult does, and it is the same lesson arriving on // the other path: on 2026-08-21 an opengist restore reported "Restore-from-unit completed" over a unit // holding manifest.json and compose/ and nothing else, and no screen could have told the customer that // no data had been returned (R-353). // // The Manifest* counts are carried BECAUSE zero-replayed has two causes and they are not the same // fact. A unit that lists no dumps means THE BACKUP held no data. A unit that lists dumps none of which // replayed means something is wrong and the customer's live data was left untouched. R-355 is the // standing rule this obeys: a claim about the APP must never be inferred from a counter — and here it // is not merely unproven but unprovable, because 07-backup-architecture §6.3 records that an app's // canonical .sql could be absent from the unit for reasons that have nothing to do with whether the app // has a database (R-361 destroyed exactly that file for four months). type UnitRestoreResult struct { // VolumesReplayed is how many named-volume tars were unpacked into live Docker volumes. VolumesReplayed int // DBsReplayed is how many .sql dumps were imported. Never inferred from the presence of a database // service — only a completed import increments it. DBsReplayed int // ManifestVolumes is len(manifest.VolumeDumps): what the unit CLAIMS it captured. The gap between // this and VolumesReplayed is the whole of Scenario C. ManifestVolumes int // ManifestDBs is len(manifest.DBDumps): the same claim for the database leg. ManifestDBs int // CountsUnknown marks a run whose counts could not be established AT ALL — today the one case is // the no-unit fallback to RestoreApp, which returns only an error and whose signature is // deliberately out of scope. // // IT EXISTS BECAUSE THE ZERO VALUE WOULD OTHERWISE LIE. Without it a fallback restore that really // replayed three volumes reports VolumesReplayed=0 / ManifestVolumes=0 — the shape the surface // reads as „ez a mentés csak a beállításokat tartalmazta, adatot nem". That is a confident false // statement, and precisely the R-88 failure direction (degrade to NO DATA rather than to UNKNOWN) // this whole change exists to remove. An unknown must be carried, never drawn as a zero. CountsUnknown bool // DataPins / DataAt (v0.275.0, R-696) — the versions the restored DATA belongs to and when it was // written, from the unit's `data` block; the definition started with it is the one they name. Empty // when the unit's versions are unknown (VersionsUnknown). DataPins []string DataAt time.Time // VersionChanged — the app now runs DataPins, which differ from what it ran before the restore (a // restore of an older version). The page then says which version came back (`07` §6.6). VersionChanged bool // VersionsUnknown — a unit written before v0.275.0 (or with an unstamped data file): restored as // before, with a WARN naming it. VersionsUnknown bool } // ErrUnitVersionMismatch is the KIND of the refusal of a restore whose unit would start its data with a // definition the data does not belong to (v0.275.0, R-696): a unit whose files were written under different // pins (Mixed), or whose compose/ names other pins than its data. Raised BEFORE anything is touched; // branch with errors.Is. The customer sentence is the bundle's. var ErrUnitVersionMismatch = errors.New("the recovery unit's definition is not the one its data belongs to") // unitVersionCheck is A3's rule, as a pure function of the manifest and the unit's own compose file: // known and matching → the pins to report; unknown → (nil, nil) and the caller WARNs; mixed or // mismatched → the refusal. func unitVersionCheck(stackName string, man *RecoveryManifest, composeDir string) ([]string, error) { if man == nil || man.Data == nil { return nil, nil } if man.Data.Mixed { return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_versions_mixed", stackName) } def := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")) if !samePins(def, man.Data.ImagePins) { return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_version_mismatch", stackName, PinsVersion(def), PinsVersion(man.Data.ImagePins)) } return man.Data.ImagePins, nil } // RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit. // // It reads the unit manifest, takes the portable secrets from the UNIT and the rest from the guest's // live app.yaml (unit wins — see reconcileRestoreSecrets), applies the fail-closed data-key gate, // restores the named-volume data from the unit's tars, then restores the app's definition from the unit // and redeploys it with the reconstructed env (re-pulling the pinned image). If no unit exists it falls // back to the legacy volume-only RestoreApp. // // D5: this no longer needs the guest. A restore with the guest's app.yaml absent succeeds, which is // pinned by TestRestoreFromRecoveryUnitWithGuestAbsent — the withheld class is regenerated (O4) and // only a data key missing from BOTH sources still refuses. // // R-102: this is now the thin caller. It names the PRIMARY unit — the app's own drive, // backups/primary/ — and hands it to RestoreFromRecoveryUnitAt, which holds the whole body. // An unresolvable drive path is still refused inside …At, in the same place and with the same // message, so the order of the checks a caller can observe is unchanged. func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, error) { return m.RestoreFromRecoveryUnitAt(stackName, m.primaryUnitDirFor(stackName)) } // primaryUnitDirFor names the PRIMARY unit a keep-side restore opens. For a deployed app that is // backups/primary/ on its own drive. For a REMOVED app (R-487) the drive is no longer known // — GetAppDrivePath falls back to the system path — so a unit kept on a data drive was unreachable // and the restore silently took the volume-only fallback. It is now found where it sits. // // R-690 (2026-09-25): "removed" is asked with isStackDeployed — the removed-app list's OWN predicate. // It used to be GetStackComposePath's ok, which in production is true for EVERY catalog app (every // template is a stack), so this branch never ran on a box: nextcloud's unit on a data drive was // missed, the restore took the volume-only fallback, and the app came back with no env and no // database. Pinned by TestR690_RemovedUnitFoundWhenTheStackStillExists (production-shaped provider). func (m *Manager) primaryUnitDirFor(stackName string) string { if m.stackProvider != nil && !m.isStackDeployed(stackName) { if u, found := m.RemovedAppUnitFor(stackName); found { return u.UnitDir } } return RecoveryUnitPath(m.namespaceRoot(m.GetAppDrivePath(stackName)), stackName) } // RestoreFromRecoveryUnitAt is RestoreFromRecoveryUnit with an EXPLICIT recovery-unit directory. // // R-102. ONE implementation, two callers — the same rule restoreDockerVolumesFrom states beside // itself in restore.go, and for the same reason: a second copy of this body is exactly how the local // path and the off-site path drifted apart until nothing compared them. // // The reason it exists: Tier-2 mirrors the app's whole recovery unit to // /backups/secondary//recovery-unit/ on every run, and until now every reader of a unit // could only name a path under backups/primary/. So in the one failure Tier-2 exists for — the // primary drive is lost, taking the primary unit with it — the surviving copy could not be opened by // any action in the product (07-backup-architecture §6.3, §7.2). // // THE SOURCE MOVES; THE DESTINATION DOES NOT. unitDir changes only where the manifest, the compose // capture, the .sql dumps and the volume tars are READ from. The app's data is written back to the // live Docker volumes and the live database container, and its definition to the guest, exactly as // before — a restore that also relocated the app's data would be a migration, not a restore. // // Everything else is pinned and unchanged: the mutation order (stop → volumes → recreate → DB-only // start → replay → start, R-47), the secret reconciliation with unit-over-guest precedence and the // fail-closed data-key gate, and the no-unit fallback to RestoreApp with its CountsUnknown handling. func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRestoreResult, error) { return m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{}) } // UnitRestoreOptions carries the caller's EXPLICIT consent for a restore this package would // otherwise refuse. It exists because of R-538, and it has exactly one member for now. type UnitRestoreOptions struct { // AcceptMissingFiles lets a unit restore proceed for an app whose own files live on the data // drive and are therefore NOT in the unit. The default — zero value, every existing caller — is // to REFUSE, because the run that produced this option replayed a database over files that were // never captured, reported „3 adatkötet és az adatbázis visszaállítva", and left Nextcloud // listing five photos that returned `Sabre\DAV\Exception\NotFound`. // // It is a per-call argument and never a field on the Manager: a consent that outlives the act it // was given for is not consent. AcceptMissingFiles bool } // ErrUnitLacksFileLegs is the refusal R-538 asks for. It names the app and the paths that are NOT in // the unit, so the caller can build an honest sentence without re-deriving anything. type ErrUnitLacksFileLegs struct { Stack string Paths []string } func (e *ErrUnitLacksFileLegs) Error() string { return fmt.Sprintf("%s: the recovery unit carries no copy of the app's files on the data drive (%s) — refusing to replay the database over them", e.Stack, strings.Join(e.Paths, ", ")) } // RestoreFromRecoveryUnitAtWith is RestoreFromRecoveryUnitAt with the caller's explicit consent // flags. See UnitRestoreOptions. func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt UnitRestoreOptions) (UnitRestoreResult, error) { var res UnitRestoreResult if m.stackProvider == nil { return res, fmt.Errorf("stack provider not configured") } // R-538 — REFUSE BEFORE ANYTHING IS TOUCHED. This runs before the lock, before the stack is // stopped and before a single volume is replaced, because the measured harm was not only the // missing files: the replayed database also stopped referencing the app's OWN wastebasket, which // still held every byte on the drive. A refusal that has already stopped the app has destroyed // the customer's last route while declining to help them. // // The condition is about the UNIT, not the app class: a unit structurally cannot hold these // paths (`CaptureRecoveryUnit` has no file-copy step, `RecoveryManifest` no field for one), and // that is true of the Tier-2 mirror of a unit as well — Tier 2's file half is a separate action // („Fájlok visszaállítása", RestoreTier2Files), which is exactly what the caller should offer. if !opt.AcceptMissingFiles { if legs := m.DeclaredDriveFileLegs(stackName); len(legs) > 0 { m.logger.Printf("[WARN] [backup] unit restore REFUSED for %s: the unit carries no file leg; %d drive path(s) would be left as they are: %s", stackName, len(legs), strings.Join(legs, ", ")) return res, &ErrUnitLacksFileLegs{Stack: stackName, Paths: legs} } } m.mu.Lock() if m.running { m.mu.Unlock() return res, fmt.Errorf("backup or restore already in progress") } m.running = true m.mu.Unlock() defer func() { m.mu.Lock() m.running = false m.mu.Unlock() }() // The DESTINATION side, and it is deliberately still resolved here: RestoreApp (the no-unit // fallback below) needs it, and 07-backup-architecture §6.3 records that the restore destination // is resolved by the same rule as the capture destination. It is no longer used to derive any // SOURCE path — that is what unitDir is for. drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { return res, fmt.Errorf("cannot determine drive path for %s", stackName) } manifest := readManifest(UnitManifestFile(unitDir)) if manifest == nil { m.logger.Printf("[WARN] [backup] No readable recovery unit for %s at %s — falling back to volume-only restore", stackName, unitDir) m.mu.Lock() m.running = false // RestoreApp re-acquires the running flag m.mu.Unlock() // The fallback has no unit and no manifest, and `RestoreApp` returns only an error — so the // counts here are genuinely UNKNOWN, not zero. Reporting zero would tell a customer whose // volumes were just restored that their backup held no data. res.CountsUnknown = true return res, m.RestoreApp(stackName, "") } // R-353: what the unit CLAIMS it holds, recorded before any mutation. Read from the manifest that // was just parsed above, so the claim and the outcome are counted from the same document. res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps) composeDir := UnitComposeDir(unitDir) // v0.275.0 (R-696, `07` §6.6) — A RESTORE NEVER STARTS DATA WITH A DEFINITION IT DOES NOT BELONG TO. // The unit keeps the definition of its data (data_versions.go); this refuses, before anything is // touched, a unit where the two disagree. Measured before the fix (A1, 9202): the unit's PostgreSQL 18 // definition over a 16 datadir tar — the tar replaced the live volume, 18 refused it, the app stayed down. dataPins, verr := unitVersionCheck(stackName, manifest, composeDir) if verr != nil { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: the unit's definition %v, its data %v (mixed=%v) — a restore never starts data with a definition it does not belong to; nothing was touched", stackName, ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")), manifest.Data.ImagePins, manifest.Data.Mixed) return res, verr } if dataPins == nil { res.VersionsUnknown = true m.logger.Printf("[WARN] [backup] Restore %s from %s: the unit does not record which versions wrote its data (written before v0.275.0, or an unstamped file) — restoring its definition as captured, as before", stackName, unitDir) } else { res.DataPins = dataPins res.DataAt, _ = manifest.Data.DataTime() if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok { if live := definitionPins(info); len(live) > 0 && !samePins(live, dataPins) { res.VersionChanged = true m.logger.Printf("[INFO] [backup] Restore %s: the data belongs to %v (written %s); the app ran %v — it comes back at its data's version, and the update climbs from there", stackName, dataPins, manifest.Data.At, live) } } } fullEnv, missing, err := m.unitRestoreEnv(stackName, composeDir, manifest) if err != nil { return res, err } // R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with // two answers, and "an absent log line is not evidence" — the drill reads this line to prove the // secondary mirror, not the primary unit, was the source. It is a path, never a secret. m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit %s: images=%d, secrets recovered=%d/%d, data_keys=%d", stackName, unitDir, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars)) // R-47: which compose service holds the database, and is there anything to replay? Resolved from // the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed // BEFORE the first mutation, so the refusal below leaves the live app completely untouched. dbServices, dsErr := DBServiceNames(filepath.Join(composeDir, "docker-compose.yml")) if dsErr != nil { // "cannot tell" is not "no database" — leave it empty and let the gate decide. m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr) } dbDumpDir := UnitDBDumpDir(unitDir) hasDumps := hasReplayableDump(dbDumpDir) if hasDumps && len(dbServices) == 0 { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName) return res, util.MsgError("err.backup.az_adatbazis_szolgaltatas_nem_azonosithato_a", stackName) } // R-640: a cut-off database copy is refused HERE, before the first mutation, so the live app is // untouched — loading it would report success over an empty (PostgreSQL) or half-replaced // (MariaDB) database. if bad := incompleteDumps(dbDumpDir); len(bad) > 0 { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: incomplete database copy %v (no completion marker)", stackName, bad) return res, util.MsgError("err.backup.adatbazis_masolat_csonka_nem_indult", stackName) } // Stop, restore named-volume data, recreate the definition, replay the DB with ONLY the database // service running, and only then start the whole stack. // F17: surface a data-restore failure instead of swallowing it (we still bring the app back up). var dataErr error if err := m.stackProvider.StopStack(stackName); err != nil { m.logger.Printf("[WARN] [backup] could not stop %s before restore: %v (continuing)", stackName, err) } // R-353: the count is captured even when the replay errors — a partial replay is a fact the // customer's sentence has to be built from, and discarding it on the error path is how Scenario C // would end up wearing Scenario B's wording. // R-354's volume-REPLAY seam, reused here rather than a second one being invented. It defaults to // the real restoreDockerVolumesFrom, so production behaviour is byte-for-byte what it was; what it // buys is that R-102's acceptance test can assert WHICH directory the tars came out of without a // Docker daemon. For 40 of the 53 catalogue apps that archive is the entire dataset, so "the // mirror was the source" has to be provable for the volume leg too, not only for the env and the // database. volReplay := m.volumeReplayFrom if volReplay == nil { volReplay = m.restoreDockerVolumesFrom } replayed, volErr := volReplay(stackName, UnitVolumeDumpDir(unitDir)) res.VolumesReplayed = replayed if volErr != nil { m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, volErr) dataErr = volErr } if err := m.stackProvider.RecreateStackDefinitionFromUnit(stackName, composeDir, fullEnv); err != nil { return res, fmt.Errorf("recreating %s from unit: %w", stackName, err) } // F17: the captured .sql dump is the authoritative logical DB state — replay it AFTER the volume // restore, so the dump WINS over any volume-tar copy of the database. // R-47: the replay happens with ONLY the database service up. This used to run after // RecreateStackFromUnit had already brought the WHOLE stack up, letting the application rebuild // schema objects underneath the replay (H4, DIAG-immich-restore-round2-2026-07-19). if hasDumps { if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil { m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err) if dataErr == nil { dataErr = err } } else if n, err := m.reimportDBDumpsAtCtx(stackName, dbDumpDir); err != nil { res.DBsReplayed = n // partial credit: whatever imported before the failure really did import m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err) if dataErr == nil { dataErr = err } } else { res.DBsReplayed = n } } if err := m.stackProvider.StartStack(stackName); err != nil { return res, fmt.Errorf("starting %s after restore from unit: %w", stackName, err) } if err := m.waitForHealthy(stackName, 90*time.Second); err != nil { m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err) } if dataErr != nil { return res, fmt.Errorf("restore of %s from unit completed with data errors: %w", stackName, dataErr) } m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s — %d volume(s) of %d listed, %d database(s) of %d listed", stackName, res.VolumesReplayed, res.ManifestVolumes, res.DBsReplayed, res.ManifestDBs) // Slice 4: the app is back on its unit's definition and data and was started — the route back an // update hold names. Lift that hold (and only that kind; see clearUpdateHoldAfterRestore). m.clearUpdateHoldAfterRestore(stackName) return res, nil } // unitRestoreEnv rebuilds the env a restore starts the unit's definition with: the unit's plain config, // its portable secrets (D5), the guest's for the withheld class, the fail-closed data-key gate, and a // generated replacement for a missing RESETTABLE secret (O4). ONE implementation for the unit restore and, // since v0.275.0, the off-site restore that brings an app back at its snapshot's version. Values are // never logged. func (m *Manager) unitRestoreEnv(stackName, composeDir string, manifest *RecoveryManifest) (fullEnv map[string]string, missing []string, err error) { nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars) // D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The // guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the // fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole // point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see // reconcileRestoreSecrets), then the fail-closed gate. guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars) fullEnv, missing, err = reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars) if err != nil { m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err) return nil, missing, err } // O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank // string" → exit 1). Generate a replacement via the deploy flow's generator instead — // RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands // encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are // never generated: the fail-closed gate above already refused if one was missing, and the // generator itself refuses data-key fields (defense-in-depth). Values are never logged. // // D5 shrinks this path to the rare case: the portable class now comes from the unit, so a // generator run means the secret was empty at capture AND absent from the guest. // // It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data // directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a // regenerated value leaves the app unable to authenticate against its own restored rows while the // dump replay, which uses the container's local trust socket, still reports success. The old wording // here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18 // DB/root-password fields and is now scoped to what is actually true. if len(missing) > 0 { dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars)) for _, dk := range manifest.DataKeyEnvVars { dataKeySet[dk] = true } var generated, unresolved []string for _, name := range missing { if !dataKeySet[name] && m.generateSecret != nil { if v, ok := m.generateSecret(stackName, name); ok && v != "" { fullEnv[name] = v generated = append(generated, name) continue } } unresolved = append(unresolved, name) } if len(generated) > 0 { m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data", stackName, generated) } if len(unresolved) > 0 { m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually", stackName, len(unresolved), unresolved) } } return fullEnv, missing, nil }