7c05b59708
gates / gates (push) Successful in 24s
179 Hungarian sentences were built deep inside a package with fmt.Errorf and printed by whoever caught them: too late to translate where they are shown, too early where they are made. Every one now carries its key across that gap. ZERO Hungarian error literals remain. util.MsgError does three things at once, each earned: - Error() is the Hungarian, byte for byte, so every un-converted printer is unchanged; - errors.Is answers for the kind AND for a wrapped cause (KindErrorf dropped the cause); - an error ARGUMENT renders recursively, so "formázás sikertelen: %w" translates whole. A foreign error — restic, docker, ssh, the stdlib — prints verbatim. It is not ours. 76 display sites go through errText, and TestNoErrErrorInPageOutput convicts any that do not. memoryVerdict returns an error rather than a sentence, so the deploy's 409 and the household's language come from one value; UpdateRefusal gained a Cause to carry it. Plurals, one rule, stated once: a key with .one/.other takes its COUNT first. Not a per-call-site flag — the producer somebody forgot would read "3 app is not running". The guard caught a real key collision (alert.deadapp.one) the day the rule landed. TWO DEFECTS FOUND IN MY OWN TOOLING, recorded rather than quietly fixed. The bulk converter silently dropped multi-line concatenations, damaging 7 producers — and the parity gate could not see it, because every surviving fragment WAS a real base literal while the CALL had lost text; two behaviour tests caught it. And the counting script was case-sensitive, so it said "0 left" while five remained. MinAgent: 0.131.0 (unchanged). No hub release needed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
464 lines
24 KiB
Go
464 lines
24 KiB
Go
package backup
|
|
|
|
import (
|
|
"fmt"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// reconcileRestoreSecrets merges the recovery unit's non-secret env with the secrets recovered from
|
|
// the unit itself (D5) and from the guest's own app.yaml, and applies the FAIL-CLOSED data-key gate.
|
|
// It is the safety-critical heart of Phase 2b and is deliberately a pure function (no I/O) so it can
|
|
// be exhaustively unit-tested — the D5 source arrives as an ARGUMENT, not as a read.
|
|
//
|
|
// Policy:
|
|
// - Regenerate NOTHING here. Secrets come from the unit (portable class) or the guest (the rest).
|
|
// - A missing DATA-ENCRYPTING key (`dataKeyNames`) is FATAL: regenerating it would render the
|
|
// restored data unreadable, so we refuse and tell the operator to do a PBS whole-guest restore.
|
|
// D5 means the key is normally IN the unit — but "normally" is not a reason to soften the gate.
|
|
// - A missing resettable secret is NON-fatal: returned in `missing` so the caller can warn or
|
|
// regenerate it (O4). No data is lost.
|
|
//
|
|
// PRECEDENCE — the UNIT WINS over the guest when both hold a value for the same name.
|
|
//
|
|
// This is not arbitrary and it is not "newest wins". The unit's secrets are captured in the SAME run
|
|
// as the dumps beside them (runVolumeDumps → captureAllRecoveryUnits, backup.go), so the unit's value
|
|
// is the one that MATCHES THE DATA ABOUT TO BE RESTORED, whereas the guest's value is merely the most
|
|
// recent. Where they disagree the guest's has been rotated since the capture, and preferring it is
|
|
// precisely the data-loss bug:
|
|
// - a rotated data-encrypting key does not decrypt data encrypted with the old one;
|
|
// - a rotated DB password does not match the scram/mysql hash inside the restored data directory
|
|
// (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so the app cannot reach its own rows.
|
|
//
|
|
// The restore persists fullEnv back to the guest's app.yaml (RecreateStackDefinitionFromUnit), so
|
|
// unit-wins also leaves the guest consistent with the data now on disk.
|
|
func reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets map[string]string, secretNames, dataKeyNames []string) (fullEnv map[string]string, missing []string, err error) {
|
|
fullEnv = make(map[string]string, len(nonSecretEnv)+len(secretNames))
|
|
for k, v := range nonSecretEnv {
|
|
fullEnv[k] = v
|
|
}
|
|
// resolve applies the precedence: unit first, guest only as a fallback.
|
|
resolve := func(n string) (string, bool) {
|
|
if v, ok := unitSecrets[n]; ok && v != "" {
|
|
return v, true
|
|
}
|
|
if v, ok := guestSecrets[n]; ok && v != "" {
|
|
return v, true
|
|
}
|
|
return "", false
|
|
}
|
|
have := func(n string) bool {
|
|
_, ok := resolve(n)
|
|
return ok
|
|
}
|
|
for _, n := range secretNames {
|
|
if v, ok := resolve(n); ok {
|
|
fullEnv[n] = v
|
|
} else {
|
|
missing = append(missing, n)
|
|
}
|
|
}
|
|
// Fail-closed: any unrecoverable data-encrypting key aborts the restore.
|
|
var missingDataKeys []string
|
|
for _, dk := range dataKeyNames {
|
|
if !have(dk) {
|
|
missingDataKeys = append(missingDataKeys, dk)
|
|
}
|
|
}
|
|
if len(missingDataKeys) > 0 {
|
|
return nil, missing, fmt.Errorf(
|
|
"refusing to restore: data-encrypting key(s) %v are in NEITHER the recovery unit nor the guest's app.yaml — "+
|
|
"a PBS whole-guest restore is required first (regenerating the key would render stored data unreadable)",
|
|
missingDataKeys)
|
|
}
|
|
return fullEnv, missing, nil
|
|
}
|
|
|
|
// readUnitEnv parses a recovery unit's app.yaml and SPLITS it into the plain config env and the
|
|
// secrets the unit carries (D5), using the manifest's portable-secret names as the discriminator.
|
|
//
|
|
// The split is driven by the MANIFEST, not by guessing from key names: the manifest and the app.yaml
|
|
// are captured together and checksummed together, so they cannot disagree about which entries are
|
|
// secrets. A schema-1 unit has no portable names, so everything lands in nonSecret — exactly the
|
|
// pre-D5 behaviour, which is what makes an old unit still restorable.
|
|
func readUnitEnv(path string, portableNames []string) (nonSecret, unitSecrets map[string]string) {
|
|
nonSecret, unitSecrets = map[string]string{}, map[string]string{}
|
|
data, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nonSecret, unitSecrets
|
|
}
|
|
var s strippedAppYaml
|
|
if yaml.Unmarshal(data, &s) != nil || s.Env == nil {
|
|
return nonSecret, unitSecrets
|
|
}
|
|
isPortable := make(map[string]bool, len(portableNames))
|
|
for _, n := range portableNames {
|
|
isPortable[n] = true
|
|
}
|
|
for k, v := range s.Env {
|
|
if isPortable[k] {
|
|
unitSecrets[k] = v
|
|
continue
|
|
}
|
|
nonSecret[k] = v
|
|
}
|
|
return nonSecret, unitSecrets
|
|
}
|
|
|
|
// hasReplayableDump reports whether dumpDir holds a .sql dump that the replay could actually use.
|
|
// The `pre-restore-` safety dumps are EXCLUDED: they live in the same directory (deliberately — an
|
|
// undo the customer cannot see is not much of one) but are never a replay source, so counting them
|
|
// would arm the DB-only phase, and its fail-closed gate, for an app that has nothing to replay.
|
|
func hasReplayableDump(dumpDir string) bool {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
return false
|
|
}
|
|
for _, e := range entries {
|
|
if e.IsDir() || filepath.Ext(e.Name()) != ".sql" {
|
|
continue
|
|
}
|
|
if !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// UnitRestoreResult is what a local recovery-unit restore actually did, so the surface can STATE it
|
|
// rather than report a bare completion.
|
|
//
|
|
// It exists for the same reason OffsiteReconstituteResult does, and it is the same lesson arriving on
|
|
// the other path: on 2026-08-21 an opengist restore reported "Restore-from-unit completed" over a unit
|
|
// holding manifest.json and compose/ and nothing else, and no screen could have told the customer that
|
|
// no data had been returned (R-353).
|
|
//
|
|
// The Manifest* counts are carried BECAUSE zero-replayed has two causes and they are not the same
|
|
// fact. A unit that lists no dumps means THE BACKUP held no data. A unit that lists dumps none of which
|
|
// replayed means something is wrong and the customer's live data was left untouched. R-355 is the
|
|
// standing rule this obeys: a claim about the APP must never be inferred from a counter — and here it
|
|
// is not merely unproven but unprovable, because 07-backup-architecture §6.3 records that an app's
|
|
// canonical .sql could be absent from the unit for reasons that have nothing to do with whether the app
|
|
// has a database (R-361 destroyed exactly that file for four months).
|
|
type UnitRestoreResult struct {
|
|
// VolumesReplayed is how many named-volume tars were unpacked into live Docker volumes.
|
|
VolumesReplayed int
|
|
// DBsReplayed is how many .sql dumps were imported. Never inferred from the presence of a database
|
|
// service — only a completed import increments it.
|
|
DBsReplayed int
|
|
// ManifestVolumes is len(manifest.VolumeDumps): what the unit CLAIMS it captured. The gap between
|
|
// this and VolumesReplayed is the whole of Scenario C.
|
|
ManifestVolumes int
|
|
// ManifestDBs is len(manifest.DBDumps): the same claim for the database leg.
|
|
ManifestDBs int
|
|
// CountsUnknown marks a run whose counts could not be established AT ALL — today the one case is
|
|
// the no-unit fallback to RestoreApp, which returns only an error and whose signature is
|
|
// deliberately out of scope.
|
|
//
|
|
// IT EXISTS BECAUSE THE ZERO VALUE WOULD OTHERWISE LIE. Without it a fallback restore that really
|
|
// replayed three volumes reports VolumesReplayed=0 / ManifestVolumes=0 — the shape the surface
|
|
// reads as „ez a mentés csak a beállításokat tartalmazta, adatot nem". That is a confident false
|
|
// statement, and precisely the R-88 failure direction (degrade to NO DATA rather than to UNKNOWN)
|
|
// this whole change exists to remove. An unknown must be carried, never drawn as a zero.
|
|
CountsUnknown bool
|
|
}
|
|
|
|
// RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit.
|
|
//
|
|
// It reads the unit manifest, takes the portable secrets from the UNIT and the rest from the guest's
|
|
// live app.yaml (unit wins — see reconcileRestoreSecrets), applies the fail-closed data-key gate,
|
|
// restores the named-volume data from the unit's tars, then restores the app's definition from the unit
|
|
// and redeploys it with the reconstructed env (re-pulling the pinned image). If no unit exists it falls
|
|
// back to the legacy volume-only RestoreApp.
|
|
//
|
|
// D5: this no longer needs the guest. A restore with the guest's app.yaml absent succeeds, which is
|
|
// pinned by TestRestoreFromRecoveryUnitWithGuestAbsent — the withheld class is regenerated (O4) and
|
|
// only a data key missing from BOTH sources still refuses.
|
|
//
|
|
// R-102: this is now the thin caller. It names the PRIMARY unit — the app's own drive,
|
|
// backups/primary/<stack> — and hands it to RestoreFromRecoveryUnitAt, which holds the whole body.
|
|
// An unresolvable drive path is still refused inside …At, in the same place and with the same
|
|
// message, so the order of the checks a caller can observe is unchanged.
|
|
func (m *Manager) RestoreFromRecoveryUnit(stackName string) (UnitRestoreResult, error) {
|
|
return m.RestoreFromRecoveryUnitAt(stackName, m.primaryUnitDirFor(stackName))
|
|
}
|
|
|
|
// primaryUnitDirFor names the PRIMARY unit a keep-side restore opens. For a deployed app that is
|
|
// backups/primary/<stack> on its own drive. For a REMOVED app (R-487) the drive is no longer known
|
|
// — GetAppDrivePath falls back to the system path — so a unit kept on a data drive was unreachable
|
|
// and the restore silently took the volume-only fallback. It is now found where it sits.
|
|
func (m *Manager) primaryUnitDirFor(stackName string) string {
|
|
if m.stackProvider != nil {
|
|
if _, deployed := m.stackProvider.GetStackComposePath(stackName); !deployed {
|
|
if u, found := m.RemovedAppUnitFor(stackName); found {
|
|
return u.UnitDir
|
|
}
|
|
}
|
|
}
|
|
return RecoveryUnitPath(m.namespaceRoot(m.GetAppDrivePath(stackName)), stackName)
|
|
}
|
|
|
|
// RestoreFromRecoveryUnitAt is RestoreFromRecoveryUnit with an EXPLICIT recovery-unit directory.
|
|
//
|
|
// R-102. ONE implementation, two callers — the same rule restoreDockerVolumesFrom states beside
|
|
// itself in restore.go, and for the same reason: a second copy of this body is exactly how the local
|
|
// path and the off-site path drifted apart until nothing compared them.
|
|
//
|
|
// The reason it exists: Tier-2 mirrors the app's whole recovery unit to
|
|
// <dest>/backups/secondary/<stack>/recovery-unit/ on every run, and until now every reader of a unit
|
|
// could only name a path under backups/primary/. So in the one failure Tier-2 exists for — the
|
|
// primary drive is lost, taking the primary unit with it — the surviving copy could not be opened by
|
|
// any action in the product (07-backup-architecture §6.3, §7.2).
|
|
//
|
|
// THE SOURCE MOVES; THE DESTINATION DOES NOT. unitDir changes only where the manifest, the compose
|
|
// capture, the .sql dumps and the volume tars are READ from. The app's data is written back to the
|
|
// live Docker volumes and the live database container, and its definition to the guest, exactly as
|
|
// before — a restore that also relocated the app's data would be a migration, not a restore.
|
|
//
|
|
// Everything else is pinned and unchanged: the mutation order (stop → volumes → recreate → DB-only
|
|
// start → replay → start, R-47), the secret reconciliation with unit-over-guest precedence and the
|
|
// fail-closed data-key gate, and the no-unit fallback to RestoreApp with its CountsUnknown handling.
|
|
func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRestoreResult, error) {
|
|
return m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{})
|
|
}
|
|
|
|
// UnitRestoreOptions carries the caller's EXPLICIT consent for a restore this package would
|
|
// otherwise refuse. It exists because of R-538, and it has exactly one member for now.
|
|
type UnitRestoreOptions struct {
|
|
// AcceptMissingFiles lets a unit restore proceed for an app whose own files live on the data
|
|
// drive and are therefore NOT in the unit. The default — zero value, every existing caller — is
|
|
// to REFUSE, because the run that produced this option replayed a database over files that were
|
|
// never captured, reported „3 adatkötet és az adatbázis visszaállítva", and left Nextcloud
|
|
// listing five photos that returned `Sabre\DAV\Exception\NotFound`.
|
|
//
|
|
// It is a per-call argument and never a field on the Manager: a consent that outlives the act it
|
|
// was given for is not consent.
|
|
AcceptMissingFiles bool
|
|
}
|
|
|
|
// ErrUnitLacksFileLegs is the refusal R-538 asks for. It names the app and the paths that are NOT in
|
|
// the unit, so the caller can build an honest sentence without re-deriving anything.
|
|
type ErrUnitLacksFileLegs struct {
|
|
Stack string
|
|
Paths []string
|
|
}
|
|
|
|
func (e *ErrUnitLacksFileLegs) Error() string {
|
|
return fmt.Sprintf("%s: the recovery unit carries no copy of the app's files on the data drive (%s) — refusing to replay the database over them",
|
|
e.Stack, strings.Join(e.Paths, ", "))
|
|
}
|
|
|
|
// RestoreFromRecoveryUnitAtWith is RestoreFromRecoveryUnitAt with the caller's explicit consent
|
|
// flags. See UnitRestoreOptions.
|
|
func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt UnitRestoreOptions) (UnitRestoreResult, error) {
|
|
var res UnitRestoreResult
|
|
if m.stackProvider == nil {
|
|
return res, fmt.Errorf("stack provider not configured")
|
|
}
|
|
|
|
// R-538 — REFUSE BEFORE ANYTHING IS TOUCHED. This runs before the lock, before the stack is
|
|
// stopped and before a single volume is replaced, because the measured harm was not only the
|
|
// missing files: the replayed database also stopped referencing the app's OWN wastebasket, which
|
|
// still held every byte on the drive. A refusal that has already stopped the app has destroyed
|
|
// the customer's last route while declining to help them.
|
|
//
|
|
// The condition is about the UNIT, not the app class: a unit structurally cannot hold these
|
|
// paths (`CaptureRecoveryUnit` has no file-copy step, `RecoveryManifest` no field for one), and
|
|
// that is true of the Tier-2 mirror of a unit as well — Tier 2's file half is a separate action
|
|
// („Fájlok visszaállítása", RestoreTier2Files), which is exactly what the caller should offer.
|
|
if !opt.AcceptMissingFiles {
|
|
if legs := m.DeclaredDriveFileLegs(stackName); len(legs) > 0 {
|
|
m.logger.Printf("[WARN] [backup] unit restore REFUSED for %s: the unit carries no file leg; %d drive path(s) would be left as they are: %s",
|
|
stackName, len(legs), strings.Join(legs, ", "))
|
|
return res, &ErrUnitLacksFileLegs{Stack: stackName, Paths: legs}
|
|
}
|
|
}
|
|
|
|
m.mu.Lock()
|
|
if m.running {
|
|
m.mu.Unlock()
|
|
return res, fmt.Errorf("backup or restore already in progress")
|
|
}
|
|
m.running = true
|
|
m.mu.Unlock()
|
|
defer func() {
|
|
m.mu.Lock()
|
|
m.running = false
|
|
m.mu.Unlock()
|
|
}()
|
|
|
|
// The DESTINATION side, and it is deliberately still resolved here: RestoreApp (the no-unit
|
|
// fallback below) needs it, and 07-backup-architecture §6.3 records that the restore destination
|
|
// is resolved by the same rule as the capture destination. It is no longer used to derive any
|
|
// SOURCE path — that is what unitDir is for.
|
|
drivePath := m.GetAppDrivePath(stackName)
|
|
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
|
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
|
|
manifest := readManifest(UnitManifestFile(unitDir))
|
|
if manifest == nil {
|
|
m.logger.Printf("[WARN] [backup] No readable recovery unit for %s at %s — falling back to volume-only restore", stackName, unitDir)
|
|
m.mu.Lock()
|
|
m.running = false // RestoreApp re-acquires the running flag
|
|
m.mu.Unlock()
|
|
// The fallback has no unit and no manifest, and `RestoreApp` returns only an error — so the
|
|
// counts here are genuinely UNKNOWN, not zero. Reporting zero would tell a customer whose
|
|
// volumes were just restored that their backup held no data.
|
|
res.CountsUnknown = true
|
|
return res, m.RestoreApp(stackName, "")
|
|
}
|
|
|
|
// R-353: what the unit CLAIMS it holds, recorded before any mutation. Read from the manifest that
|
|
// was just parsed above, so the claim and the outcome are counted from the same document.
|
|
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
|
|
|
|
composeDir := UnitComposeDir(unitDir)
|
|
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
|
|
|
|
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
|
|
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
|
|
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
|
|
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
|
|
// reconcileRestoreSecrets), then the fail-closed gate.
|
|
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
|
|
fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
|
|
return res, err
|
|
}
|
|
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
|
|
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
|
|
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
|
|
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
|
|
// never generated: the fail-closed gate above already refused if one was missing, and the
|
|
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
|
|
//
|
|
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
|
|
// generator run means the secret was empty at capture AND absent from the guest.
|
|
//
|
|
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
|
|
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
|
|
// regenerated value leaves the app unable to authenticate against its own restored rows while the
|
|
// dump replay, which uses the container's local trust socket, still reports success. The old wording
|
|
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
|
|
// DB/root-password fields and is now scoped to what is actually true.
|
|
if len(missing) > 0 {
|
|
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
|
|
for _, dk := range manifest.DataKeyEnvVars {
|
|
dataKeySet[dk] = true
|
|
}
|
|
var generated, unresolved []string
|
|
for _, name := range missing {
|
|
if !dataKeySet[name] && m.generateSecret != nil {
|
|
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
|
|
fullEnv[name] = v
|
|
generated = append(generated, name)
|
|
continue
|
|
}
|
|
}
|
|
unresolved = append(unresolved, name)
|
|
}
|
|
if len(generated) > 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
|
|
stackName, generated)
|
|
}
|
|
if len(unresolved) > 0 {
|
|
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
|
|
stackName, len(unresolved), unresolved)
|
|
}
|
|
}
|
|
// R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with
|
|
// two answers, and "an absent log line is not evidence" — the drill reads this line to prove the
|
|
// secondary mirror, not the primary unit, was the source. It is a path, never a secret.
|
|
m.logger.Printf("[INFO] [backup] Restoring %s from recovery unit %s: images=%d, secrets recovered=%d/%d, data_keys=%d",
|
|
stackName, unitDir, len(manifest.ImagePins), len(manifest.SecretEnvVars)-len(missing), len(manifest.SecretEnvVars), len(manifest.DataKeyEnvVars))
|
|
|
|
// R-47: which compose service holds the database, and is there anything to replay? Resolved from
|
|
// the UNIT's compose, because that file is about to BECOME the live one. Both answers are needed
|
|
// BEFORE the first mutation, so the refusal below leaves the live app completely untouched.
|
|
dbServices, dsErr := DBServiceNames(filepath.Join(composeDir, "docker-compose.yml"))
|
|
if dsErr != nil {
|
|
// "cannot tell" is not "no database" — leave it empty and let the gate decide.
|
|
m.logger.Printf("[WARN] [backup] %s: could not read the unit's compose services: %v", stackName, dsErr)
|
|
}
|
|
dbDumpDir := UnitDBDumpDir(unitDir)
|
|
hasDumps := hasReplayableDump(dbDumpDir)
|
|
if hasDumps && len(dbServices) == 0 {
|
|
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the unit's compose", stackName)
|
|
return res, util.MsgError("err.backup.az_adatbazis_szolgaltatas_nem_azonosithato_a", stackName)
|
|
}
|
|
|
|
// Stop, restore named-volume data, recreate the definition, replay the DB with ONLY the database
|
|
// service running, and only then start the whole stack.
|
|
// F17: surface a data-restore failure instead of swallowing it (we still bring the app back up).
|
|
var dataErr error
|
|
if err := m.stackProvider.StopStack(stackName); err != nil {
|
|
m.logger.Printf("[WARN] [backup] could not stop %s before restore: %v (continuing)", stackName, err)
|
|
}
|
|
// R-353: the count is captured even when the replay errors — a partial replay is a fact the
|
|
// customer's sentence has to be built from, and discarding it on the error path is how Scenario C
|
|
// would end up wearing Scenario B's wording.
|
|
// R-354's volume-REPLAY seam, reused here rather than a second one being invented. It defaults to
|
|
// the real restoreDockerVolumesFrom, so production behaviour is byte-for-byte what it was; what it
|
|
// buys is that R-102's acceptance test can assert WHICH directory the tars came out of without a
|
|
// Docker daemon. For 40 of the 53 catalogue apps that archive is the entire dataset, so "the
|
|
// mirror was the source" has to be provable for the volume leg too, not only for the env and the
|
|
// database.
|
|
volReplay := m.volumeReplayFrom
|
|
if volReplay == nil {
|
|
volReplay = m.restoreDockerVolumesFrom
|
|
}
|
|
replayed, volErr := volReplay(stackName, UnitVolumeDumpDir(unitDir))
|
|
res.VolumesReplayed = replayed
|
|
if volErr != nil {
|
|
m.logger.Printf("[ERROR] [backup] volume restore for %s: %v", stackName, volErr)
|
|
dataErr = volErr
|
|
}
|
|
if err := m.stackProvider.RecreateStackDefinitionFromUnit(stackName, composeDir, fullEnv); err != nil {
|
|
return res, fmt.Errorf("recreating %s from unit: %w", stackName, err)
|
|
}
|
|
// F17: the captured .sql dump is the authoritative logical DB state — replay it AFTER the volume
|
|
// restore, so the dump WINS over any volume-tar copy of the database.
|
|
// R-47: the replay happens with ONLY the database service up. This used to run after
|
|
// RecreateStackFromUnit had already brought the WHOLE stack up, letting the application rebuild
|
|
// schema objects underneath the replay (H4, DIAG-immich-restore-round2-2026-07-19).
|
|
if hasDumps {
|
|
if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
} else if n, err := m.reimportDBDumpsAtCtx(stackName, dbDumpDir); err != nil {
|
|
res.DBsReplayed = n // partial credit: whatever imported before the failure really did import
|
|
m.logger.Printf("[ERROR] [backup] DB re-import for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
} else {
|
|
res.DBsReplayed = n
|
|
}
|
|
}
|
|
if err := m.stackProvider.StartStack(stackName); err != nil {
|
|
return res, fmt.Errorf("starting %s after restore from unit: %w", stackName, err)
|
|
}
|
|
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
|
|
m.logger.Printf("[WARN] [backup] %s restored but health check failed: %v", stackName, err)
|
|
}
|
|
|
|
if dataErr != nil {
|
|
return res, fmt.Errorf("restore of %s from unit completed with data errors: %w", stackName, dataErr)
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s — %d volume(s) of %d listed, %d database(s) of %d listed",
|
|
stackName, res.VolumesReplayed, res.ManifestVolumes, res.DBsReplayed, res.ManifestDBs)
|
|
// Slice 4: the app is back on its unit's definition and data and was started — the route back an
|
|
// update hold names. Lift that hold (and only that kind; see clearUpdateHoldAfterRestore).
|
|
m.clearUpdateHoldAfterRestore(stackName)
|
|
return res, nil
|
|
}
|