package backup import ( "context" "fmt" "os" "os/exec" "path/filepath" "sort" "strings" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // Offsite reconstitution (R-43, v0.148.0) — the leg that was missing. // // Until v0.148.0 NO offsite path could restore a database. The two „visszaállítás" buttons staged // files into a scratch folder and never touched postgres; the place-to-live button merged only the // files MISSING from the live tree (`rsync --ignore-existing`) and never replayed a dump. For a // DB-indexed app — most of the catalog — that combination cannot bring content back: the bytes // return and the application still cannot see them, because its index lives in the database. // Measured live on 2026-07-19 (DIAG-immich-restore-2026-07-19): 11 photos, files intact on disk, // timeline empty, two "successful" restores that merged 0 files. // // ReconstituteFromOffsite is the honest version of that operation: it takes the CHOSEN snapshot's // coherent pair and makes the live app equal to it — files overwritten to the snapshot's version, // database replayed from the same snapshot's dump, app restarted. It is deliberately a different // function from PlaceOffsiteRestore rather than a flag on it, because the two have opposite file // semantics and conflating them is exactly how the missing-only merge came to be presented as a // restore. // // Two invariants hold throughout: // // - NOTHING IS EVER DELETED. The file copy overwrites and adds; it never carries `--delete`. A // file the customer created after the snapshot survives the restore as an extra. That is the // house boundary — a restore that silently removed newer work would be a data-loss event // wearing a recovery button's label. // - THE UNDO EXISTS BEFORE THE ACT. A safety dump of the live database is written, and verified // present on disk, BEFORE anything is stopped, overwritten or replayed. If that dump cannot be // taken, the whole operation refuses with zero changes — a replay whose previous state was not // captured is not a restore, it is an overwrite with no way back. // offsitePreDump runs the coherence pre-phase's dump leg (nil seam → runDBDumpsInternal, which also // refreshes the recovery units so the manifests enumerate the dumps just written). Extracted as a // seam because the ORDER — dumps strictly before the restic capture — is the entire mechanism of // R-44, and an ordering guarantee that no test can observe is one refactor away from silently // reverting to the behaviour that produced DIAG-immich-restore-2026-07-19. func (m *Manager) offsitePreDump(ctx context.Context) error { if m.offsitePreDumpFn != nil { return m.offsitePreDumpFn(ctx) } return m.runDBDumpsInternal(ctx) } // SetOffsitePreDumpFn overrides the offsite dump pre-phase (tests; no Docker needed). func (m *Manager) SetOffsitePreDumpFn(fn func(ctx context.Context) error) { m.offsitePreDumpFn = fn } // preRestoreDumpPrefix marks the safety dumps taken immediately before a reconstitution. They live // in the app's own unit db-dumps dir so `ListDumpFiles` surfaces them beside the regular dumps — // they ARE the undo, and an undo the customer cannot see is not much of one. The regular replay // loop matches `-.sql` exactly, so a prefixed file is never mistaken for a source. const preRestoreDumpPrefix = "pre-restore-" // OffsiteReconstituteResult reports what a reconstitution actually did, so the flash can state an // OUTCOME instead of a mechanism. Every field here exists because the v0.147 flash could not say it. type OffsiteReconstituteResult struct { SnapshotID string FilesPlaced int DBsReplayed int // VolumesReplayed (R-354) is how many named-volume archives came back from the snapshot. It is on // the result for the same reason every other field here is: so the OUTCOME can state what // happened rather than a mechanism. Without it the message said "5 fájl visszaállítva" over a // restore that had silently dropped a 1.4 MB volume archive — a true sentence leaving a false // impression, which is the shape this surface keeps having removed from it. VolumesReplayed int SafetyDump string // path of the pre-restore dump (the undo), "" when the app has no DB // RolledBack (R-379) is true when the database replay FAILED and this run put the customer's own // pre-restore copy back. It is on the result rather than inferred from the error, because "the // restore failed" and "your data is as it was" are two different facts and the surface has to be // able to say both. RolledBack bool DumpsAt time.Time // when the snapshot's DB half was taken (zero = unknown/legacy unit) OffsiteRunID string // "" for a pre-v0.148 snapshot — an unverified pair Skewed bool // the snapshot carries no coherence stamp: files and DB may differ in age LooksEmpty bool // R-44 sniff on the dump about to be replayed // Placement (R-351) is what the backup recorded about where this app's data lived, compared // against where this restore actually wrote. Carried on the RESULT and not only on the refusal, // so a restore that proceeded into a different destination says so in its own outcome rather // than reporting a bare success — a warning beside a success is read as a success, so the // difference has to survive into the message. Placement PlacementCheck } // fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite). // Deliberately NOT placeCopier(): that one is `--ignore-existing`, whose whole purpose is to leave // live files alone, which is precisely what a full restore must not do. func (m *Manager) fullPlaceCopier() func(src, dst string) (int, error) { if m.offboxFullPlaceCopier != nil { return m.offboxFullPlaceCopier } return rsyncRestoreOverwrite } // rsyncRestoreOverwrite copies src over dst: `rsync -a --itemize-changes`, with NO // `--ignore-existing` (a changed file becomes the snapshot's version) and NO `--delete` (an extra // file at dst survives). Returns the number of regular files transferred. func rsyncRestoreOverwrite(src, dst string) (int, error) { if err := os.MkdirAll(dst, 0755); err != nil { return 0, fmt.Errorf("mkdir %s: %w", dst, err) } ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute) defer cancel() cmd := exec.CommandContext(ctx, "rsync", "-a", "--itemize-changes", strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/") out, err := cmd.CombinedOutput() if err != nil { return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out))) } return countRestoredFiles(string(out)), nil } // safetyDumpSet is what ONE reconstitution's undo consists of: the stamp that identifies this run's // files, and one written path per database the app has. // // R-379: it exists because `writeSafetyDump` used to return only the FIRST path, and the rollback // added in v0.220.0 must re-apply EVERY database's undo or it restores one and leaves the other // half-written — the defect it exists to close, one database over. The stamp is the IDENTITY: three // runs against `docmost` on 2026-08-22 left three `pre-restore-*` files in the same directory, so // matching on the prefix would replay an arbitrary older state. Match on this stamp, never on the // prefix, never on age or size. type safetyDumpSet struct { Stamp string // 20060102T150405Z — this run's, and only this run's Files []safetyDumpFile // one per database, in discovery order } // safetyDumpFile pairs an undo file with the database it came from, so the rollback can hand each // dump back to the container it belongs to instead of guessing from the filename. type safetyDumpFile struct { DB DiscoveredDB Path string } // First returns the first written path, or "" — the value the pre-v0.220.0 signature returned, kept // because the customer-facing message names one file and changing that is not this task. func (s safetyDumpSet) First() string { if len(s.Files) == 0 { return "" } return s.Files[0].Path } // undoCopyPhrase is the sentence the DOUBLE-FAILURE message uses to describe the customer's undo // copy — and it says what is TRUE, which is the whole of R-383. // // THE BUG THIS EXISTS TO KILL. The double-failure branch ended with „a korábbi állapot mentése // megvan: " — *the previous state's backup EXISTS* — built from the path `writeSafetyDump` // returned and WITHOUT ever asking the filesystem. But one of the two ways `rollbackSafetyDump` fails // is that the file is not there, so in exactly the case that sentence is printed it is most likely to // be false. Measured twice live, on v0.220.2 and v0.221.1. // // A false reassurance is worse than no sentence: it is read at the moment the customer is deciding // whether their data is recoverable, and it points support at a file that is not there. // // WHY NOT SIMPLY DROP THE FILENAME. R-351's lesson: a refusal that names nothing forces a person to // remember what the product already knows. The operator needs the path either way — to fetch the // file, or to look for it. So the absent case still names WHERE it should have been, and says // plainly that it is not there. // // The check is `os.Stat`, deliberately not a readability or integrity test: this runs at the end of a // failed restore on a machine that may be unwell, and the honest claim available here is presence. // A zero-length file is reported as MISSING — a 0-byte dump restores nothing, and calling it present // is the same false reassurance one step smaller. func undoCopyPhrase(set safetyDumpSet) string { var present, absent []string for _, f := range set.Files { if f.Path == "" { continue } if st, err := os.Stat(f.Path); err == nil && !st.IsDir() && st.Size() > 0 { present = append(present, filepath.Base(f.Path)) continue } absent = append(absent, filepath.Base(f.Path)) } switch { case len(present) > 0 && len(absent) == 0: return "a korábbi állapot mentése megvan: " + strings.Join(present, ", ") case len(present) > 0: // Partial: name both halves. An app with two databases whose undo is half there is a // different situation from either whole one, and support must not have to guess which. return "a korábbi állapot mentése RÉSZBEN van meg — megvan: " + strings.Join(present, ", ") + "; HIÁNYZIK: " + strings.Join(absent, ", ") case len(absent) > 0: return "a korábbi állapot mentését NEM találjuk a helyén (" + strings.Join(absent, ", ") + ")" default: return "a korábbi állapotról nem készült menthető másolat" } } // writeSafetyDump dumps every live database of stack into the app's unit db-dumps dir under the // `pre-restore-` prefix, and returns the SET it wrote. Returns (zero, nil) when the app has no // database at all — a no-DB app has nothing to undo and must flow exactly as it did before // v0.148.0 (no dump, no replay, no behaviour change). // // A discovered database that CANNOT be dumped is a hard error: it means the undo would not exist. func (m *Manager) writeSafetyDump(ctx context.Context, stackName, nsRoot string) (safetyDumpSet, error) { discover := m.discoverDBs if discover == nil { discover = func(ctx context.Context) ([]DiscoveredDB, error) { return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames()) } } dbs, err := discover(ctx) if err != nil { return safetyDumpSet{}, fmt.Errorf("a biztonsági mentés előtt nem sikerült felderíteni az adatbázisokat: %w", err) } var mine []DiscoveredDB for _, db := range dbs { if db.StackName == stackName { mine = append(mine, db) } } if len(mine) == 0 { return safetyDumpSet{}, nil // no DB → nothing to undo → scenario E flows unchanged } dumpDir := AppDBDumpPath(nsRoot, stackName) if err := os.MkdirAll(dumpDir, 0755); err != nil { return safetyDumpSet{}, fmt.Errorf("a biztonsági mentés könyvtára nem hozható létre: %w", err) } set := safetyDumpSet{Stamp: time.Now().UTC().Format("20060102T150405Z")} for _, db := range mine { // R-361: the undo copy is dumped STRAIGHT to its own name. It used to be dumped to the app's // canonical `-.sql` and renamed afterwards, and the comment here asserted that // the rename meant it "can never overwrite the app's real dump". THAT WAS FALSE AS WRITTEN: // `DumpOne` writes the canonical name, so every safety dump destroyed the app's own backup and // then moved it away — leaving the app with NO database backup until the next nightly run, and // a local restore-from-unit in that window telling the customer the app never had a database. // Measured live on demo-hp 2026-08-22: `docmost` and `bookstack` both held only `pre-restore-*` // files and no canonical dump. // // THE INVARIANT, AND HOW IT IS NOW ENFORCED: nothing but the app's own dump is ever written to // the canonical name, because the safety dump never names it — `DumpOneTo` takes the final path // and derives its own `.tmp` from it, so neither the destination nor the scratch file can // collide with a nightly dump running beside it. Pinned by // TestR361_SafetyDumpLeavesTheCanonicalDumpByteIdentical. safe := filepath.Join(dumpDir, fmt.Sprintf("%s%s-%s-%s.sql", preRestoreDumpPrefix, set.Stamp, stackName, db.DBType)) res := m.dumpForSafety(ctx, db, safe) if res.Error != nil { return safetyDumpSet{}, fmt.Errorf("a jelenlegi adatbázis biztonsági mentése sikertelen (%s): %w — a visszaállítás nem indult el", db.ContainerName, res.Error) } // EVERY file, not just the first — R-379, and the reason is on safetyDumpSet. set.Files = append(set.Files, safetyDumpFile{DB: db, Path: safe}) m.logger.Printf("[INFO] [offbox] %s: pre-restore safety dump written → %s (%s)", stackName, filepath.Base(safe), humanizeBytes(res.Size)) } return set, nil } // shortID trims a docker id for logs. Never used for identity — only for reading. func shortID(id string) string { if len(id) > 12 { return id[:12] } return id } // maxUndoCopiesPerApp is how many `pre-restore-` copies an app keeps. // // THREE, and the reasoning rather than a number pulled from the air. One is not enough: the case // that needs an undo is a restore that went wrong, and the second-guess attempt is exactly when the // customer reaches for the state before the FIRST attempt. Many is not free: they live inside the // recovery unit, so every one is also mirrored to Tier 2 AND pushed off-site permanently — four // accumulated on `docmost` in a single afternoon on 2026-08-22 (135 KB + 135 KB + 141 KB + 138 KB), // each of them forever. Three keeps two prior attempts and bounds the off-site growth. const maxUndoCopiesPerApp = 3 // pruneUndoCopies keeps the newest maxUndoCopiesPerApp undo copies for an app and removes the rest. // // DELIBERATELY NOT CALLED FROM THE RESTORE PATH. A delete on the failure path is how an undo goes // missing at exactly the moment it is needed; this runs from the capture side, where nothing is // depending on the files right now. It is called AFTER a successful capture, never before one. // // Ordering is by the stamp IN THE FILENAME, not by mtime and never by size: mtime moves when a file // is copied or a filesystem is restored, and the stamp is the identity writeSafetyDump assigned. // The newest is never a deletion candidate even if the list is somehow malformed. func (m *Manager) pruneUndoCopies(dumpDir, stack string) { entries, err := os.ReadDir(dumpDir) if err != nil { return } type undo struct{ name, stamp string } var undos []undo for _, e := range entries { if e.IsDir() || !strings.HasSuffix(e.Name(), ".sql") { continue } base := strings.TrimSuffix(e.Name(), ".sql") if !strings.HasPrefix(base, preRestoreDumpPrefix) { continue } after := strings.TrimPrefix(base, preRestoreDumpPrefix) i := strings.Index(after, "-") if i <= 0 { continue // not the shape writeSafetyDump writes — leave it alone rather than guess } undos = append(undos, undo{name: e.Name(), stamp: after[:i]}) } if len(undos) <= maxUndoCopiesPerApp { return } sort.Slice(undos, func(a, b int) bool { return undos[a].stamp > undos[b].stamp }) // newest first for _, u := range undos[maxUndoCopiesPerApp:] { p := filepath.Join(dumpDir, u.name) if err := os.Remove(p); err != nil { m.logger.Printf("[WARN] [backup] %s: could not prune old undo copy %s: %v", stack, u.name, err) continue } m.logger.Printf("[INFO] [backup] %s: pruned old undo copy %s (keeping the newest %d)", stack, u.name, maxUndoCopiesPerApp) } } // RestoreHoldFor reports whether an app is being held stopped after a failed restore + failed // rollback, and returns the customer-facing reason. Every start path consults this — the customer's // button, the app-stop Recover() starter, and the boot reconciler — because a hold that only one // path honours is not a hold. // // Nil settings ⇒ NOT held. That direction is deliberate and is the opposite of the usual fail-closed // rule: with no settings there is no hold recorded, so refusing every start would strand every app // on a misconfigured box. The write side logs loudly when it cannot persist (see // holdAppAfterFailedRollback), which is where that case is caught. func (m *Manager) RestoreHoldFor(stack string) (bool, string) { if m == nil || m.settings == nil { return false, "" } h, ok := m.settings.GetRestoreHold(stack) if !ok { return false, "" } when := h.At if t, err := time.Parse(time.RFC3339, h.At); err == nil { when = t.Format("2006-01-02 15:04") } return true, fmt.Sprintf("a(z) %s adatainak visszaállítása %s-kor megszakadt, és a korábbi állapotot sem sikerült visszatölteni. "+ "Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek tovább. Vedd fel velünk a kapcsolatot", stack, when) } // holdAppAfterFailedRollback records the R-379/R-380 hold and makes sure nothing restarts the app // behind our back. // // OPERATOR RULING, 2026-08-22: when the replay fails AND the rollback fails, the app is HELD // STOPPED rather than started. A running app on a half-written database lets the customer type into // it, and that turns a recoverable state into a permanent one. The alternative — start it and mark // it — was considered and declined. // // It ENDS the app-stop marker deliberately. The marker means "owed a restart"; a held app is not // owed one, and leaving the marker active would have Recover() start the broken app at the next // controller boot. The hold is the thing that persists, not the marker. func (m *Manager) holdAppAfterFailedRollback(stack string, replayErr, rollbackErr error) { if m.settings == nil { m.logger.Printf("[ERROR] [offbox] %s: cannot persist the restore hold — no settings wired; the app is stopped but NOTHING will refuse a restart", stack) return } h := settings.RestoreHold{ Stack: stack, At: time.Now().UTC().Format(time.RFC3339), } if replayErr != nil { h.ReplayError = replayErr.Error() } if rollbackErr != nil { h.RollbackErr = rollbackErr.Error() } if err := m.settings.SetRestoreHold(h); err != nil { m.logger.Printf("[ERROR] [offbox] %s: persisting the restore hold FAILED: %v — the app is stopped and unguarded", stack, err) } // The app is not owed a restart; it is deliberately held. See the doc comment. if m.appStop != nil { m.appStop.End() } if m.restoreHoldNotify != nil { m.restoreHoldNotify(stack, replayErr, rollbackErr) } } // SetRestoreHoldNotify wires the operator notification for a held app (cmd/controller/main.go). func (m *Manager) SetRestoreHoldNotify(fn func(stack string, replayErr, rollbackErr error)) { m.restoreHoldNotify = fn } // rollbackSafetyDump re-applies THIS RUN's undo set, database by database, and is the whole of // R-379's fix: it is the same ImportDump call a person made by hand on 2026-08-22 to recover // `docmost` and `bookstack` after a failed replay, moved into the product. // // It runs with the DB service still up (the replay's own window) and BEFORE any restart, so the // app never observes the half-written state. An error here means the app cannot be trusted to run — // see the hold in ReconstituteFromOffsite. func (m *Manager) rollbackSafetyDump(ctx context.Context, stack string, set safetyDumpSet) error { if len(set.Files) == 0 { return nil } imp := m.rollbackImport if imp == nil { imp = func(ctx context.Context, db DiscoveredDB, path string) error { return ImportDump(ctx, db, path, m.logger, m.isDebug()) } } // RE-DISCOVER THE CONTAINERS. The undo FILE is stable; the container it must be poured into is // NOT. `writeSafetyDump` captured its DiscoveredDB before the stop, and by the time the rollback // runs the stack has been stopped and the DB service re-created — a NEW container id. // // MEASURED LIVE ON demo-hp 2026-08-22, which is the only reason this is here: the first live run // of this code captured `docmost-postgres id=9adbc14f9af6` at 16:05:44, the DB-only start // re-created it as `309795897b82` at 16:05:47, and the rollback's `docker exec` against the dead // id sat in `waitDBReady` until it timed out 30 s later — so the app was HELD for an // infrastructure reason when its data was recoverable. The unit tests could not see it: they // inject the import seam and never touch container identity. `reimportDBDumpsFrom` already // re-discovers for exactly this reason. discover := m.discoverDBs if discover == nil { discover = func(ctx context.Context) ([]DiscoveredDB, error) { return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames()) } } live, dErr := discover(ctx) if dErr != nil { return fmt.Errorf("a visszavonás előtt nem sikerült felderíteni az adatbázisokat: %w", dErr) } liveFor := func(want DiscoveredDB) (DiscoveredDB, bool) { for _, db := range live { if db.StackName == want.StackName && db.DBType == want.DBType { return db, true } } return DiscoveredDB{}, false } for _, f := range set.Files { if _, sErr := os.Stat(f.Path); sErr != nil { return fmt.Errorf("a visszavonáshoz szükséges mentés nem található (%s): %w", filepath.Base(f.Path), sErr) } target, ok := liveFor(f.DB) if !ok { // Fail closed: pouring an undo into a container we cannot identify is worse than saying // we could not do it. return fmt.Errorf("a(z) %s adatbázis-tárolója nem található a visszavonáshoz", f.DB.ContainerName) } if target.ContainerID != f.DB.ContainerID { m.logger.Printf("[DEBUG] [offbox] %s: %s was re-created during the restore (%s → %s) — rolling back into the live container", stack, f.DB.ContainerName, shortID(f.DB.ContainerID), shortID(target.ContainerID)) } m.logger.Printf("[INFO] [offbox] %s: rolling back to the pre-restore state from %s", stack, filepath.Base(f.Path)) if err := imp(ctx, target, f.Path); err != nil { return fmt.Errorf("a korábbi állapot visszaállítása sikertelen (%s): %w", target.ContainerName, err) } } m.logger.Printf("[INFO] [offbox] %s: rollback complete — %d database(s) returned to the pre-restore state", stack, len(set.Files)) return nil } // dumpForSafety is the dump seam for the safety dump (tests inject; nil → the real DumpOneTo). // // R-361: it takes the FINAL PATH, not a directory. A directory argument is what allowed the callee to // choose the canonical name, which is the whole defect. func (m *Manager) dumpForSafety(ctx context.Context, db DiscoveredDB, finalPath string) DumpResult { if m.safetyDumpFn != nil { return m.safetyDumpFn(ctx, db, finalPath) } return DumpOneTo(ctx, db, finalPath, m.logger, m.isDebug()) } // ReconstituteFromOffsite makes the live app equal to a restored full-scratch snapshot: files // overwritten to the snapshot's version (extras survive, nothing deleted), then the snapshot's own // DB dump replayed, with a safety dump of the current database taken first. Requires a completed // FULL scratch restore (RestoreOffboxScratch with full=true). Single-flight. // ackPlacementChange (R-351) is the customer's DELIBERATE acknowledgement that the destination // differs from the one the backup recorded. It is a separate act from the restore's own confirm: // folding it into `confirm=1` would mean one click carried two decisions, which is precisely what // R-48 exists to prevent. func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ackPlacementChange bool) (OffsiteReconstituteResult, error) { var res OffsiteReconstituteResult if !m.OffboxConfigured() { return res, fmt.Errorf("off-box backup not configured") } if !isSafeStackName(stack) { return res, fmt.Errorf("invalid stack name") } if m.stackProvider == nil { return res, fmt.Errorf("stack provider not configured") } if err := m.acquireRunning(); err != nil { return res, fmt.Errorf("egy másik mentési/visszaállítási művelet már fut") } defer m.releaseRunning() scratch, _, err := m.offboxRestoreScratchDir(stack) if err != nil { return res, err } if _, sErr := os.Stat(scratch); sErr != nil { return res, fmt.Errorf("nincs előkészített teljes visszaállítás — futtass előbb egy teljes visszaállítást") } id, paths, err := m.offboxLatestSnapshot(ctx, stack) if err != nil { return res, err } res.SnapshotID = id // R-253: the same sentence the restore page now shows, so the page and the refusal cannot // drift apart again. It is a REFUSAL, not a failure — the data is untouched and the customer // has one step to take. The restore deliberately does NOT deploy the app itself: the // destination is the app's own HDD path, which is a drive the CUSTOMER chooses at deploy // time, and picking it for them is the decision this whole recovery path exists to leave // with them. // R-351: the refusal now NAMES the place the backup recorded, when it can read it. The // prepared scratch already contains the unit, so this is a local file read — no network call, // nothing restored, and it happens on a path that was going to refuse anyway. Telling // somebody to reinstall without telling them where the data belongs is what forced the // 2026-08-21 operator to remember two values the backup already held. // R-356: this refusal used to be reached by `GetStackHDDPath(stack) == ""` — one predicate // answering two questions. It now covers ONLY "the app is not deployed", and it stopped // covering "the app has no drive". The reason the two came apart: the drive choice is the // CUSTOMER's, and 40 of the 53 catalog apps were never offered one — they have no choice to // leave with them, and their data lives on the system data path by design. The R-253 decision // above is untouched for the 13 apps that DO have a drive to get wrong. if !m.isStackDeployed(stack) { if rec := m.recordedPlacementFromScratch(scratch); rec.Known() { return res, fmt.Errorf("a(z) %s nincs telepítve, ezért nincs hová visszaállítani az adatait. "+ "A mentése szerint az adatai itt voltak: %s. Telepítsd újra az alkalmazást (Alkalmazások) "+ "ugyanerre a helyre, utána ez a visszaállítás működni fog", stack, rec.Drive) } return res, fmt.Errorf("a(z) %s nincs telepítve, ezért nincs hová visszaállítani az adatait — "+ "telepítsd újra az alkalmazást (Alkalmazások), utána ez a visszaállítás működni fog", stack) } // The destination is resolved by the SAME rule the capture side used to write this snapshot // (CaptureRecoveryUnit → GetAppDrivePath): the app's drive if it has one, the system data path // otherwise. Anything else and the restore would aim at a different place than the backup came // from, which is the mismatch prompt firing on a box where nothing actually moved. hdd := strings.TrimSpace(m.GetAppDrivePath(stack)) if hdd == "" { // A DIFFERENT failure from the one above, so it gets a different sentence: the app IS // installed, but the box cannot name its own data root (systemDataPath unset). Saying // "nincs telepítve" here would send the customer to reinstall an app that is already // running, and the real fault would stay invisible. return res, fmt.Errorf("a(z) %s telepítve van, de a vezérlő nem tudja megállapítani, hová tartoznak az adatai "+ "(nincs beállítva rendszer-adatterület). Ellenőrizd a tárhely beállításait (Tárhely), utána indítsd újra a visszaállítást", stack) } liveNs := m.namespaceRoot(hdd) // --- R-357: FREE SPACE, BEFORE ANYTHING IS TOUCHED ------------------------------------------ // // This file contained ZERO references to offboxFree until now. The three headroom gates that // existed all guarded NON-destructive paths (offbox_restore.go: the download sizer, the prepare // gate, and PlaceOffsiteRestore's missing-only merge). The one path that stops the customer's app // and overwrites their live data had none. // // Measured on demo-hp 2026-08-21: it stopped the app, ran out of disk part-way, left 2 of 5 planted // items in place and restarted the app — a half-restored dataset presented as a completed restore. // // POSITION IS THE WHOLE FIX. This sits before mapOffsiteRestorePaths, before writeSafetyDump and // well before StopStack, so a refusal costs the customer nothing at all — the app never goes down. // A gate after StopStack would turn a refusal into an outage, which is the shape it exists to // prevent. Scenario D asserts the non-effect (StopStack call count 0), not the error string. // // NO HEADROOM MULTIPLIER, deliberately, and stated so the next reader does not "fix" it: // OffboxRestorePrepareFull uses ×1.1 because it is sizing a DOWNLOAD whose final size it is // predicting. This is a local copy of a tree that already exists on disk, so its size is known // exactly — the same reasoning PlaceOffsiteRestore's gate uses, and this matches it. free, need := m.offboxFree()(liveNs), m.offboxSize()(scratch) switch { case need <= 0: // FAIL CLOSED. Without this the comparison below is `free < 0`, which is false, and an // unmeasurable scratch would sail straight through into the destructive phase — the gate // present and inert, which is worse than no gate because it reads as protection. m.logger.Printf("[ERROR] [offbox] %s: REFUSING the destructive restore — the scratch size could not be measured (scratch=%s)", stack, scratch) return res, fmt.Errorf(offsiteSizeUnknownMsg) case free <= 0: // Same direction for the other probe. The customer sentence is shared with the case above // (the operator asked for one wording); the LOG line above and below is what distinguishes // which probe failed. m.logger.Printf("[ERROR] [offbox] %s: REFUSING the destructive restore — free space on the live namespace could not be measured (liveNs=%s)", stack, liveNs) return res, fmt.Errorf(offsiteSizeUnknownMsg) case free < need: m.logger.Printf("[WARN] [offbox] %s: REFUSING the destructive restore — need %d B, free %d B on %s; the app was NOT stopped", stack, need, free, liveNs) return res, fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free)) } placements, err := mapOffsiteRestorePaths(paths, stack, scratch, liveNs) if err != nil { return res, err // whole-placement refusal (no partial writes) } // Stat pre-pass over EVERY placement before the first copy — an incomplete scratch (e.g. only a // unit-only restore was run) refuses with ZERO copies. for _, pl := range placements { if _, sErr := os.Stat(pl.src); sErr != nil { return res, fmt.Errorf("a teljes visszaállítás hiányos (%s nincs meg) — futtass előbb egy teljes visszaállítást", filepath.Base(pl.src)) } } // The snapshot's coherence stamp, read from the RESTORED unit manifest (not the live one). scratchUnit := "" for _, pl := range placements { if pl.isUnit { scratchUnit = pl.src break } } if scratchUnit == "" { return res, fmt.Errorf("a pillanatképben nincs mentési egység — a visszaállítás nem indítható") } scratchDumpDir := filepath.Join(scratchUnit, "db-dumps") man := readManifest(filepath.Join(scratchUnit, "manifest.json")) if man != nil { res.OffsiteRunID = man.OffsiteRunID if man.DumpsAt != "" { if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil { res.DumpsAt = t } } } // --- WHERE THE BACKUP SAYS THIS DATA LIVED (R-351) ------------------------------------------ // The manifest we just opened has carried `drive` and `namespace_root` since schema 1, and until // now nothing read them back. Compared HERE, before the safety dump and before the first byte is // placed, so the refusal costs nothing and leaves the app completely untouched. // // An UNKNOWN recording (a pre-field unit, or one we could not read) is not a mismatch and does // not refuse: blocking on an absence would strand every older backup, and CheckPlacement returns // that case explicitly rather than letting it fall through as "they match". res.Placement = CheckPlacement(man, hdd, liveNs) if res.Placement.Mismatch && !ackPlacementChange { return res, fmt.Errorf("%s", PlacementMismatchMessage(stack, res.Placement)) } if res.Placement.Mismatch { m.logger.Printf("[WARN] [offbox] %s: restoring into %s, but the backup recorded %s — the customer acknowledged the change", stack, res.Placement.LiveDrive, res.Placement.Recorded.Drive) } // A pre-v0.148 snapshot carries no stamp: its dump was whatever the 02:30 local run left behind, // so the pair's two halves may be hours or days apart. Surfaced, never blocked — the confirm // dialog says so and the safety dump makes it reversible. res.Skewed = res.OffsiteRunID == "" res.LooksEmpty = m.sniffScratchDump(scratchDumpDir, stack) // --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------ // Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir, // so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first // mutation so the refusal below costs nothing. var dbServices []string if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" { svcs, dsErr := DBServiceNames(composePath) if dsErr != nil { // "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse. m.logger.Printf("[WARN] [offbox] %s: could not read the live compose services: %v", stack, dsErr) } dbServices = svcs } // --- THE UNDO, BEFORE THE ACT --------------------------------------------------------------- // Taken while the stack is still UP (a stopped database cannot be dumped) and before a single // byte is overwritten, so a failure here aborts with the live app completely untouched. safetySet, err := m.writeSafetyDump(ctx, stack, liveNs) if err != nil { return res, err } safety := safetySet.First() res.SafetyDump = safety hasDB := safety != "" if hasDB { if _, sErr := os.Stat(safety); sErr != nil { // Fail-closed: never replay when the undo is not verifiably on disk. return res, fmt.Errorf("a biztonsági mentés nem található a lemezen — a visszaállítás biztonsági okból nem indult el") } // Fail-closed (R-47): the app HAS a database but no compose service can be identified to // start alone for the replay. The only alternative would be to start everything and replay // into the race that produced H4 — refusing with the live app untouched is the better outcome. if len(dbServices) == 0 { return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stack) } } // --- FILES ---------------------------------------------------------------------------------- // R-166: mark the stop→restore→start window BEFORE stopping. A controller killed anywhere inside // it used to leave the app down with nothing on disk recording that it was owed a restart — and a // full offsite restore is a LONG window, so this is the shape most likely to be interrupted. if err := m.appStop.Begin("offbox-reconstitute:"+stack, ReasonOffboxReconstitute, []string{stack}); err != nil { return res, fmt.Errorf("a(z) %s leállítása előtti jelölő nem menthető: %w", stack, err) } // restartStack starts the app and clears the marker ONLY when the start actually succeeded — a // failed start leaves the marker so the next startup retries. Every bring-up below goes through // it; a bare StartStack here would clear nothing and strand the marker on the success path. restartStack := func() error { err := m.stackProvider.StartStack(stack) if err == nil { m.appStop.End() } else { // R-330: same rule as the volume-dump path — a restart that was attempted and broke // leaves the app genuinely down, so the alarm suppression must go immediately. m.appStop.ReleaseFailed(stack) } return err } if err := m.stackProvider.StopStack(stack); err != nil { m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err) } copier := m.fullPlaceCopier() for _, pl := range placements { if pl.isUnit { // The live recovery unit is still never overwritten — it is the LOCAL restore path's // source and clobbering it would trade one recovery route for another. THAT reason is // sound and still holds; it is why this skip stays. // // R-354 — THE SECOND HALF OF THIS COMMENT USED TO BE FALSE AND IS CORRECTED HERE. It said // "the snapshot's dump is replayed from the scratch unit instead, so nothing is lost by // skipping it". That was true of the DATABASE dump and false of the VOLUME archives, which // live in the same unit and were replayed by nothing at all. Skipping the placement is // correct; treating the skip as harmless was not. Measured live 2026-08-21: calibre-web's // 1 422 848-byte `calibre_web_config.tar` was in the unit, in the snapshot and in the // verification folder, and the restore reported "5 fájl visszaállítva" without it. // // Both legs are now replayed FROM THE SCRATCH UNIT below — volumes first, then the DB, so // the logical dump still wins over any volume-tar copy of the same database. continue } n, cErr := copier(pl.src, pl.dst) if cErr != nil { // Best-effort bring-up: leaving the app stopped after a partial copy would turn a failed // restore into an outage. if sErr := restartStack(); sErr != nil { m.logger.Printf("[WARN] [offbox] %s: restart after failed placement also failed: %v", stack, sErr) } return res, fmt.Errorf("a(z) %s fájljainak visszaállítása sikertelen: %w", stack, cErr) } res.FilesPlaced += n } // --- NAMED VOLUMES (R-354) ------------------------------------------------------------------ // Replayed from the SCRATCH unit, exactly as the database dump is, and for the same reason: the // live unit is never overwritten by a placement, so the snapshot's copy exists only under the // scratch. Same helper as the local restore path — one implementation, two callers. // // ORDER IS LOAD-BEARING and mirrors RestoreFromRecoveryUnit: volumes FIRST, database after, so a // logical .sql dump still wins over whatever copy of the same database a volume tar happens to // contain. It also has to happen inside the stopped window, because replacing a named volume means // removing it, and Docker refuses that while a container holds it. volReplay := m.volumeReplayFrom if volReplay == nil { volReplay = m.restoreDockerVolumesFrom } nVols, vErr := volReplay(stack, filepath.Join(scratchUnit, "volume-dumps")) res.VolumesReplayed = nVols if vErr != nil { // A partial replay must never read as a completion. Bring the app back up rather than leaving // an outage, then surface it — the same shape the file leg above uses. if sErr := restartStack(); sErr != nil { m.logger.Printf("[WARN] [offbox] %s: restart after failed volume replay also failed: %v", stack, sErr) } return res, fmt.Errorf("a(z) %s adatkötetének visszaállítása sikertelen: %w", stack, vErr) } // --- DATABASE ------------------------------------------------------------------------------- // The DB container must be UP for the replay (ImportDump talks to it with its own discovered // credentials), but NOTHING ELSE may be — R-47. Until v0.153.0 this was a full StartStack, which // gave the application a window to rebuild the very schema objects the dump was about to create: // measured at 2 s on 2026-07-19, and the replay aborted `relation "clip_index" already exists` // under ON_ERROR_STOP=1 (H4). Starting only the database service closes that window entirely. if hasDB { if err := m.stackProvider.StartStackServices(stack, dbServices); err != nil { // Best-effort bring-up: a failed restore must not also be an outage. if sErr := restartStack(); sErr != nil { m.logger.Printf("[WARN] [offbox] %s: full start after failed DB-only start also failed: %v", stack, sErr) } return res, fmt.Errorf("a(z) %s adatbázis-szolgáltatásának indítása sikertelen: %w", stack, err) } n, iErr := m.reimportDBDumpsFrom(ctx, stack, scratchDumpDir) res.DBsReplayed = n if iErr != nil { // --- R-379/R-380: PUT THE CUSTOMER'S OWN COPY BACK --------------------------------- // Until v0.220.0 this branch restarted the app onto a HALF-WRITTEN database and named // the undo file in the message. Measured 2026-08-22 on demo-hp: Postgres was left // emptied and crash-looping; MariaDB was left partly applied while the app reported // `health=healthy`. Both are the same failure — a half state — and the only difference // was whether it looked broken. MariaDB's structural statements are not transactional, // so no engine flag can prevent the half state; putting the undo back is what removes // it. This is the same ImportDump call a person ran by hand that day to recover both // apps, moved into the product. // // The rollback runs BEFORE any restart and with the DB service still up, so the app // never observes the half state. The ORIGINAL replay error is never swallowed: it is // logged here in full and named in the customer's sentence. m.logger.Printf("[ERROR] [offbox] %s: database replay failed, rolling back to the pre-restore state: %v", stack, iErr) if rbErr := m.rollbackSafetyDump(ctx, stack, safetySet); rbErr != nil { // BOTH failed. Do NOT start the app: a running app on a half-written database lets // the customer type into it and makes the damage permanent. Hold it instead — // operator ruling, 2026-08-22. m.logger.Printf("[ERROR] [offbox] %s: ROLLBACK ALSO FAILED (%v) — holding the app stopped; replay error was: %v", stack, rbErr, iErr) m.holdAppAfterFailedRollback(stack, iErr, rbErr) // R-383: the undo copy is DESCRIBED FROM DISK, never from the path alone. See // undoCopyPhrase — this sentence used to assert the file existed in exactly the // branch where a missing file is one of the two causes. return res, fmt.Errorf("a(z) %s adatbázisának visszaállítása sikertelen, és a korábbi állapot visszatöltése sem sikerült. "+ "Az alkalmazást biztonsági okból LEÁLLÍTVA hagytuk, hogy az adatai ne sérüljenek tovább. "+ "Vedd fel velünk a kapcsolatot — %s", stack, undoCopyPhrase(safetySet)) } res.RolledBack = true if sErr := restartStack(); sErr != nil { m.logger.Printf("[WARN] [offbox] %s: start after a successful rollback failed: %v", stack, sErr) } // Says BOTH things. A message that reported only the failure would leave the customer // believing their data was gone when it is back — the omission of a GAIN is as // misleading as the omission of a loss. return res, fmt.Errorf("a(z) %s adatbázisának visszaállítása sikertelen — az adataid visszakerültek a visszaállítás előtti állapotba, "+ "az alkalmazás fut tovább. Ha újra megpróbálnád, előbb vedd fel velünk a kapcsolatot", stack) } } if err := restartStack(); err != nil { return res, fmt.Errorf("a(z) %s újraindítása sikertelen a fájlok visszaállítása után: %w", stack, err) } if err := m.waitForHealthy(stack, 90*time.Second); err != nil { m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err) } // R-382: VolumesReplayed was set above and never printed, so the operator log said // "0 file(s) placed, 1 DB dump(s) replayed" on a run that returned a 52 MB Postgres data // directory — less informative than the customer's own flash, which already named the volumes. m.logger.Printf("[INFO] [offbox] reconstituted %s from snapshot %s: %d file(s) placed, %d volume(s) replayed, %d DB dump(s) replayed, safety dump=%s, skewed=%v", stack, id, res.FilesPlaced, res.VolumesReplayed, res.DBsReplayed, filepath.Base(safety), res.Skewed) return res, nil } // OffsitePairInfo describes the {DB, files} pair sitting in a prepared full-restore scratch, so the // confirm dialog can tell the customer what they are about to restore BEFORE they commit to it. // Everything here is honesty-surface: none of it blocks the operation. type OffsitePairInfo struct { Ready bool DumpsAt time.Time // when the DB half was taken (zero = legacy unit, age unknown) Skewed bool // no coherence stamp → the two halves may be from different times LooksEmpty bool // R-44 sniff: the dump has an accounts table with no rows HasDump bool } // OffsiteScratchPair reads the prepared scratch's unit manifest and reports what the pair looks // like. Cheap and read-only — safe to call from a page render. func (m *Manager) OffsiteScratchPair(stack string) OffsitePairInfo { var info OffsitePairInfo if !isSafeStackName(stack) { return info } scratch, _, err := m.offboxRestoreScratchDir(stack) if err != nil { return info } // The unit sits at //backups/primary/; the old namespace is unknown here, // so find it rather than reconstructing it. unit := findScratchUnitDir(scratch, stack) if unit == "" { return info } info.Ready = true dumpDir := filepath.Join(unit, "db-dumps") if entries, rErr := os.ReadDir(dumpDir); rErr == nil { for _, e := range entries { if !e.IsDir() && filepath.Ext(e.Name()) == ".sql" && !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) { info.HasDump = true break } } } if man := readManifest(filepath.Join(unit, "manifest.json")); man != nil { if man.DumpsAt != "" { if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil { info.DumpsAt = t } } info.Skewed = man.OffsiteRunID == "" } else { info.Skewed = true } if info.HasDump { info.LooksEmpty = m.sniffScratchDump(dumpDir, stack) } return info } // findScratchUnitDir locates `backups/primary/` anywhere under a restored scratch. restic // rebuilds absolute source paths under the target, and the snapshot may have come from a drive that // no longer exists on this box, so the prefix cannot be assumed. func findScratchUnitDir(scratch, stack string) string { found := "" suffix := filepath.Join("backups", "primary", stack) _ = filepath.Walk(scratch, func(path string, fi os.FileInfo, err error) error { if err != nil || found != "" { return nil //nolint:nilerr // a walk error on one branch must not abort the search } if fi.IsDir() && strings.HasSuffix(path, suffix) { found = path } return nil }) return found } // sniffScratchDump runs the R-44 content sniff over the dump about to be replayed. Best-effort and // warn-level: any failure to read simply reports "no warning", because a sniff that blocks a // restore is worse than the skew it describes. func (m *Manager) sniffScratchDump(dumpDir, stack string) bool { entries, err := os.ReadDir(dumpDir) if err != nil { return false } for _, e := range entries { name := e.Name() if e.IsDir() || filepath.Ext(name) != ".sql" || strings.HasPrefix(name, preRestoreDumpPrefix) { continue } dbType := DBTypePostgres if strings.Contains(name, string(DBTypeMariaDB)) { dbType = DBTypeMariaDB } if v := ValidateDump(filepath.Join(dumpDir, name), dbType); v.LooksEmpty { m.logger.Printf("[WARN] [offbox] %s: the snapshot dump %s has no account rows — it may predate the customer's data", stack, name) return true } } return false }