bc278944a3
gates / gates (push) Successful in 25s
app_update_undone / app_update_held events (09 decision 15), on by default and seeded once on existing boxes; R-606 update sentences as key+args rendered per reader; R-646 startup applied-meta backfill for apps current with the catalog; R-620 a disabled notifier WARNs once per event type. Needs hub v0.120.0. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
966 lines
49 KiB
Go
966 lines
49 KiB
Go
package backup
|
||
|
||
import (
|
||
"context"
|
||
"fmt"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||
"os"
|
||
"os/exec"
|
||
"path/filepath"
|
||
"sort"
|
||
"strings"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||
)
|
||
|
||
// Offsite reconstitution (R-43, v0.148.0) — the leg that was missing.
|
||
//
|
||
// Until v0.148.0 NO offsite path could restore a database. The two „visszaállítás" buttons staged
|
||
// files into a scratch folder and never touched postgres; the place-to-live button merged only the
|
||
// files MISSING from the live tree (`rsync --ignore-existing`) and never replayed a dump. For a
|
||
// DB-indexed app — most of the catalog — that combination cannot bring content back: the bytes
|
||
// return and the application still cannot see them, because its index lives in the database.
|
||
// Measured live on 2026-07-19 (DIAG-immich-restore-2026-07-19): 11 photos, files intact on disk,
|
||
// timeline empty, two "successful" restores that merged 0 files.
|
||
//
|
||
// ReconstituteFromOffsite is the honest version of that operation: it takes the CHOSEN snapshot's
|
||
// coherent pair and makes the live app equal to it — files overwritten to the snapshot's version,
|
||
// database replayed from the same snapshot's dump, app restarted. It is deliberately a different
|
||
// function from PlaceOffsiteRestore rather than a flag on it, because the two have opposite file
|
||
// semantics and conflating them is exactly how the missing-only merge came to be presented as a
|
||
// restore.
|
||
//
|
||
// Two invariants hold throughout:
|
||
//
|
||
// - NOTHING IS EVER DELETED. The file copy overwrites and adds; it never carries `--delete`. A
|
||
// file the customer created after the snapshot survives the restore as an extra. That is the
|
||
// house boundary — a restore that silently removed newer work would be a data-loss event
|
||
// wearing a recovery button's label.
|
||
// - THE UNDO EXISTS BEFORE THE ACT. A safety dump of the live database is written, and verified
|
||
// present on disk, BEFORE anything is stopped, overwritten or replayed. If that dump cannot be
|
||
// taken, the whole operation refuses with zero changes — a replay whose previous state was not
|
||
// captured is not a restore, it is an overwrite with no way back.
|
||
|
||
// offsitePreDump runs the coherence pre-phase's dump leg (nil seam → runDBDumpsInternal, which also
|
||
// refreshes the recovery units so the manifests enumerate the dumps just written). Extracted as a
|
||
// seam because the ORDER — dumps strictly before the restic capture — is the entire mechanism of
|
||
// R-44, and an ordering guarantee that no test can observe is one refactor away from silently
|
||
// reverting to the behaviour that produced DIAG-immich-restore-2026-07-19.
|
||
func (m *Manager) offsitePreDump(ctx context.Context) error {
|
||
if m.offsitePreDumpFn != nil {
|
||
return m.offsitePreDumpFn(ctx)
|
||
}
|
||
return m.runDBDumpsInternal(ctx)
|
||
}
|
||
|
||
// SetOffsitePreDumpFn overrides the offsite dump pre-phase (tests; no Docker needed).
|
||
func (m *Manager) SetOffsitePreDumpFn(fn func(ctx context.Context) error) { m.offsitePreDumpFn = fn }
|
||
|
||
// preRestoreDumpPrefix marks the safety dumps taken immediately before a reconstitution. They live
|
||
// in the app's own unit db-dumps dir so `ListDumpFiles` surfaces them beside the regular dumps —
|
||
// they ARE the undo, and an undo the customer cannot see is not much of one. The regular replay
|
||
// loop matches `<stack>-<dbtype>.sql` exactly, so a prefixed file is never mistaken for a source.
|
||
const preRestoreDumpPrefix = "pre-restore-"
|
||
|
||
// OffsiteReconstituteResult reports what a reconstitution actually did, so the flash can state an
|
||
// OUTCOME instead of a mechanism. Every field here exists because the v0.147 flash could not say it.
|
||
type OffsiteReconstituteResult struct {
|
||
SnapshotID string
|
||
FilesPlaced int
|
||
DBsReplayed int
|
||
// VolumesReplayed (R-354) is how many named-volume archives came back from the snapshot. It is on
|
||
// the result for the same reason every other field here is: so the OUTCOME can state what
|
||
// happened rather than a mechanism. Without it the message said "5 fájl visszaállítva" over a
|
||
// restore that had silently dropped a 1.4 MB volume archive — a true sentence leaving a false
|
||
// impression, which is the shape this surface keeps having removed from it.
|
||
VolumesReplayed int
|
||
SafetyDump string // path of the pre-restore dump (the undo), "" when the app has no DB
|
||
// RolledBack (R-379) is true when the database replay FAILED and this run put the customer's own
|
||
// pre-restore copy back. It is on the result rather than inferred from the error, because "the
|
||
// restore failed" and "your data is as it was" are two different facts and the surface has to be
|
||
// able to say both.
|
||
RolledBack bool
|
||
DumpsAt time.Time // when the snapshot's DB half was taken (zero = unknown/legacy unit)
|
||
OffsiteRunID string // "" for a pre-v0.148 snapshot — an unverified pair
|
||
Skewed bool // the snapshot carries no coherence stamp: files and DB may differ in age
|
||
LooksEmpty bool // R-44 sniff on the dump about to be replayed
|
||
// Placement (R-351) is what the backup recorded about where this app's data lived, compared
|
||
// against where this restore actually wrote. Carried on the RESULT and not only on the refusal,
|
||
// so a restore that proceeded into a different destination says so in its own outcome rather
|
||
// than reporting a bare success — a warning beside a success is read as a success, so the
|
||
// difference has to survive into the message.
|
||
Placement PlacementCheck
|
||
}
|
||
|
||
// fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite).
|
||
// Deliberately NOT placeCopier(): that one is `--ignore-existing`, whose whole purpose is to leave
|
||
// live files alone, which is precisely what a full restore must not do.
|
||
func (m *Manager) fullPlaceCopier() func(src, dst string) (int, error) {
|
||
if m.offboxFullPlaceCopier != nil {
|
||
return m.offboxFullPlaceCopier
|
||
}
|
||
return rsyncRestoreOverwrite
|
||
}
|
||
|
||
// rsyncRestoreOverwrite copies src over dst: `rsync -a --itemize-changes`, with NO
|
||
// `--ignore-existing` (a changed file becomes the snapshot's version) and NO `--delete` (an extra
|
||
// file at dst survives). Returns the number of regular files transferred.
|
||
func rsyncRestoreOverwrite(src, dst string) (int, error) {
|
||
if err := os.MkdirAll(dst, 0755); err != nil {
|
||
return 0, fmt.Errorf("mkdir %s: %w", dst, err)
|
||
}
|
||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
|
||
defer cancel()
|
||
cmd := exec.CommandContext(ctx, "rsync", "-a", "--itemize-changes",
|
||
strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/")
|
||
out, err := cmd.CombinedOutput()
|
||
if err != nil {
|
||
return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out)))
|
||
}
|
||
return countRestoredFiles(string(out)), nil
|
||
}
|
||
|
||
// safetyDumpSet is what ONE reconstitution's undo consists of: the stamp that identifies this run's
|
||
// files, and one written path per database the app has.
|
||
//
|
||
// R-379: it exists because `writeSafetyDump` used to return only the FIRST path, and the rollback
|
||
// added in v0.220.0 must re-apply EVERY database's undo or it restores one and leaves the other
|
||
// half-written — the defect it exists to close, one database over. The stamp is the IDENTITY: three
|
||
// runs against `docmost` on 2026-08-22 left three `pre-restore-*` files in the same directory, so
|
||
// matching on the prefix would replay an arbitrary older state. Match on this stamp, never on the
|
||
// prefix, never on age or size.
|
||
type safetyDumpSet struct {
|
||
Stamp string // 20060102T150405Z — this run's, and only this run's
|
||
Files []safetyDumpFile // one per database, in discovery order
|
||
}
|
||
|
||
// safetyDumpFile pairs an undo file with the database it came from, so the rollback can hand each
|
||
// dump back to the container it belongs to instead of guessing from the filename.
|
||
type safetyDumpFile struct {
|
||
DB DiscoveredDB
|
||
Path string
|
||
}
|
||
|
||
// First returns the first written path, or "" — the value the pre-v0.220.0 signature returned, kept
|
||
// because the customer-facing message names one file and changing that is not this task.
|
||
func (s safetyDumpSet) First() string {
|
||
if len(s.Files) == 0 {
|
||
return ""
|
||
}
|
||
return s.Files[0].Path
|
||
}
|
||
|
||
// undoCopyPhrase is the sentence the DOUBLE-FAILURE message uses to describe the customer's undo
|
||
// copy — and it says what is TRUE, which is the whole of R-383.
|
||
//
|
||
// THE BUG THIS EXISTS TO KILL. The double-failure branch ended with „a korábbi állapot mentése
|
||
// megvan: <file>" — *the previous state's backup EXISTS* — built from the path `writeSafetyDump`
|
||
// returned and WITHOUT ever asking the filesystem. But one of the two ways `rollbackSafetyDump` fails
|
||
// is that the file is not there, so in exactly the case that sentence is printed it is most likely to
|
||
// be false. Measured twice live, on v0.220.2 and v0.221.1.
|
||
//
|
||
// A false reassurance is worse than no sentence: it is read at the moment the customer is deciding
|
||
// whether their data is recoverable, and it points support at a file that is not there.
|
||
//
|
||
// WHY NOT SIMPLY DROP THE FILENAME. R-351's lesson: a refusal that names nothing forces a person to
|
||
// remember what the product already knows. The operator needs the path either way — to fetch the
|
||
// file, or to look for it. So the absent case still names WHERE it should have been, and says
|
||
// plainly that it is not there.
|
||
//
|
||
// The check is `os.Stat`, deliberately not a readability or integrity test: this runs at the end of a
|
||
// failed restore on a machine that may be unwell, and the honest claim available here is presence.
|
||
// A zero-length file is reported as MISSING — a 0-byte dump restores nothing, and calling it present
|
||
// is the same false reassurance one step smaller.
|
||
func (m *Manager) undoCopyPhrase(set safetyDumpSet) string {
|
||
var present, absent []string
|
||
for _, f := range set.Files {
|
||
if f.Path == "" {
|
||
continue
|
||
}
|
||
if st, err := os.Stat(f.Path); err == nil && !st.IsDir() && st.Size() > 0 {
|
||
present = append(present, filepath.Base(f.Path))
|
||
continue
|
||
}
|
||
absent = append(absent, filepath.Base(f.Path))
|
||
}
|
||
switch {
|
||
case len(present) > 0 && len(absent) == 0:
|
||
return m.note("note.undo.present", strings.Join(present, ", "))
|
||
case len(present) > 0:
|
||
// Partial: name both halves. An app with two databases whose undo is half there is a
|
||
// different situation from either whole one, and support must not have to guess which.
|
||
return m.note("note.undo.partial", strings.Join(present, ", "), strings.Join(absent, ", "))
|
||
case len(absent) > 0:
|
||
return m.note("note.undo.absent", strings.Join(absent, ", "))
|
||
default:
|
||
return m.note("note.undo.none")
|
||
}
|
||
}
|
||
|
||
// writeSafetyDump dumps every live database of stack into the app's unit db-dumps dir under the
|
||
// `pre-restore-` prefix, and returns the SET it wrote. Returns (zero, nil) when the app has no
|
||
// database at all — a no-DB app has nothing to undo and must flow exactly as it did before
|
||
// v0.148.0 (no dump, no replay, no behaviour change).
|
||
//
|
||
// A discovered database that CANNOT be dumped is a hard error: it means the undo would not exist.
|
||
func (m *Manager) writeSafetyDump(ctx context.Context, stackName, nsRoot string) (safetyDumpSet, error) {
|
||
discover := m.discoverDBs
|
||
if discover == nil {
|
||
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
||
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||
}
|
||
}
|
||
dbs, err := discover(ctx)
|
||
if err != nil {
|
||
return safetyDumpSet{}, util.MsgError("err.backup.a_biztonsagi_mentes_elott_nem_sikerult", err)
|
||
}
|
||
var mine []DiscoveredDB
|
||
for _, db := range dbs {
|
||
if db.StackName == stackName {
|
||
mine = append(mine, db)
|
||
}
|
||
}
|
||
if len(mine) == 0 {
|
||
return safetyDumpSet{}, nil // no DB → nothing to undo → scenario E flows unchanged
|
||
}
|
||
|
||
dumpDir := AppDBDumpPath(nsRoot, stackName)
|
||
if err := os.MkdirAll(dumpDir, 0755); err != nil {
|
||
return safetyDumpSet{}, util.MsgError("err.backup.a_biztonsagi_mentes_konyvtara_nem_hozhato", err)
|
||
}
|
||
set := safetyDumpSet{Stamp: time.Now().UTC().Format("20060102T150405Z")}
|
||
for _, db := range mine {
|
||
// R-361: the undo copy is dumped STRAIGHT to its own name. It used to be dumped to the app's
|
||
// canonical `<stack>-<dbtype>.sql` and renamed afterwards, and the comment here asserted that
|
||
// the rename meant it "can never overwrite the app's real dump". THAT WAS FALSE AS WRITTEN:
|
||
// `DumpOne` writes the canonical name, so every safety dump destroyed the app's own backup and
|
||
// then moved it away — leaving the app with NO database backup until the next nightly run, and
|
||
// a local restore-from-unit in that window telling the customer the app never had a database.
|
||
// Measured live on demo-hp 2026-08-22: `docmost` and `bookstack` both held only `pre-restore-*`
|
||
// files and no canonical dump.
|
||
//
|
||
// THE INVARIANT, AND HOW IT IS NOW ENFORCED: nothing but the app's own dump is ever written to
|
||
// the canonical name, because the safety dump never names it — `DumpOneTo` takes the final path
|
||
// and derives its own `.tmp` from it, so neither the destination nor the scratch file can
|
||
// collide with a nightly dump running beside it. Pinned by
|
||
// TestR361_SafetyDumpLeavesTheCanonicalDumpByteIdentical.
|
||
safe := filepath.Join(dumpDir, fmt.Sprintf("%s%s-%s-%s.sql", preRestoreDumpPrefix, set.Stamp, stackName, db.DBType))
|
||
res := m.dumpForSafety(ctx, db, safe)
|
||
if res.Error != nil {
|
||
return safetyDumpSet{}, util.MsgError("err.backup.a_jelenlegi_adatbazis_biztonsagi_mentese_sikertelen", db.ContainerName, res.Error)
|
||
}
|
||
// EVERY file, not just the first — R-379, and the reason is on safetyDumpSet.
|
||
set.Files = append(set.Files, safetyDumpFile{DB: db, Path: safe})
|
||
m.logger.Printf("[INFO] [offbox] %s: pre-restore safety dump written → %s (%s)", stackName, filepath.Base(safe), humanizeBytes(res.Size))
|
||
}
|
||
return set, nil
|
||
}
|
||
|
||
// shortID trims a docker id for logs. Never used for identity — only for reading.
|
||
func shortID(id string) string {
|
||
if len(id) > 12 {
|
||
return id[:12]
|
||
}
|
||
return id
|
||
}
|
||
|
||
// maxUndoCopiesPerApp is how many `pre-restore-` copies an app keeps.
|
||
//
|
||
// THREE, and the reasoning rather than a number pulled from the air. One is not enough: the case
|
||
// that needs an undo is a restore that went wrong, and the second-guess attempt is exactly when the
|
||
// customer reaches for the state before the FIRST attempt. Many is not free: they live inside the
|
||
// recovery unit, so every one is also mirrored to Tier 2 AND pushed off-site permanently — four
|
||
// accumulated on `docmost` in a single afternoon on 2026-08-22 (135 KB + 135 KB + 141 KB + 138 KB),
|
||
// each of them forever. Three keeps two prior attempts and bounds the off-site growth.
|
||
const maxUndoCopiesPerApp = 3
|
||
|
||
// pruneUndoCopies keeps the newest maxUndoCopiesPerApp undo copies for an app and removes the rest.
|
||
//
|
||
// DELIBERATELY NOT CALLED FROM THE RESTORE PATH. A delete on the failure path is how an undo goes
|
||
// missing at exactly the moment it is needed; this runs from the capture side, where nothing is
|
||
// depending on the files right now. It is called AFTER a successful capture, never before one.
|
||
//
|
||
// Ordering is by the stamp IN THE FILENAME, not by mtime and never by size: mtime moves when a file
|
||
// is copied or a filesystem is restored, and the stamp is the identity writeSafetyDump assigned.
|
||
// The newest is never a deletion candidate even if the list is somehow malformed.
|
||
func (m *Manager) pruneUndoCopies(dumpDir, stack string) {
|
||
entries, err := os.ReadDir(dumpDir)
|
||
if err != nil {
|
||
return
|
||
}
|
||
type undo struct{ name, stamp string }
|
||
var undos []undo
|
||
for _, e := range entries {
|
||
if e.IsDir() || !strings.HasSuffix(e.Name(), ".sql") {
|
||
continue
|
||
}
|
||
base := strings.TrimSuffix(e.Name(), ".sql")
|
||
if !strings.HasPrefix(base, preRestoreDumpPrefix) {
|
||
continue
|
||
}
|
||
after := strings.TrimPrefix(base, preRestoreDumpPrefix)
|
||
i := strings.Index(after, "-")
|
||
if i <= 0 {
|
||
continue // not the shape writeSafetyDump writes — leave it alone rather than guess
|
||
}
|
||
undos = append(undos, undo{name: e.Name(), stamp: after[:i]})
|
||
}
|
||
if len(undos) <= maxUndoCopiesPerApp {
|
||
return
|
||
}
|
||
sort.Slice(undos, func(a, b int) bool { return undos[a].stamp > undos[b].stamp }) // newest first
|
||
for _, u := range undos[maxUndoCopiesPerApp:] {
|
||
p := filepath.Join(dumpDir, u.name)
|
||
if err := os.Remove(p); err != nil {
|
||
m.logger.Printf("[WARN] [backup] %s: could not prune old undo copy %s: %v", stack, u.name, err)
|
||
continue
|
||
}
|
||
m.logger.Printf("[INFO] [backup] %s: pruned old undo copy %s (keeping the newest %d)", stack, u.name, maxUndoCopiesPerApp)
|
||
}
|
||
}
|
||
|
||
// RestoreHoldFor reports whether an app is being held stopped after a failed restore + failed
|
||
// rollback, and returns the customer-facing reason. Every start path consults this — the customer's
|
||
// button, the app-stop Recover() starter, and the boot reconciler — because a hold that only one
|
||
// path honours is not a hold.
|
||
//
|
||
// Nil settings ⇒ NOT held. That direction is deliberate and is the opposite of the usual fail-closed
|
||
// rule: with no settings there is no hold recorded, so refusing every start would strand every app
|
||
// on a misconfigured box. The write side logs loudly when it cannot persist (see
|
||
// holdAppAfterFailedRollback), which is where that case is caught.
|
||
func (m *Manager) RestoreHoldFor(stack string) (bool, string) {
|
||
return m.RestoreHoldForLang(stack, m.boxLang())
|
||
}
|
||
|
||
// RestoreHoldForLang is RestoreHoldFor with the sentence in lang (v0.264.0, R-606). The page and the
|
||
// household mail ask for the READER's language; every other caller takes the box's. The Hungarian is
|
||
// byte-identical to the literals it replaced (UpdateHoldFmt & co. — TestR606_HoldSentenceHungarianUnchanged).
|
||
func (m *Manager) RestoreHoldForLang(stack, lang string) (bool, string) {
|
||
if m == nil || m.settings == nil {
|
||
return false, ""
|
||
}
|
||
h, ok := m.settings.GetRestoreHold(stack)
|
||
if !ok {
|
||
return false, ""
|
||
}
|
||
// Slice 4: one storage, two reasons. An update hold names the copy it can be restored from; a
|
||
// restore hold names nothing, because the restore it refers to already consumed the copy.
|
||
if h.Reason == settings.HoldReasonUpdateFailed {
|
||
return true, m.undoHoldPrefix(lang, h.UndoState) + updateHoldSentence(lang, stack, h)
|
||
}
|
||
when := h.At
|
||
if t, err := time.Parse(time.RFC3339, h.At); err == nil {
|
||
when = t.Format("2006-01-02 15:04")
|
||
}
|
||
return true, util.Text(lang, "note.reconstitute.held", stack, when)
|
||
}
|
||
|
||
// undoHoldPrefix (v0.263.0) opens the hold sentence when the box already TRIED to undo the update and
|
||
// that failed too: what was tried, then what state the data is in. "" when no undo was attempted, so
|
||
// every hold written before v0.263.0 reads exactly as it did.
|
||
func (m *Manager) undoHoldPrefix(lang, state string) string {
|
||
switch state {
|
||
case "untouched", "half", "not_started":
|
||
return util.Text(lang, "hold.update.undo_failed") + " " + util.Text(lang, "hold.update.undo_state."+state) + " "
|
||
case "":
|
||
return ""
|
||
}
|
||
m.logger.Printf("[WARN] [backup] unknown undo state %q on a hold — rendering the plain prefix", state)
|
||
return util.Text(lang, "hold.update.undo_failed") + " "
|
||
}
|
||
|
||
// updateHoldSentence is the update hold's own sentence (slice 4, R-475, R-479) in lang. Its Hungarian
|
||
// is UpdateHoldFmt / UpdateHoldTierFmt / UpdateHoldLegacyFmt byte for byte (pinned by a test).
|
||
func updateHoldSentence(lang, stack string, h settings.RestoreHold) string {
|
||
copyDate := util.Text(lang, "note.reconstitute.copy_latest")
|
||
if h.CopyDate != "" {
|
||
copyDate = fmtHoldTime(h.CopyDate)
|
||
}
|
||
// R-475: name the tier when the hold recorded one; an older hold keeps its own sentence.
|
||
if label := UpdateTierLabelIn(lang, h.CopyTier); label != "" && h.CopyDate != "" {
|
||
if h.CopyHolds != "" { // R-479: name what the copy holds
|
||
return util.Text(lang, "hold.update.sentence", stack, fmtHoldTime(h.At), label, copyDate, copyHoldsIn(lang, h.CopyHolds))
|
||
}
|
||
return util.Text(lang, "hold.update.sentence_tier", stack, fmtHoldTime(h.At), label, copyDate)
|
||
}
|
||
return util.Text(lang, "hold.update.sentence_legacy", stack, fmtHoldTime(h.At), copyDate)
|
||
}
|
||
|
||
// holdAppAfterFailedRollback records the R-379/R-380 hold and makes sure nothing restarts the app
|
||
// behind our back.
|
||
//
|
||
// OPERATOR RULING, 2026-08-22: when the replay fails AND the rollback fails, the app is HELD
|
||
// STOPPED rather than started. A running app on a half-written database lets the customer type into
|
||
// it, and that turns a recoverable state into a permanent one. The alternative — start it and mark
|
||
// it — was considered and declined.
|
||
//
|
||
// It ENDS the app-stop marker deliberately. The marker means "owed a restart"; a held app is not
|
||
// owed one, and leaving the marker active would have Recover() start the broken app at the next
|
||
// controller boot. The hold is the thing that persists, not the marker.
|
||
func (m *Manager) holdAppAfterFailedRollback(stack string, replayErr, rollbackErr error) {
|
||
if m.settings == nil {
|
||
m.logger.Printf("[ERROR] [offbox] %s: cannot persist the restore hold — no settings wired; the app is stopped but NOTHING will refuse a restart", stack)
|
||
return
|
||
}
|
||
h := settings.RestoreHold{
|
||
Stack: stack,
|
||
At: time.Now().UTC().Format(time.RFC3339),
|
||
}
|
||
if replayErr != nil {
|
||
h.ReplayError = replayErr.Error()
|
||
}
|
||
if rollbackErr != nil {
|
||
h.RollbackErr = rollbackErr.Error()
|
||
}
|
||
if err := m.settings.SetRestoreHold(h); err != nil {
|
||
m.logger.Printf("[ERROR] [offbox] %s: persisting the restore hold FAILED: %v — the app is stopped and unguarded", stack, err)
|
||
}
|
||
// The app is not owed a restart; it is deliberately held. See the doc comment.
|
||
if m.appStop != nil {
|
||
m.appStop.End()
|
||
}
|
||
if m.restoreHoldNotify != nil {
|
||
m.restoreHoldNotify(stack, replayErr, rollbackErr)
|
||
}
|
||
}
|
||
|
||
// SetRestoreHoldNotify wires the operator notification for a held app (cmd/controller/main.go).
|
||
func (m *Manager) SetRestoreHoldNotify(fn func(stack string, replayErr, rollbackErr error)) {
|
||
m.restoreHoldNotify = fn
|
||
}
|
||
|
||
// rollbackSafetyDump re-applies THIS RUN's undo set, database by database, and is the whole of
|
||
// R-379's fix: it is the same ImportDump call a person made by hand on 2026-08-22 to recover
|
||
// `docmost` and `bookstack` after a failed replay, moved into the product.
|
||
//
|
||
// It runs with the DB service still up (the replay's own window) and BEFORE any restart, so the
|
||
// app never observes the half-written state. An error here means the app cannot be trusted to run —
|
||
// see the hold in ReconstituteFromOffsite.
|
||
func (m *Manager) rollbackSafetyDump(ctx context.Context, stack string, set safetyDumpSet) error {
|
||
if len(set.Files) == 0 {
|
||
return nil
|
||
}
|
||
imp := m.rollbackImport
|
||
if imp == nil {
|
||
imp = func(ctx context.Context, db DiscoveredDB, path string) error {
|
||
return ImportDump(ctx, db, path, m.logger, m.isDebug())
|
||
}
|
||
}
|
||
// RE-DISCOVER THE CONTAINERS. The undo FILE is stable; the container it must be poured into is
|
||
// NOT. `writeSafetyDump` captured its DiscoveredDB before the stop, and by the time the rollback
|
||
// runs the stack has been stopped and the DB service re-created — a NEW container id.
|
||
//
|
||
// MEASURED LIVE ON demo-hp 2026-08-22, which is the only reason this is here: the first live run
|
||
// of this code captured `docmost-postgres id=9adbc14f9af6` at 16:05:44, the DB-only start
|
||
// re-created it as `309795897b82` at 16:05:47, and the rollback's `docker exec` against the dead
|
||
// id sat in `waitDBReady` until it timed out 30 s later — so the app was HELD for an
|
||
// infrastructure reason when its data was recoverable. The unit tests could not see it: they
|
||
// inject the import seam and never touch container identity. `reimportDBDumpsFrom` already
|
||
// re-discovers for exactly this reason.
|
||
discover := m.discoverDBs
|
||
if discover == nil {
|
||
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
||
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||
}
|
||
}
|
||
live, dErr := discover(ctx)
|
||
if dErr != nil {
|
||
return util.MsgError("err.backup.a_visszavonas_elott_nem_sikerult_felderiteni", dErr)
|
||
}
|
||
liveFor := func(want DiscoveredDB) (DiscoveredDB, bool) {
|
||
for _, db := range live {
|
||
if db.StackName == want.StackName && db.DBType == want.DBType {
|
||
return db, true
|
||
}
|
||
}
|
||
return DiscoveredDB{}, false
|
||
}
|
||
|
||
for _, f := range set.Files {
|
||
if _, sErr := os.Stat(f.Path); sErr != nil {
|
||
return util.MsgError("err.backup.a_visszavonashoz_szukseges_mentes_nem_talalhato", filepath.Base(f.Path), sErr)
|
||
}
|
||
target, ok := liveFor(f.DB)
|
||
if !ok {
|
||
// Fail closed: pouring an undo into a container we cannot identify is worse than saying
|
||
// we could not do it.
|
||
return util.MsgError("err.backup.a_z_adatbazis_taroloja_nem_talalhato", f.DB.ContainerName)
|
||
}
|
||
if target.ContainerID != f.DB.ContainerID {
|
||
m.logger.Printf("[DEBUG] [offbox] %s: %s was re-created during the restore (%s → %s) — rolling back into the live container",
|
||
stack, f.DB.ContainerName, shortID(f.DB.ContainerID), shortID(target.ContainerID))
|
||
}
|
||
m.logger.Printf("[INFO] [offbox] %s: rolling back to the pre-restore state from %s", stack, filepath.Base(f.Path))
|
||
if err := imp(ctx, target, f.Path); err != nil {
|
||
return util.MsgError("err.backup.a_korabbi_allapot_visszaallitasa_sikertelen", target.ContainerName, err)
|
||
}
|
||
}
|
||
m.logger.Printf("[INFO] [offbox] %s: rollback complete — %d database(s) returned to the pre-restore state", stack, len(set.Files))
|
||
return nil
|
||
}
|
||
|
||
// dumpForSafety is the dump seam for the safety dump (tests inject; nil → the real DumpOneTo).
|
||
//
|
||
// R-361: it takes the FINAL PATH, not a directory. A directory argument is what allowed the callee to
|
||
// choose the canonical name, which is the whole defect.
|
||
func (m *Manager) dumpForSafety(ctx context.Context, db DiscoveredDB, finalPath string) DumpResult {
|
||
if m.safetyDumpFn != nil {
|
||
return m.safetyDumpFn(ctx, db, finalPath)
|
||
}
|
||
return DumpOneTo(ctx, db, finalPath, m.logger, m.isDebug())
|
||
}
|
||
|
||
// ReconstituteFromOffsite makes the live app equal to a restored full-scratch snapshot: files
|
||
// overwritten to the snapshot's version (extras survive, nothing deleted), then the snapshot's own
|
||
// DB dump replayed, with a safety dump of the current database taken first. Requires a completed
|
||
// FULL scratch restore (RestoreOffboxScratch with full=true). Single-flight.
|
||
// ackPlacementChange (R-351) is the customer's DELIBERATE acknowledgement that the destination
|
||
// differs from the one the backup recorded. It is a separate act from the restore's own confirm:
|
||
// folding it into `confirm=1` would mean one click carried two decisions, which is precisely what
|
||
// R-48 exists to prevent.
|
||
func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ackPlacementChange bool) (OffsiteReconstituteResult, error) {
|
||
var res OffsiteReconstituteResult
|
||
if !m.OffboxConfigured() {
|
||
return res, fmt.Errorf("off-box backup not configured")
|
||
}
|
||
if !isSafeStackName(stack) {
|
||
return res, fmt.Errorf("invalid stack name")
|
||
}
|
||
if m.stackProvider == nil {
|
||
return res, fmt.Errorf("stack provider not configured")
|
||
}
|
||
if err := m.acquireRunning(); err != nil {
|
||
return res, util.MsgError("err.backup.egy_masik_mentesi_visszaallitasi_muvelet_mar")
|
||
}
|
||
defer m.releaseRunning()
|
||
|
||
scratch, _, err := m.offboxRestoreScratchDir(stack)
|
||
if err != nil {
|
||
return res, err
|
||
}
|
||
if _, sErr := os.Stat(scratch); sErr != nil {
|
||
return res, util.MsgError("err.backup.nincs_elokeszitett_teljes_visszaallitas_futtass_elobb")
|
||
}
|
||
id, paths, err := m.offboxLatestSnapshot(ctx, stack)
|
||
if err != nil {
|
||
return res, err
|
||
}
|
||
res.SnapshotID = id
|
||
|
||
// R-253: the same sentence the restore page now shows, so the page and the refusal cannot
|
||
// drift apart again. It is a REFUSAL, not a failure — the data is untouched and the customer
|
||
// has one step to take. The restore deliberately does NOT deploy the app itself: the
|
||
// destination is the app's own HDD path, which is a drive the CUSTOMER chooses at deploy
|
||
// time, and picking it for them is the decision this whole recovery path exists to leave
|
||
// with them.
|
||
// R-351: the refusal now NAMES the place the backup recorded, when it can read it. The
|
||
// prepared scratch already contains the unit, so this is a local file read — no network call,
|
||
// nothing restored, and it happens on a path that was going to refuse anyway. Telling
|
||
// somebody to reinstall without telling them where the data belongs is what forced the
|
||
// 2026-08-21 operator to remember two values the backup already held.
|
||
// R-356: this refusal used to be reached by `GetStackHDDPath(stack) == ""` — one predicate
|
||
// answering two questions. It now covers ONLY "the app is not deployed", and it stopped
|
||
// covering "the app has no drive". The reason the two came apart: the drive choice is the
|
||
// CUSTOMER's, and 40 of the 53 catalog apps were never offered one — they have no choice to
|
||
// leave with them, and their data lives on the system data path by design. The R-253 decision
|
||
// above is untouched for the 13 apps that DO have a drive to get wrong.
|
||
if !m.isStackDeployed(stack) {
|
||
if rec := m.recordedPlacementFromScratch(scratch); rec.Known() {
|
||
return res, util.MsgError("err.backup.not_installed_known_drive", stack, rec.Drive)
|
||
}
|
||
return res, util.MsgError("err.backup.not_installed", stack)
|
||
}
|
||
// The destination is resolved by the SAME rule the capture side used to write this snapshot
|
||
// (CaptureRecoveryUnit → GetAppDrivePath): the app's drive if it has one, the system data path
|
||
// otherwise. Anything else and the restore would aim at a different place than the backup came
|
||
// from, which is the mismatch prompt firing on a box where nothing actually moved.
|
||
hdd := strings.TrimSpace(m.GetAppDrivePath(stack))
|
||
if hdd == "" {
|
||
// A DIFFERENT failure from the one above, so it gets a different sentence: the app IS
|
||
// installed, but the box cannot name its own data root (systemDataPath unset). Saying
|
||
// "nincs telepítve" here would send the customer to reinstall an app that is already
|
||
// running, and the real fault would stay invisible.
|
||
return res, util.MsgError("err.backup.no_data_root", stack)
|
||
}
|
||
liveNs := m.namespaceRoot(hdd)
|
||
|
||
// --- R-357: FREE SPACE, BEFORE ANYTHING IS TOUCHED ------------------------------------------
|
||
//
|
||
// This file contained ZERO references to offboxFree until now. The three headroom gates that
|
||
// existed all guarded NON-destructive paths (offbox_restore.go: the download sizer, the prepare
|
||
// gate, and PlaceOffsiteRestore's missing-only merge). The one path that stops the customer's app
|
||
// and overwrites their live data had none.
|
||
//
|
||
// Measured on demo-hp 2026-08-21: it stopped the app, ran out of disk part-way, left 2 of 5 planted
|
||
// items in place and restarted the app — a half-restored dataset presented as a completed restore.
|
||
//
|
||
// POSITION IS THE WHOLE FIX. This sits before mapOffsiteRestorePaths, before writeSafetyDump and
|
||
// well before StopStack, so a refusal costs the customer nothing at all — the app never goes down.
|
||
// A gate after StopStack would turn a refusal into an outage, which is the shape it exists to
|
||
// prevent. Scenario D asserts the non-effect (StopStack call count 0), not the error string.
|
||
//
|
||
// NO HEADROOM MULTIPLIER, deliberately, and stated so the next reader does not "fix" it:
|
||
// OffboxRestorePrepareFull uses ×1.1 because it is sizing a DOWNLOAD whose final size it is
|
||
// predicting. This is a local copy of a tree that already exists on disk, so its size is known
|
||
// exactly — the same reasoning PlaceOffsiteRestore's gate uses, and this matches it.
|
||
free, need := m.offboxFree()(liveNs), m.offboxSize()(scratch)
|
||
switch {
|
||
case need <= 0:
|
||
// FAIL CLOSED. Without this the comparison below is `free < 0`, which is false, and an
|
||
// unmeasurable scratch would sail straight through into the destructive phase — the gate
|
||
// present and inert, which is worse than no gate because it reads as protection.
|
||
m.logger.Printf("[ERROR] [offbox] %s: REFUSING the destructive restore — the scratch size could not be measured (scratch=%s)", stack, scratch)
|
||
return res, fmt.Errorf(offsiteSizeUnknownMsg)
|
||
case free <= 0:
|
||
// Same direction for the other probe. The customer sentence is shared with the case above
|
||
// (the operator asked for one wording); the LOG line above and below is what distinguishes
|
||
// which probe failed.
|
||
m.logger.Printf("[ERROR] [offbox] %s: REFUSING the destructive restore — free space on the live namespace could not be measured (liveNs=%s)", stack, liveNs)
|
||
return res, fmt.Errorf(offsiteSizeUnknownMsg)
|
||
case free < need:
|
||
m.logger.Printf("[WARN] [offbox] %s: REFUSING the destructive restore — need %d B, free %d B on %s; the app was NOT stopped", stack, need, free, liveNs)
|
||
return res, fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free))
|
||
}
|
||
|
||
placements, err := mapOffsiteRestorePaths(paths, stack, scratch, liveNs)
|
||
if err != nil {
|
||
return res, err // whole-placement refusal (no partial writes)
|
||
}
|
||
// Stat pre-pass over EVERY placement before the first copy — an incomplete scratch (e.g. only a
|
||
// unit-only restore was run) refuses with ZERO copies.
|
||
for _, pl := range placements {
|
||
if _, sErr := os.Stat(pl.src); sErr != nil {
|
||
return res, util.MsgError("err.backup.a_teljes_visszaallitas_hianyos_nincs_meg", filepath.Base(pl.src))
|
||
}
|
||
}
|
||
|
||
// The snapshot's coherence stamp, read from the RESTORED unit manifest (not the live one).
|
||
scratchUnit := ""
|
||
for _, pl := range placements {
|
||
if pl.isUnit {
|
||
scratchUnit = pl.src
|
||
break
|
||
}
|
||
}
|
||
if scratchUnit == "" {
|
||
return res, util.MsgError("err.backup.a_pillanatkepben_nincs_mentesi_egyseg_a")
|
||
}
|
||
scratchDumpDir := filepath.Join(scratchUnit, "db-dumps")
|
||
man := readManifest(filepath.Join(scratchUnit, "manifest.json"))
|
||
if man != nil {
|
||
res.OffsiteRunID = man.OffsiteRunID
|
||
if man.DumpsAt != "" {
|
||
if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil {
|
||
res.DumpsAt = t
|
||
}
|
||
}
|
||
}
|
||
|
||
// --- WHERE THE BACKUP SAYS THIS DATA LIVED (R-351) ------------------------------------------
|
||
// The manifest we just opened has carried `drive` and `namespace_root` since schema 1, and until
|
||
// now nothing read them back. Compared HERE, before the safety dump and before the first byte is
|
||
// placed, so the refusal costs nothing and leaves the app completely untouched.
|
||
//
|
||
// An UNKNOWN recording (a pre-field unit, or one we could not read) is not a mismatch and does
|
||
// not refuse: blocking on an absence would strand every older backup, and CheckPlacement returns
|
||
// that case explicitly rather than letting it fall through as "they match".
|
||
res.Placement = CheckPlacement(man, hdd, liveNs)
|
||
if res.Placement.Mismatch && !ackPlacementChange {
|
||
return res, fmt.Errorf("%s", PlacementMismatchMessage(stack, res.Placement))
|
||
}
|
||
if res.Placement.Mismatch {
|
||
m.logger.Printf("[WARN] [offbox] %s: restoring into %s, but the backup recorded %s — the customer acknowledged the change",
|
||
stack, res.Placement.LiveDrive, res.Placement.Recorded.Drive)
|
||
}
|
||
// A pre-v0.148 snapshot carries no stamp: its dump was whatever the 02:30 local run left behind,
|
||
// so the pair's two halves may be hours or days apart. Surfaced, never blocked — the confirm
|
||
// dialog says so and the safety dump makes it reversible.
|
||
res.Skewed = res.OffsiteRunID == ""
|
||
res.LooksEmpty = m.sniffScratchDump(scratchDumpDir, stack)
|
||
|
||
// --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------
|
||
// Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir,
|
||
// so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first
|
||
// mutation so the refusal below costs nothing.
|
||
var dbServices []string
|
||
if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
|
||
svcs, dsErr := DBServiceNames(composePath)
|
||
if dsErr != nil {
|
||
// "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse.
|
||
m.logger.Printf("[WARN] [offbox] %s: could not read the live compose services: %v", stack, dsErr)
|
||
}
|
||
dbServices = svcs
|
||
}
|
||
|
||
// --- THE UNDO, BEFORE THE ACT ---------------------------------------------------------------
|
||
// Taken while the stack is still UP (a stopped database cannot be dumped) and before a single
|
||
// byte is overwritten, so a failure here aborts with the live app completely untouched.
|
||
safetySet, err := m.writeSafetyDump(ctx, stack, liveNs)
|
||
if err != nil {
|
||
return res, err
|
||
}
|
||
safety := safetySet.First()
|
||
res.SafetyDump = safety
|
||
hasDB := safety != ""
|
||
if hasDB {
|
||
if _, sErr := os.Stat(safety); sErr != nil {
|
||
// Fail-closed: never replay when the undo is not verifiably on disk.
|
||
return res, util.MsgError("err.backup.a_biztonsagi_mentes_nem_talalhato_a")
|
||
}
|
||
// Fail-closed (R-47): the app HAS a database but no compose service can be identified to
|
||
// start alone for the replay. The only alternative would be to start everything and replay
|
||
// into the race that produced H4 — refusing with the live app untouched is the better outcome.
|
||
if len(dbServices) == 0 {
|
||
return res, util.MsgError("err.backup.az_adatbazis_szolgaltatas_nem_azonosithato_a", stack)
|
||
}
|
||
}
|
||
|
||
// --- FILES ----------------------------------------------------------------------------------
|
||
// R-166: mark the stop→restore→start window BEFORE stopping. A controller killed anywhere inside
|
||
// it used to leave the app down with nothing on disk recording that it was owed a restart — and a
|
||
// full offsite restore is a LONG window, so this is the shape most likely to be interrupted.
|
||
if err := m.appStop.Begin("offbox-reconstitute:"+stack, ReasonOffboxReconstitute, []string{stack}); err != nil {
|
||
return res, util.MsgError("err.backup.a_z_leallitasa_elotti_jelolo_nem", stack, err)
|
||
}
|
||
// restartStack starts the app and clears the marker ONLY when the start actually succeeded — a
|
||
// failed start leaves the marker so the next startup retries. Every bring-up below goes through
|
||
// it; a bare StartStack here would clear nothing and strand the marker on the success path.
|
||
restartStack := func() error {
|
||
err := m.stackProvider.StartStack(stack)
|
||
if err == nil {
|
||
m.appStop.End()
|
||
} else {
|
||
// R-330: same rule as the volume-dump path — a restart that was attempted and broke
|
||
// leaves the app genuinely down, so the alarm suppression must go immediately.
|
||
m.appStop.ReleaseFailed(stack)
|
||
}
|
||
return err
|
||
}
|
||
if err := m.stackProvider.StopStack(stack); err != nil {
|
||
m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err)
|
||
}
|
||
copier := m.fullPlaceCopier()
|
||
for _, pl := range placements {
|
||
if pl.isUnit {
|
||
// The live recovery unit is still never overwritten — it is the LOCAL restore path's
|
||
// source and clobbering it would trade one recovery route for another. THAT reason is
|
||
// sound and still holds; it is why this skip stays.
|
||
//
|
||
// R-354 — THE SECOND HALF OF THIS COMMENT USED TO BE FALSE AND IS CORRECTED HERE. It said
|
||
// "the snapshot's dump is replayed from the scratch unit instead, so nothing is lost by
|
||
// skipping it". That was true of the DATABASE dump and false of the VOLUME archives, which
|
||
// live in the same unit and were replayed by nothing at all. Skipping the placement is
|
||
// correct; treating the skip as harmless was not. Measured live 2026-08-21: calibre-web's
|
||
// 1 422 848-byte `calibre_web_config.tar` was in the unit, in the snapshot and in the
|
||
// verification folder, and the restore reported "5 fájl visszaállítva" without it.
|
||
//
|
||
// Both legs are now replayed FROM THE SCRATCH UNIT below — volumes first, then the DB, so
|
||
// the logical dump still wins over any volume-tar copy of the same database.
|
||
continue
|
||
}
|
||
n, cErr := copier(pl.src, pl.dst)
|
||
if cErr != nil {
|
||
// Best-effort bring-up: leaving the app stopped after a partial copy would turn a failed
|
||
// restore into an outage.
|
||
if sErr := restartStack(); sErr != nil {
|
||
m.logger.Printf("[WARN] [offbox] %s: restart after failed placement also failed: %v", stack, sErr)
|
||
}
|
||
return res, util.MsgError("err.backup.a_z_fajljainak_visszaallitasa_sikertelen", stack, cErr)
|
||
}
|
||
res.FilesPlaced += n
|
||
}
|
||
|
||
// --- NAMED VOLUMES (R-354) ------------------------------------------------------------------
|
||
// Replayed from the SCRATCH unit, exactly as the database dump is, and for the same reason: the
|
||
// live unit is never overwritten by a placement, so the snapshot's copy exists only under the
|
||
// scratch. Same helper as the local restore path — one implementation, two callers.
|
||
//
|
||
// ORDER IS LOAD-BEARING and mirrors RestoreFromRecoveryUnit: volumes FIRST, database after, so a
|
||
// logical .sql dump still wins over whatever copy of the same database a volume tar happens to
|
||
// contain. It also has to happen inside the stopped window, because replacing a named volume means
|
||
// removing it, and Docker refuses that while a container holds it.
|
||
volReplay := m.volumeReplayFrom
|
||
if volReplay == nil {
|
||
volReplay = m.restoreDockerVolumesFrom
|
||
}
|
||
nVols, vErr := volReplay(stack, filepath.Join(scratchUnit, "volume-dumps"))
|
||
res.VolumesReplayed = nVols
|
||
if vErr != nil {
|
||
// A partial replay must never read as a completion. Bring the app back up rather than leaving
|
||
// an outage, then surface it — the same shape the file leg above uses.
|
||
if sErr := restartStack(); sErr != nil {
|
||
m.logger.Printf("[WARN] [offbox] %s: restart after failed volume replay also failed: %v", stack, sErr)
|
||
}
|
||
return res, util.MsgError("err.backup.a_z_adatkotetenek_visszaallitasa_sikertelen", stack, vErr)
|
||
}
|
||
|
||
// --- DATABASE -------------------------------------------------------------------------------
|
||
// The DB container must be UP for the replay (ImportDump talks to it with its own discovered
|
||
// credentials), but NOTHING ELSE may be — R-47. Until v0.153.0 this was a full StartStack, which
|
||
// gave the application a window to rebuild the very schema objects the dump was about to create:
|
||
// measured at 2 s on 2026-07-19, and the replay aborted `relation "clip_index" already exists`
|
||
// under ON_ERROR_STOP=1 (H4). Starting only the database service closes that window entirely.
|
||
if hasDB {
|
||
if err := m.stackProvider.StartStackServices(stack, dbServices); err != nil {
|
||
// Best-effort bring-up: a failed restore must not also be an outage.
|
||
if sErr := restartStack(); sErr != nil {
|
||
m.logger.Printf("[WARN] [offbox] %s: full start after failed DB-only start also failed: %v", stack, sErr)
|
||
}
|
||
return res, util.MsgError("err.backup.a_z_adatbazis_szolgaltatasanak_inditasa_sikertelen", stack, err)
|
||
}
|
||
n, iErr := m.reimportDBDumpsFrom(ctx, stack, scratchDumpDir)
|
||
res.DBsReplayed = n
|
||
if iErr != nil {
|
||
// --- R-379/R-380: PUT THE CUSTOMER'S OWN COPY BACK ---------------------------------
|
||
// Until v0.220.0 this branch restarted the app onto a HALF-WRITTEN database and named
|
||
// the undo file in the message. Measured 2026-08-22 on demo-hp: Postgres was left
|
||
// emptied and crash-looping; MariaDB was left partly applied while the app reported
|
||
// `health=healthy`. Both are the same failure — a half state — and the only difference
|
||
// was whether it looked broken. MariaDB's structural statements are not transactional,
|
||
// so no engine flag can prevent the half state; putting the undo back is what removes
|
||
// it. This is the same ImportDump call a person ran by hand that day to recover both
|
||
// apps, moved into the product.
|
||
//
|
||
// The rollback runs BEFORE any restart and with the DB service still up, so the app
|
||
// never observes the half state. The ORIGINAL replay error is never swallowed: it is
|
||
// logged here in full and named in the customer's sentence.
|
||
m.logger.Printf("[ERROR] [offbox] %s: database replay failed, rolling back to the pre-restore state: %v", stack, iErr)
|
||
if rbErr := m.rollbackSafetyDump(ctx, stack, safetySet); rbErr != nil {
|
||
// BOTH failed. Do NOT start the app: a running app on a half-written database lets
|
||
// the customer type into it and makes the damage permanent. Hold it instead —
|
||
// operator ruling, 2026-08-22.
|
||
m.logger.Printf("[ERROR] [offbox] %s: ROLLBACK ALSO FAILED (%v) — holding the app stopped; replay error was: %v", stack, rbErr, iErr)
|
||
m.holdAppAfterFailedRollback(stack, iErr, rbErr)
|
||
// R-383: the undo copy is DESCRIBED FROM DISK, never from the path alone. See
|
||
// undoCopyPhrase — this sentence used to assert the file existed in exactly the
|
||
// branch where a missing file is one of the two causes.
|
||
return res, util.MsgError("err.backup.db_restore_and_rollback_failed", stack, m.undoCopyPhrase(safetySet))
|
||
}
|
||
res.RolledBack = true
|
||
if sErr := restartStack(); sErr != nil {
|
||
m.logger.Printf("[WARN] [offbox] %s: start after a successful rollback failed: %v", stack, sErr)
|
||
}
|
||
// Says BOTH things. A message that reported only the failure would leave the customer
|
||
// believing their data was gone when it is back — the omission of a GAIN is as
|
||
// misleading as the omission of a loss.
|
||
return res, util.MsgError("err.backup.db_restore_failed_rolled_back", stack)
|
||
}
|
||
}
|
||
if err := restartStack(); err != nil {
|
||
return res, util.MsgError("err.backup.a_z_ujrainditasa_sikertelen_a_fajlok", stack, err)
|
||
}
|
||
// R-475: an update hold may now name the OFF-SITE copy, and this is the route back it names. The
|
||
// unit restore clears it in RestoreFromRecoveryUnitAt; this path never went through that function,
|
||
// so without this line a successful off-site restore would leave the app refusing its next start.
|
||
// Pinned by TestR475_OffsiteRestoreClearsAnUpdateHold.
|
||
m.clearUpdateHoldAfterRestore(stack)
|
||
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
|
||
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
|
||
}
|
||
|
||
// R-382: VolumesReplayed was set above and never printed, so the operator log said
|
||
// "0 file(s) placed, 1 DB dump(s) replayed" on a run that returned a 52 MB Postgres data
|
||
// directory — less informative than the customer's own flash, which already named the volumes.
|
||
m.logger.Printf("[INFO] [offbox] reconstituted %s from snapshot %s: %d file(s) placed, %d volume(s) replayed, %d DB dump(s) replayed, safety dump=%s, skewed=%v",
|
||
stack, id, res.FilesPlaced, res.VolumesReplayed, res.DBsReplayed, filepath.Base(safety), res.Skewed)
|
||
return res, nil
|
||
}
|
||
|
||
// OffsitePairInfo describes the {DB, files} pair sitting in a prepared full-restore scratch, so the
|
||
// confirm dialog can tell the customer what they are about to restore BEFORE they commit to it.
|
||
// Everything here is honesty-surface: none of it blocks the operation.
|
||
type OffsitePairInfo struct {
|
||
Ready bool
|
||
DumpsAt time.Time // when the DB half was taken (zero = legacy unit, age unknown)
|
||
Skewed bool // no coherence stamp → the two halves may be from different times
|
||
LooksEmpty bool // R-44 sniff: the dump has an accounts table with no rows
|
||
HasDump bool
|
||
}
|
||
|
||
// OffsiteScratchPair reads the prepared scratch's unit manifest and reports what the pair looks
|
||
// like. Cheap and read-only — safe to call from a page render.
|
||
func (m *Manager) OffsiteScratchPair(stack string) OffsitePairInfo {
|
||
var info OffsitePairInfo
|
||
if !isSafeStackName(stack) {
|
||
return info
|
||
}
|
||
scratch, _, err := m.offboxRestoreScratchDir(stack)
|
||
if err != nil {
|
||
return info
|
||
}
|
||
// The unit sits at <scratch>/<oldNs>/backups/primary/<stack>; the old namespace is unknown here,
|
||
// so find it rather than reconstructing it.
|
||
unit := findScratchUnitDir(scratch, stack)
|
||
if unit == "" {
|
||
return info
|
||
}
|
||
info.Ready = true
|
||
dumpDir := filepath.Join(unit, "db-dumps")
|
||
if entries, rErr := os.ReadDir(dumpDir); rErr == nil {
|
||
for _, e := range entries {
|
||
if !e.IsDir() && filepath.Ext(e.Name()) == ".sql" && !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
|
||
info.HasDump = true
|
||
break
|
||
}
|
||
}
|
||
}
|
||
if man := readManifest(filepath.Join(unit, "manifest.json")); man != nil {
|
||
if man.DumpsAt != "" {
|
||
if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil {
|
||
info.DumpsAt = t
|
||
}
|
||
}
|
||
info.Skewed = man.OffsiteRunID == ""
|
||
} else {
|
||
info.Skewed = true
|
||
}
|
||
if info.HasDump {
|
||
info.LooksEmpty = m.sniffScratchDump(dumpDir, stack)
|
||
}
|
||
return info
|
||
}
|
||
|
||
// findScratchUnitDir locates `backups/primary/<stack>` anywhere under a restored scratch. restic
|
||
// rebuilds absolute source paths under the target, and the snapshot may have come from a drive that
|
||
// no longer exists on this box, so the prefix cannot be assumed.
|
||
func findScratchUnitDir(scratch, stack string) string {
|
||
found := ""
|
||
suffix := filepath.Join("backups", "primary", stack)
|
||
_ = filepath.Walk(scratch, func(path string, fi os.FileInfo, err error) error {
|
||
if err != nil || found != "" {
|
||
return nil //nolint:nilerr // a walk error on one branch must not abort the search
|
||
}
|
||
if fi.IsDir() && strings.HasSuffix(path, suffix) {
|
||
found = path
|
||
}
|
||
return nil
|
||
})
|
||
return found
|
||
}
|
||
|
||
// sniffScratchDump runs the R-44 content sniff over the dump about to be replayed. Best-effort and
|
||
// warn-level: any failure to read simply reports "no warning", because a sniff that blocks a
|
||
// restore is worse than the skew it describes.
|
||
func (m *Manager) sniffScratchDump(dumpDir, stack string) bool {
|
||
entries, err := os.ReadDir(dumpDir)
|
||
if err != nil {
|
||
return false
|
||
}
|
||
for _, e := range entries {
|
||
name := e.Name()
|
||
if e.IsDir() || filepath.Ext(name) != ".sql" || strings.HasPrefix(name, preRestoreDumpPrefix) {
|
||
continue
|
||
}
|
||
dbType := DBTypePostgres
|
||
if strings.Contains(name, string(DBTypeMariaDB)) {
|
||
dbType = DBTypeMariaDB
|
||
}
|
||
if v := ValidateDump(filepath.Join(dumpDir, name), dbType); v.LooksEmpty {
|
||
m.logger.Printf("[WARN] [offbox] %s: the snapshot dump %s has no account rows — it may predate the customer's data", stack, name)
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|