Files
felhom-controller/controller/internal/backup/offbox_reconstitute.go
T
admin 08eb1a6e3a
gates / gates (push) Failing after 12s
R-356: the off-site restore refused every app that has no data drive
ReconstituteFromOffsite and PlaceOffsiteRestore both resolved the restore
destination with the RAW HDD_PATH and read an empty answer as "the app is not
installed". For 40 of the 53 catalogue apps that answer is correctly empty and
permanent, so both actions refused forever for a running, healthy app — and told
the customer to reinstall it "in the same place", which those apps never offer.

Separate the two questions. "Installed?" is asked of ListDeployedStacks via a new
Manager.isStackDeployed that fails CLOSED on a nil provider. "Where?" is answered
by GetAppDrivePath — the same resolver CaptureRecoveryUnit wrote the snapshot
with, so the restore aims at the place the backup came from.

The 13 drive apps are unchanged: own drive, mismatch check, ack still required.
A third refusal, with its own sentence, covers installed-but-no-resolvable-root.

Fixtures that marked an app "installed" by giving it an HDD path now state
deployment as its own fact. No assertion weakened.
2026-08-22 13:08:46 +02:00

553 lines
27 KiB
Go

package backup
import (
"context"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
"time"
)
// Offsite reconstitution (R-43, v0.148.0) — the leg that was missing.
//
// Until v0.148.0 NO offsite path could restore a database. The two „visszaállítás" buttons staged
// files into a scratch folder and never touched postgres; the place-to-live button merged only the
// files MISSING from the live tree (`rsync --ignore-existing`) and never replayed a dump. For a
// DB-indexed app — most of the catalog — that combination cannot bring content back: the bytes
// return and the application still cannot see them, because its index lives in the database.
// Measured live on 2026-07-19 (DIAG-immich-restore-2026-07-19): 11 photos, files intact on disk,
// timeline empty, two "successful" restores that merged 0 files.
//
// ReconstituteFromOffsite is the honest version of that operation: it takes the CHOSEN snapshot's
// coherent pair and makes the live app equal to it — files overwritten to the snapshot's version,
// database replayed from the same snapshot's dump, app restarted. It is deliberately a different
// function from PlaceOffsiteRestore rather than a flag on it, because the two have opposite file
// semantics and conflating them is exactly how the missing-only merge came to be presented as a
// restore.
//
// Two invariants hold throughout:
//
// - NOTHING IS EVER DELETED. The file copy overwrites and adds; it never carries `--delete`. A
// file the customer created after the snapshot survives the restore as an extra. That is the
// house boundary — a restore that silently removed newer work would be a data-loss event
// wearing a recovery button's label.
// - THE UNDO EXISTS BEFORE THE ACT. A safety dump of the live database is written, and verified
// present on disk, BEFORE anything is stopped, overwritten or replayed. If that dump cannot be
// taken, the whole operation refuses with zero changes — a replay whose previous state was not
// captured is not a restore, it is an overwrite with no way back.
// offsitePreDump runs the coherence pre-phase's dump leg (nil seam → runDBDumpsInternal, which also
// refreshes the recovery units so the manifests enumerate the dumps just written). Extracted as a
// seam because the ORDER — dumps strictly before the restic capture — is the entire mechanism of
// R-44, and an ordering guarantee that no test can observe is one refactor away from silently
// reverting to the behaviour that produced DIAG-immich-restore-2026-07-19.
func (m *Manager) offsitePreDump(ctx context.Context) error {
if m.offsitePreDumpFn != nil {
return m.offsitePreDumpFn(ctx)
}
return m.runDBDumpsInternal(ctx)
}
// SetOffsitePreDumpFn overrides the offsite dump pre-phase (tests; no Docker needed).
func (m *Manager) SetOffsitePreDumpFn(fn func(ctx context.Context) error) { m.offsitePreDumpFn = fn }
// preRestoreDumpPrefix marks the safety dumps taken immediately before a reconstitution. They live
// in the app's own unit db-dumps dir so `ListDumpFiles` surfaces them beside the regular dumps —
// they ARE the undo, and an undo the customer cannot see is not much of one. The regular replay
// loop matches `<stack>-<dbtype>.sql` exactly, so a prefixed file is never mistaken for a source.
const preRestoreDumpPrefix = "pre-restore-"
// OffsiteReconstituteResult reports what a reconstitution actually did, so the flash can state an
// OUTCOME instead of a mechanism. Every field here exists because the v0.147 flash could not say it.
type OffsiteReconstituteResult struct {
SnapshotID string
FilesPlaced int
DBsReplayed int
// VolumesReplayed (R-354) is how many named-volume archives came back from the snapshot. It is on
// the result for the same reason every other field here is: so the OUTCOME can state what
// happened rather than a mechanism. Without it the message said "5 fájl visszaállítva" over a
// restore that had silently dropped a 1.4 MB volume archive — a true sentence leaving a false
// impression, which is the shape this surface keeps having removed from it.
VolumesReplayed int
SafetyDump string // path of the pre-restore dump (the undo), "" when the app has no DB
DumpsAt time.Time // when the snapshot's DB half was taken (zero = unknown/legacy unit)
OffsiteRunID string // "" for a pre-v0.148 snapshot — an unverified pair
Skewed bool // the snapshot carries no coherence stamp: files and DB may differ in age
LooksEmpty bool // R-44 sniff on the dump about to be replayed
// Placement (R-351) is what the backup recorded about where this app's data lived, compared
// against where this restore actually wrote. Carried on the RESULT and not only on the refusal,
// so a restore that proceeded into a different destination says so in its own outcome rather
// than reporting a bare success — a warning beside a success is read as a success, so the
// difference has to survive into the message.
Placement PlacementCheck
}
// fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite).
// Deliberately NOT placeCopier(): that one is `--ignore-existing`, whose whole purpose is to leave
// live files alone, which is precisely what a full restore must not do.
func (m *Manager) fullPlaceCopier() func(src, dst string) (int, error) {
if m.offboxFullPlaceCopier != nil {
return m.offboxFullPlaceCopier
}
return rsyncRestoreOverwrite
}
// rsyncRestoreOverwrite copies src over dst: `rsync -a --itemize-changes`, with NO
// `--ignore-existing` (a changed file becomes the snapshot's version) and NO `--delete` (an extra
// file at dst survives). Returns the number of regular files transferred.
func rsyncRestoreOverwrite(src, dst string) (int, error) {
if err := os.MkdirAll(dst, 0755); err != nil {
return 0, fmt.Errorf("mkdir %s: %w", dst, err)
}
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
defer cancel()
cmd := exec.CommandContext(ctx, "rsync", "-a", "--itemize-changes",
strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/")
out, err := cmd.CombinedOutput()
if err != nil {
return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out)))
}
return countRestoredFiles(string(out)), nil
}
// writeSafetyDump dumps every live database of stack into the app's unit db-dumps dir under the
// `pre-restore-` prefix, and returns the first dump's path. Returns ("", nil) when the app has no
// database at all — a no-DB app has nothing to undo and must flow exactly as it did before
// v0.148.0 (no dump, no replay, no behaviour change).
//
// A discovered database that CANNOT be dumped is a hard error: it means the undo would not exist.
func (m *Manager) writeSafetyDump(ctx context.Context, stackName, nsRoot string) (string, error) {
discover := m.discoverDBs
if discover == nil {
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
}
}
dbs, err := discover(ctx)
if err != nil {
return "", fmt.Errorf("a biztonsági mentés előtt nem sikerült felderíteni az adatbázisokat: %w", err)
}
var mine []DiscoveredDB
for _, db := range dbs {
if db.StackName == stackName {
mine = append(mine, db)
}
}
if len(mine) == 0 {
return "", nil // no DB → nothing to undo → scenario E flows unchanged
}
dumpDir := AppDBDumpPath(nsRoot, stackName)
if err := os.MkdirAll(dumpDir, 0755); err != nil {
return "", fmt.Errorf("a biztonsági mentés könyvtára nem hozható létre: %w", err)
}
stamp := time.Now().UTC().Format("20060102T150405Z")
first := ""
for _, db := range mine {
res := m.dumpForSafety(ctx, db, dumpDir)
if res.Error != nil {
return "", fmt.Errorf("a jelenlegi adatbázis biztonsági mentése sikertelen (%s): %w — a visszaállítás nem indult el", db.ContainerName, res.Error)
}
// DumpOne writes `<stack>-<dbtype>.sql`; rename it under the safety prefix so it can never be
// picked up as a replay SOURCE and can never overwrite the app's real dump.
safe := filepath.Join(dumpDir, fmt.Sprintf("%s%s-%s-%s.sql", preRestoreDumpPrefix, stamp, stackName, db.DBType))
if res.FilePath != safe {
if err := os.Rename(res.FilePath, safe); err != nil {
return "", fmt.Errorf("a biztonsági mentés véglegesítése sikertelen: %w", err)
}
}
if first == "" {
first = safe
}
m.logger.Printf("[INFO] [offbox] %s: pre-restore safety dump written → %s (%s)", stackName, filepath.Base(safe), humanizeBytes(res.Size))
}
return first, nil
}
// dumpForSafety is the DumpOne seam for the safety dump (tests inject; nil → the real DumpOne).
func (m *Manager) dumpForSafety(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
if m.safetyDumpFn != nil {
return m.safetyDumpFn(ctx, db, dumpDir)
}
return DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
}
// ReconstituteFromOffsite makes the live app equal to a restored full-scratch snapshot: files
// overwritten to the snapshot's version (extras survive, nothing deleted), then the snapshot's own
// DB dump replayed, with a safety dump of the current database taken first. Requires a completed
// FULL scratch restore (RestoreOffboxScratch with full=true). Single-flight.
// ackPlacementChange (R-351) is the customer's DELIBERATE acknowledgement that the destination
// differs from the one the backup recorded. It is a separate act from the restore's own confirm:
// folding it into `confirm=1` would mean one click carried two decisions, which is precisely what
// R-48 exists to prevent.
func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ackPlacementChange bool) (OffsiteReconstituteResult, error) {
var res OffsiteReconstituteResult
if !m.OffboxConfigured() {
return res, fmt.Errorf("off-box backup not configured")
}
if !isSafeStackName(stack) {
return res, fmt.Errorf("invalid stack name")
}
if m.stackProvider == nil {
return res, fmt.Errorf("stack provider not configured")
}
if err := m.acquireRunning(); err != nil {
return res, fmt.Errorf("egy másik mentési/visszaállítási művelet már fut")
}
defer m.releaseRunning()
scratch, _, err := m.offboxRestoreScratchDir(stack)
if err != nil {
return res, err
}
if _, sErr := os.Stat(scratch); sErr != nil {
return res, fmt.Errorf("nincs előkészített teljes visszaállítás — futtass előbb egy teljes visszaállítást")
}
id, paths, err := m.offboxLatestSnapshot(ctx, stack)
if err != nil {
return res, err
}
res.SnapshotID = id
// R-253: the same sentence the restore page now shows, so the page and the refusal cannot
// drift apart again. It is a REFUSAL, not a failure — the data is untouched and the customer
// has one step to take. The restore deliberately does NOT deploy the app itself: the
// destination is the app's own HDD path, which is a drive the CUSTOMER chooses at deploy
// time, and picking it for them is the decision this whole recovery path exists to leave
// with them.
// R-351: the refusal now NAMES the place the backup recorded, when it can read it. The
// prepared scratch already contains the unit, so this is a local file read — no network call,
// nothing restored, and it happens on a path that was going to refuse anyway. Telling
// somebody to reinstall without telling them where the data belongs is what forced the
// 2026-08-21 operator to remember two values the backup already held.
// R-356: this refusal used to be reached by `GetStackHDDPath(stack) == ""` — one predicate
// answering two questions. It now covers ONLY "the app is not deployed", and it stopped
// covering "the app has no drive". The reason the two came apart: the drive choice is the
// CUSTOMER's, and 40 of the 53 catalog apps were never offered one — they have no choice to
// leave with them, and their data lives on the system data path by design. The R-253 decision
// above is untouched for the 13 apps that DO have a drive to get wrong.
if !m.isStackDeployed(stack) {
if rec := m.recordedPlacementFromScratch(scratch); rec.Known() {
return res, fmt.Errorf("a(z) %s nincs telepítve, ezért nincs hová visszaállítani az adatait. "+
"A mentése szerint az adatai itt voltak: %s. Telepítsd újra az alkalmazást (Alkalmazások) "+
"ugyanerre a helyre, utána ez a visszaállítás működni fog", stack, rec.Drive)
}
return res, fmt.Errorf("a(z) %s nincs telepítve, ezért nincs hová visszaállítani az adatait — "+
"telepítsd újra az alkalmazást (Alkalmazások), utána ez a visszaállítás működni fog", stack)
}
// The destination is resolved by the SAME rule the capture side used to write this snapshot
// (CaptureRecoveryUnit → GetAppDrivePath): the app's drive if it has one, the system data path
// otherwise. Anything else and the restore would aim at a different place than the backup came
// from, which is the mismatch prompt firing on a box where nothing actually moved.
hdd := strings.TrimSpace(m.GetAppDrivePath(stack))
if hdd == "" {
// A DIFFERENT failure from the one above, so it gets a different sentence: the app IS
// installed, but the box cannot name its own data root (systemDataPath unset). Saying
// "nincs telepítve" here would send the customer to reinstall an app that is already
// running, and the real fault would stay invisible.
return res, fmt.Errorf("a(z) %s telepítve van, de a vezérlő nem tudja megállapítani, hová tartoznak az adatai "+
"(nincs beállítva rendszer-adatterület). Ellenőrizd a tárhely beállításait (Tárhely), utána indítsd újra a visszaállítást", stack)
}
liveNs := m.namespaceRoot(hdd)
placements, err := mapOffsiteRestorePaths(paths, stack, scratch, liveNs)
if err != nil {
return res, err // whole-placement refusal (no partial writes)
}
// Stat pre-pass over EVERY placement before the first copy — an incomplete scratch (e.g. only a
// unit-only restore was run) refuses with ZERO copies.
for _, pl := range placements {
if _, sErr := os.Stat(pl.src); sErr != nil {
return res, fmt.Errorf("a teljes visszaállítás hiányos (%s nincs meg) — futtass előbb egy teljes visszaállítást", filepath.Base(pl.src))
}
}
// The snapshot's coherence stamp, read from the RESTORED unit manifest (not the live one).
scratchUnit := ""
for _, pl := range placements {
if pl.isUnit {
scratchUnit = pl.src
break
}
}
if scratchUnit == "" {
return res, fmt.Errorf("a pillanatképben nincs mentési egység — a visszaállítás nem indítható")
}
scratchDumpDir := filepath.Join(scratchUnit, "db-dumps")
man := readManifest(filepath.Join(scratchUnit, "manifest.json"))
if man != nil {
res.OffsiteRunID = man.OffsiteRunID
if man.DumpsAt != "" {
if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil {
res.DumpsAt = t
}
}
}
// --- WHERE THE BACKUP SAYS THIS DATA LIVED (R-351) ------------------------------------------
// The manifest we just opened has carried `drive` and `namespace_root` since schema 1, and until
// now nothing read them back. Compared HERE, before the safety dump and before the first byte is
// placed, so the refusal costs nothing and leaves the app completely untouched.
//
// An UNKNOWN recording (a pre-field unit, or one we could not read) is not a mismatch and does
// not refuse: blocking on an absence would strand every older backup, and CheckPlacement returns
// that case explicitly rather than letting it fall through as "they match".
res.Placement = CheckPlacement(man, hdd, liveNs)
if res.Placement.Mismatch && !ackPlacementChange {
return res, fmt.Errorf("%s", PlacementMismatchMessage(stack, res.Placement))
}
if res.Placement.Mismatch {
m.logger.Printf("[WARN] [offbox] %s: restoring into %s, but the backup recorded %s — the customer acknowledged the change",
stack, res.Placement.LiveDrive, res.Placement.Recorded.Drive)
}
// A pre-v0.148 snapshot carries no stamp: its dump was whatever the 02:30 local run left behind,
// so the pair's two halves may be hours or days apart. Surfaced, never blocked — the confirm
// dialog says so and the safety dump makes it reversible.
res.Skewed = res.OffsiteRunID == ""
res.LooksEmpty = m.sniffScratchDump(scratchDumpDir, stack)
// --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------
// Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir,
// so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first
// mutation so the refusal below costs nothing.
var dbServices []string
if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
svcs, dsErr := DBServiceNames(composePath)
if dsErr != nil {
// "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse.
m.logger.Printf("[WARN] [offbox] %s: could not read the live compose services: %v", stack, dsErr)
}
dbServices = svcs
}
// --- THE UNDO, BEFORE THE ACT ---------------------------------------------------------------
// Taken while the stack is still UP (a stopped database cannot be dumped) and before a single
// byte is overwritten, so a failure here aborts with the live app completely untouched.
safety, err := m.writeSafetyDump(ctx, stack, liveNs)
if err != nil {
return res, err
}
res.SafetyDump = safety
hasDB := safety != ""
if hasDB {
if _, sErr := os.Stat(safety); sErr != nil {
// Fail-closed: never replay when the undo is not verifiably on disk.
return res, fmt.Errorf("a biztonsági mentés nem található a lemezen — a visszaállítás biztonsági okból nem indult el")
}
// Fail-closed (R-47): the app HAS a database but no compose service can be identified to
// start alone for the replay. The only alternative would be to start everything and replay
// into the race that produced H4 — refusing with the live app untouched is the better outcome.
if len(dbServices) == 0 {
return res, fmt.Errorf("Az adatbázis-szolgáltatás nem azonosítható a(z) %s alkalmazásban — a visszaállítás biztonsági okból nem indult el.", stack)
}
}
// --- FILES ----------------------------------------------------------------------------------
// R-166: mark the stop→restore→start window BEFORE stopping. A controller killed anywhere inside
// it used to leave the app down with nothing on disk recording that it was owed a restart — and a
// full offsite restore is a LONG window, so this is the shape most likely to be interrupted.
if err := m.appStop.Begin("offbox-reconstitute:"+stack, ReasonOffboxReconstitute, []string{stack}); err != nil {
return res, fmt.Errorf("a(z) %s leállítása előtti jelölő nem menthető: %w", stack, err)
}
// restartStack starts the app and clears the marker ONLY when the start actually succeeded — a
// failed start leaves the marker so the next startup retries. Every bring-up below goes through
// it; a bare StartStack here would clear nothing and strand the marker on the success path.
restartStack := func() error {
err := m.stackProvider.StartStack(stack)
if err == nil {
m.appStop.End()
}
return err
}
if err := m.stackProvider.StopStack(stack); err != nil {
m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err)
}
copier := m.fullPlaceCopier()
for _, pl := range placements {
if pl.isUnit {
// The live recovery unit is still never overwritten — it is the LOCAL restore path's
// source and clobbering it would trade one recovery route for another. THAT reason is
// sound and still holds; it is why this skip stays.
//
// R-354 — THE SECOND HALF OF THIS COMMENT USED TO BE FALSE AND IS CORRECTED HERE. It said
// "the snapshot's dump is replayed from the scratch unit instead, so nothing is lost by
// skipping it". That was true of the DATABASE dump and false of the VOLUME archives, which
// live in the same unit and were replayed by nothing at all. Skipping the placement is
// correct; treating the skip as harmless was not. Measured live 2026-08-21: calibre-web's
// 1 422 848-byte `calibre_web_config.tar` was in the unit, in the snapshot and in the
// verification folder, and the restore reported "5 fájl visszaállítva" without it.
//
// Both legs are now replayed FROM THE SCRATCH UNIT below — volumes first, then the DB, so
// the logical dump still wins over any volume-tar copy of the same database.
continue
}
n, cErr := copier(pl.src, pl.dst)
if cErr != nil {
// Best-effort bring-up: leaving the app stopped after a partial copy would turn a failed
// restore into an outage.
if sErr := restartStack(); sErr != nil {
m.logger.Printf("[WARN] [offbox] %s: restart after failed placement also failed: %v", stack, sErr)
}
return res, fmt.Errorf("a(z) %s fájljainak visszaállítása sikertelen: %w", stack, cErr)
}
res.FilesPlaced += n
}
// --- NAMED VOLUMES (R-354) ------------------------------------------------------------------
// Replayed from the SCRATCH unit, exactly as the database dump is, and for the same reason: the
// live unit is never overwritten by a placement, so the snapshot's copy exists only under the
// scratch. Same helper as the local restore path — one implementation, two callers.
//
// ORDER IS LOAD-BEARING and mirrors RestoreFromRecoveryUnit: volumes FIRST, database after, so a
// logical .sql dump still wins over whatever copy of the same database a volume tar happens to
// contain. It also has to happen inside the stopped window, because replacing a named volume means
// removing it, and Docker refuses that while a container holds it.
volReplay := m.volumeReplayFrom
if volReplay == nil {
volReplay = m.restoreDockerVolumesFrom
}
nVols, vErr := volReplay(stack, filepath.Join(scratchUnit, "volume-dumps"))
res.VolumesReplayed = nVols
if vErr != nil {
// A partial replay must never read as a completion. Bring the app back up rather than leaving
// an outage, then surface it — the same shape the file leg above uses.
if sErr := restartStack(); sErr != nil {
m.logger.Printf("[WARN] [offbox] %s: restart after failed volume replay also failed: %v", stack, sErr)
}
return res, fmt.Errorf("a(z) %s adatkötetének visszaállítása sikertelen: %w", stack, vErr)
}
// --- DATABASE -------------------------------------------------------------------------------
// The DB container must be UP for the replay (ImportDump talks to it with its own discovered
// credentials), but NOTHING ELSE may be — R-47. Until v0.153.0 this was a full StartStack, which
// gave the application a window to rebuild the very schema objects the dump was about to create:
// measured at 2 s on 2026-07-19, and the replay aborted `relation "clip_index" already exists`
// under ON_ERROR_STOP=1 (H4). Starting only the database service closes that window entirely.
if hasDB {
if err := m.stackProvider.StartStackServices(stack, dbServices); err != nil {
// Best-effort bring-up: a failed restore must not also be an outage.
if sErr := restartStack(); sErr != nil {
m.logger.Printf("[WARN] [offbox] %s: full start after failed DB-only start also failed: %v", stack, sErr)
}
return res, fmt.Errorf("a(z) %s adatbázis-szolgáltatásának indítása sikertelen: %w", stack, err)
}
n, iErr := m.reimportDBDumpsFrom(ctx, stack, scratchDumpDir)
res.DBsReplayed = n
if iErr != nil {
if sErr := restartStack(); sErr != nil {
m.logger.Printf("[WARN] [offbox] %s: full start after failed replay also failed: %v", stack, sErr)
}
return res, fmt.Errorf("az adatbázis visszaállítása sikertelen: %w — a korábbi állapot mentése megvan: %s", iErr, filepath.Base(safety))
}
}
if err := restartStack(); err != nil {
return res, fmt.Errorf("a(z) %s újraindítása sikertelen a fájlok visszaállítása után: %w", stack, err)
}
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
}
m.logger.Printf("[INFO] [offbox] reconstituted %s from snapshot %s: %d file(s) placed, %d DB dump(s) replayed, safety dump=%s, skewed=%v",
stack, id, res.FilesPlaced, res.DBsReplayed, filepath.Base(safety), res.Skewed)
return res, nil
}
// OffsitePairInfo describes the {DB, files} pair sitting in a prepared full-restore scratch, so the
// confirm dialog can tell the customer what they are about to restore BEFORE they commit to it.
// Everything here is honesty-surface: none of it blocks the operation.
type OffsitePairInfo struct {
Ready bool
DumpsAt time.Time // when the DB half was taken (zero = legacy unit, age unknown)
Skewed bool // no coherence stamp → the two halves may be from different times
LooksEmpty bool // R-44 sniff: the dump has an accounts table with no rows
HasDump bool
}
// OffsiteScratchPair reads the prepared scratch's unit manifest and reports what the pair looks
// like. Cheap and read-only — safe to call from a page render.
func (m *Manager) OffsiteScratchPair(stack string) OffsitePairInfo {
var info OffsitePairInfo
if !isSafeStackName(stack) {
return info
}
scratch, _, err := m.offboxRestoreScratchDir(stack)
if err != nil {
return info
}
// The unit sits at <scratch>/<oldNs>/backups/primary/<stack>; the old namespace is unknown here,
// so find it rather than reconstructing it.
unit := findScratchUnitDir(scratch, stack)
if unit == "" {
return info
}
info.Ready = true
dumpDir := filepath.Join(unit, "db-dumps")
if entries, rErr := os.ReadDir(dumpDir); rErr == nil {
for _, e := range entries {
if !e.IsDir() && filepath.Ext(e.Name()) == ".sql" && !strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
info.HasDump = true
break
}
}
}
if man := readManifest(filepath.Join(unit, "manifest.json")); man != nil {
if man.DumpsAt != "" {
if t, pErr := time.Parse(time.RFC3339, man.DumpsAt); pErr == nil {
info.DumpsAt = t
}
}
info.Skewed = man.OffsiteRunID == ""
} else {
info.Skewed = true
}
if info.HasDump {
info.LooksEmpty = m.sniffScratchDump(dumpDir, stack)
}
return info
}
// findScratchUnitDir locates `backups/primary/<stack>` anywhere under a restored scratch. restic
// rebuilds absolute source paths under the target, and the snapshot may have come from a drive that
// no longer exists on this box, so the prefix cannot be assumed.
func findScratchUnitDir(scratch, stack string) string {
found := ""
suffix := filepath.Join("backups", "primary", stack)
_ = filepath.Walk(scratch, func(path string, fi os.FileInfo, err error) error {
if err != nil || found != "" {
return nil //nolint:nilerr // a walk error on one branch must not abort the search
}
if fi.IsDir() && strings.HasSuffix(path, suffix) {
found = path
}
return nil
})
return found
}
// sniffScratchDump runs the R-44 content sniff over the dump about to be replayed. Best-effort and
// warn-level: any failure to read simply reports "no warning", because a sniff that blocks a
// restore is worse than the skew it describes.
func (m *Manager) sniffScratchDump(dumpDir, stack string) bool {
entries, err := os.ReadDir(dumpDir)
if err != nil {
return false
}
for _, e := range entries {
name := e.Name()
if e.IsDir() || filepath.Ext(name) != ".sql" || strings.HasPrefix(name, preRestoreDumpPrefix) {
continue
}
dbType := DBTypePostgres
if strings.Contains(name, string(DBTypeMariaDB)) {
dbType = DBTypeMariaDB
}
if v := ValidateDump(filepath.Join(dumpDir, name), dbType); v.LooksEmpty {
m.logger.Printf("[WARN] [offbox] %s: the snapshot dump %s has no account rows — it may predate the customer's data", stack, name)
return true
}
}
return false
}