0f9b796615
Tier-2 mirrors each app's whole recovery unit to <dest>/backups/secondary/<app>/recovery-unit/ on every run and has done for months. Nothing read it. In the one failure Tier-2 exists for - the primary drive is lost, and the primary unit with it - the surviving copy could not be opened by any action in the product (07-backup-architecture 6.3, 7.2). Part 1.2: RestoreFromRecoveryUnitAt(stack, unitDir) holds the whole body; RestoreFromRecoveryUnit is the thin caller naming the primary unit. ONE implementation, two callers. The SOURCE moves; the DESTINATION does not - live Docker volumes, the live database container, the guest's definition, all unchanged. The R-47 mutation order, the secret reconciliation with unit-over-guest precedence, the fail-closed data-key gate and the no-unit fallback with CountsUnknown are untouched. reimportDBDumpsAtCtx is the bounded-context twin of reimportDBDumpsFrom; the 35-minute bound is now named once so the two paths cannot drift. The R-354 volume-replay seam is reused rather than a second one invented, which is what lets the acceptance test assert the volume leg's source directory. Part 1.3: RestoreTier2Unit resolves the recorded copy, refuses fail-closed unless the mirror carries a parseable manifest - a directory is not a package - and delegates. The single-writer flag is taken inside RestoreFromRecoveryUnitAt, not beside it. Part 2.1: Tier2Coverage gains UnitRestorable and the copy's dates. CanRestore() is NOT widened; it still answers only 'can the file restore run?'. One predicate answering two questions is R-356, which refused 40 running apps for months. Tests: A2-A6 and B1-B5, plus two non-regression guards. The Tier-2 fixtures build their mirror with the production RunTier2, so the claim is 'the copy Tier-2 writes is the copy this restore reads'. Red-proofs: A5 (swap volumes/recreate -> fails on the order), B2 (point the reader back at the primary -> fails with the mirror never reaching the redeploy, and with permission denied once the primary tree is unreadable).
357 lines
18 KiB
Go
357 lines
18 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// Tier-2 in-place file restore (TASK C2, closes drill finding F2): the customer-facing recovery for
|
|
// class-C data — HDD bind-mount user files under appdata/<stack>. Restores MISSING files from the
|
|
// recorded Tier-2 copy back into the live appdata dir, and touches NOTHING else:
|
|
//
|
|
// - a file that exists live is NEVER overwritten (a customer edit after the last Tier-2 run wins);
|
|
// - a live file absent from the backup is NEVER deleted (that is what rsyncMirror's --delete would
|
|
// do in this direction — the catastrophic trap this helper exists to avoid);
|
|
// - only files present in the copy and missing live are copied back (attrs preserved).
|
|
//
|
|
// This exactly serves the "I deleted my files" scenario. Corruption / point-in-time rollback stays
|
|
// with the offbox restore-to-verify + operator paths — deliberately out of scope.
|
|
|
|
// Refusal reasons (customer-readable — they surface verbatim in the flash message).
|
|
var (
|
|
errNoTier2Copy = errors.New("nincs másodlagos fájlmásolat ehhez az alkalmazáshoz")
|
|
errTier2DriveGone = errors.New("a másodlagos meghajtó nincs csatlakoztatva")
|
|
errLiveDriveGone = errors.New("az alkalmazás meghajtója nincs csatlakoztatva")
|
|
errLiveDriveDecommed = errors.New("az alkalmazás meghajtója le van szerelve")
|
|
// errTier2OldLayout (3b, §7-G2): the recorded copy predates the v2 relpath-mirroring layout (no
|
|
// marker). Refuse rather than read a flat layout we no longer understand — safe, because tier-2
|
|
// restore is missing-file recovery and the live data still exists in that scenario.
|
|
errTier2OldLayout = errors.New("A 2. mentés régi formátumú — futtass előbb egy új másodlagos mentést.")
|
|
// ErrTier2NoRestorableData (C9-F1) — this app HAS a Tier-2 copy, but that copy contains no subtree
|
|
// this restore can read: its data lives entirely in Docker named volumes, which are captured into
|
|
// recovery-unit/ (db-dumps + volume-dumps) and NEVER read by this path. 43 of the 53 catalog apps
|
|
// are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action
|
|
// that does work, instead of taking an outage and reporting "no missing files".
|
|
ErrTier2NoRestorableData = errors.New("ennek az alkalmazásnak az adatai nem ebből a másolatból állíthatók vissza")
|
|
// ErrTier2NoUnitInCopy (R-102) — the recorded Tier-2 copy holds no OPENABLE recovery unit: either
|
|
// recovery-unit/ is absent, or it is a directory without a readable manifest.json. Exported so the
|
|
// handler can refuse before beginning any op. FAIL CLOSED is the whole point of the second half:
|
|
// a directory that exists is not a package, and reading a half-copied mirror as if it were one is
|
|
// how a restore would overwrite live data with nothing.
|
|
ErrTier2NoUnitInCopy = errors.New("a másodlagos másolatban nincs megnyitható mentési egység ehhez az alkalmazáshoz")
|
|
)
|
|
|
|
// Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1
|
|
// is about. Computed from the RECORDED copy on disk, never guessed from the catalog, so an app whose
|
|
// template changed is judged by what its actual copy holds.
|
|
//
|
|
// The distinction that matters: Legs are the subtrees RestoreTier2Files reads (hdd/, userdata/);
|
|
// HasUnit means the copy ALSO holds a full recovery unit — the app's database dumps and named-volume
|
|
// tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in
|
|
// which case the restore is a guaranteed no-op no matter how much data was lost.
|
|
type Tier2Coverage struct {
|
|
Legs []string // subtrees the FILE restore reads and that exist in the copy: "hdd", "userdata"
|
|
HasUnit bool // recovery-unit/ present as a DIRECTORY — the disclosure fact, see below
|
|
|
|
// UnitRestorable (R-102) — the mirror is a real PACKAGE, not merely a directory: recovery-unit/
|
|
// exists AND carries a manifest.json that parses. This is the gate for the UNIT restore.
|
|
//
|
|
// It is a SECOND FIELD and not a widening of HasUnit, and the distinction is load-bearing in both
|
|
// directions. HasUnit answers "is there captured data this FILE restore is not looking at?" — the
|
|
// question tier2UnitNotCoveredMsg is appended for, and the honest answer for a half-copied mirror
|
|
// is still yes. UnitRestorable answers "can the unit restore open this?" — and for that same
|
|
// half-copied mirror the answer is no. Collapsing them would either silence a true disclosure or
|
|
// arm a restore over an unopenable package.
|
|
UnitRestorable bool
|
|
|
|
// CopyLastRun / CopyLastSuccess — WHEN the copy this restore would read was written, so the
|
|
// surface can name the date before it overwrites anything with it (Scenario E). Filled only by
|
|
// Tier2RestoreCoverage, which is the path that holds the settings; tier2CoverageAt is a pure
|
|
// filesystem inspection and leaves them empty. They are STRINGS in the recorded RFC3339 form,
|
|
// carried verbatim — no formatting decision is taken in this package.
|
|
//
|
|
// R-101 applies here exactly as it does on the backup card: CopyLastRun is the ATTEMPT clock and
|
|
// CopyLastSuccess is the only evidence a copy was actually made. A surface that shows one must
|
|
// not present it as the other.
|
|
CopyLastRun string
|
|
CopyLastSuccess string
|
|
}
|
|
|
|
// CanRestore reports whether the FILE restore has any subtree to read at all.
|
|
//
|
|
// R-102/R-103: this answers exactly one question and must keep answering only that one. Widening it
|
|
// to include the unit is R-356 arriving a second time — there, ONE predicate meant both "has this app
|
|
// a drive?" and "is this app installed?", and it refused 40 running apps for months while telling
|
|
// their owners to reinstall them somewhere those apps never offer. Two questions, two predicates.
|
|
func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 }
|
|
|
|
// CanRestoreUnit reports whether the UNIT restore can run from this copy — the second predicate.
|
|
func (c Tier2Coverage) CanRestoreUnit() bool { return c.UnitRestorable }
|
|
|
|
// tier2UnitDir returns the recovery-unit directory inside a resolved Tier-2 copy. ONE expression of
|
|
// where Tier-2 puts the mirror; tier2.go writes it at the same relative name ("Unit leg (always)").
|
|
func tier2UnitDir(destBase string) string {
|
|
return filepath.Join(destBase, "recovery-unit")
|
|
}
|
|
|
|
// tier2UnitIsOpenable reports whether a mirrored unit directory is a PACKAGE and not just a
|
|
// directory. Fail-closed by construction: the manifest must be present AND parse (readManifest
|
|
// returns nil for both a missing file and malformed JSON), because an unopenable unit that armed a
|
|
// restore would stop the app, replay nothing, and rewrite its definition from an empty capture.
|
|
//
|
|
// This is the R-358 lesson one tier over — the off-site scratch marker — arriving on Tier-2.
|
|
func tier2UnitIsOpenable(unitDir string) bool {
|
|
if fi, err := os.Stat(unitDir); err != nil || !fi.IsDir() {
|
|
return false
|
|
}
|
|
return readManifest(UnitManifestFile(unitDir)) != nil
|
|
}
|
|
|
|
// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat/read — no side effects.
|
|
func tier2CoverageAt(destBase string) Tier2Coverage {
|
|
var c Tier2Coverage
|
|
for _, leg := range []string{"hdd", "userdata"} {
|
|
if fi, err := os.Stat(filepath.Join(destBase, leg)); err == nil && fi.IsDir() {
|
|
c.Legs = append(c.Legs, leg)
|
|
}
|
|
}
|
|
unitDir := tier2UnitDir(destBase)
|
|
if fi, err := os.Stat(unitDir); err == nil && fi.IsDir() {
|
|
c.HasUnit = true
|
|
}
|
|
c.UnitRestorable = tier2UnitIsOpenable(unitDir)
|
|
return c
|
|
}
|
|
|
|
// Tier2RestoreCoverage resolves the app's RECORDED Tier-2 copy and reports what a restore could
|
|
// return from it. Errors are the same refusals RestoreTier2Files itself would raise, so the caller
|
|
// can surface them before starting anything — this is what lets the handler refuse without an outage.
|
|
func (m *Manager) Tier2RestoreCoverage(stackName string) (Tier2Coverage, error) {
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return Tier2Coverage{}, err
|
|
}
|
|
cov := tier2CoverageAt(destBase)
|
|
// R-102: carry WHEN the copy was written, so the surface can name the date on an action that
|
|
// overwrites live data with it. Read from the same recorded config tier2RecordedCopyDir just
|
|
// resolved the path from, so the date and the directory cannot describe different runs.
|
|
if m.settings != nil {
|
|
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil {
|
|
cov.CopyLastRun, cov.CopyLastSuccess = cfg.LastRun, cfg.LastSuccess
|
|
}
|
|
}
|
|
return cov, nil
|
|
}
|
|
|
|
// RestoreTier2Unit runs the FULL recovery-unit restore from the app's Tier-2 copy on the SECOND
|
|
// DRIVE — R-102, and the reason this task exists.
|
|
//
|
|
// Tier-2 has mirrored each app's whole recovery unit to
|
|
// <dest>/backups/secondary/<app>/recovery-unit/ on every run for months, and no code path read it:
|
|
// every reader of a unit could only name a path under backups/primary/. So in the exact failure
|
|
// Tier-2 exists for — the primary drive is lost, and the primary unit with it — the surviving copy
|
|
// was unopenable by any customer action (07-backup-architecture §6.3, §7.2).
|
|
//
|
|
// It is NOT the additive file restore beside it. This one OVERWRITES: named volumes are recreated
|
|
// from the mirror's tars and the database is replayed from the mirror's dump. The surface must carry
|
|
// that difference; see tier2UnitConfirm in the web package.
|
|
//
|
|
// THE SINGLE-WRITER FLAG IS TAKEN INSIDE RestoreFromRecoveryUnitAt, exactly as on the primary path —
|
|
// do NOT add an acquireRunning() here. A second acquire would refuse the restore it is guarding.
|
|
func (m *Manager) RestoreTier2Unit(stackName string) (UnitRestoreResult, error) {
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return UnitRestoreResult{}, err
|
|
}
|
|
unitDir := tier2UnitDir(destBase)
|
|
if !tier2UnitIsOpenable(unitDir) {
|
|
m.logger.Printf("[WARN] [backup] Tier-2 unit restore refused for %s: no openable recovery unit in the recorded copy — the app was NOT stopped", stackName)
|
|
return UnitRestoreResult{}, ErrTier2NoUnitInCopy
|
|
}
|
|
// WARN, not INFO: this is the destructive one of the two Tier-2 restores. The unit directory is a
|
|
// path and never a secret, and naming it is what makes "the SECONDARY mirror was the source" a
|
|
// positive observable in the log rather than an absence to be argued from.
|
|
m.logger.Printf("[WARN] [backup] Tier-2 UNIT restore for %s from the secondary mirror %s — this OVERWRITES live app data", stackName, unitDir)
|
|
return m.RestoreFromRecoveryUnitAt(stackName, unitDir)
|
|
}
|
|
|
|
// Tier2CopyDate returns the date the surface should name for this app's Tier-2 copy, preferring the
|
|
// last SUCCESS over the last ATTEMPT (R-101: a timestamp that records "we tried" cannot answer "did
|
|
// it work"), and reports whether the returned value is a proven success.
|
|
//
|
|
// It exists so the confirm text and the outcome sentence cannot disagree about which copy is being
|
|
// restored: one resolver, two readers.
|
|
func (c Tier2Coverage) Tier2CopyDate() (date string, proven bool) {
|
|
if c.CopyLastSuccess != "" {
|
|
return c.CopyLastSuccess, true
|
|
}
|
|
return c.CopyLastRun, false
|
|
}
|
|
|
|
// tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every
|
|
// source-side refusal in one place so the pre-flight check and the restore itself cannot drift.
|
|
func (m *Manager) tier2RecordedCopyDir(stackName string) (string, error) {
|
|
var destBase string
|
|
if m.settings != nil {
|
|
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil && cfg.LastRun != "" && cfg.DestinationPath != "" {
|
|
if m.settings.IsDisconnected(cfg.DestinationPath) {
|
|
return "", errTier2DriveGone
|
|
}
|
|
destBase = filepath.Join(cfg.DestinationPath, "backups", "secondary", stackName)
|
|
}
|
|
}
|
|
if destBase == "" {
|
|
return "", errNoTier2Copy
|
|
}
|
|
if _, statErr := os.Stat(destBase); statErr != nil {
|
|
return "", errNoTier2Copy // recorded but the copy dir is gone — same honest refusal
|
|
}
|
|
// §7-G2 marker gate: a pre-v2 (flat) copy has no marker → refuse rather than read a layout we no
|
|
// longer understand (live data still exists for missing-file recovery).
|
|
if _, mErr := os.Stat(filepath.Join(destBase, tier2LayoutMarker)); mErr != nil {
|
|
return "", errTier2OldLayout
|
|
}
|
|
return destBase, nil
|
|
}
|
|
|
|
// RestoreTier2Files restores the app's MISSING user files in place from its recorded Tier-2 copy
|
|
// (additive-only; see the package comment above). Returns how many regular files were copied back.
|
|
//
|
|
// The source is the RECORDED Tier-2 destination (settings.CrossDriveBackup.DestinationPath) — never
|
|
// a fresh selectTier2Target, which could re-pick a different (empty) drive and "restore" nothing.
|
|
// All refusals happen BEFORE the app is stopped. Stop-first is the locked consistency policy: the
|
|
// app must not be reorganizing its data dir mid-copy.
|
|
func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err error) {
|
|
if m.stackProvider == nil {
|
|
return 0, fmt.Errorf("stack provider not configured")
|
|
}
|
|
if err := m.acquireRunning(); err != nil {
|
|
return 0, err // shares the backup/restore single-flight — must not race a running backup
|
|
}
|
|
defer m.releaseRunning()
|
|
|
|
// Live side: the app's drive must be present and in service.
|
|
drive := m.GetAppDrivePath(stackName)
|
|
if drive == "" || !filepath.IsAbs(drive) {
|
|
return 0, fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
if m.settings != nil {
|
|
if m.settings.IsDisconnected(drive) {
|
|
return 0, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
|
|
}
|
|
if m.settings.IsDecommissioned(drive) {
|
|
return 0, fmt.Errorf("%w (%s)", errLiveDriveDecommed, drive)
|
|
}
|
|
}
|
|
// v2 relpath-mirroring: liveNsRoot == the app's HDD_PATH (Model A). The dest hdd/ and userdata/
|
|
// subtrees mirror the live relpath structure exactly, so restore is two whole-subtree merges (N>1
|
|
// dirs + nested binds handled natively — no per-appdata-dir resolution, no N>1 refusal).
|
|
liveNsRoot := m.namespaceRoot(drive)
|
|
|
|
// Source side: the RECORDED Tier-2 copy must exist, its drive connected, and it must be v2.
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
|
|
// C9-F1: refuse BEFORE the app is stopped if this copy holds nothing this path can read. Without
|
|
// this the app was stopped, zero files were copied, it was restarted, and the customer was told
|
|
// "Nincs hiányzó fájl — minden fájl megvan a helyén." — an outage plus a claim about data the
|
|
// restore never looked at. Placed with the other source-side refusals, all of which precede the
|
|
// stop, so the promise "all refusals happen BEFORE the app is stopped" stays true.
|
|
cov := tier2CoverageAt(destBase)
|
|
if !cov.CanRestore() {
|
|
m.logger.Printf("[WARN] [backup] Tier-2 file restore refused for %s: the recorded copy has no restorable subtree (unit_present=%v) — the app was NOT stopped",
|
|
stackName, cov.HasUnit)
|
|
return 0, ErrTier2NoRestorableData
|
|
}
|
|
|
|
copier := m.restoreFilesCopier
|
|
if copier == nil {
|
|
copier = rsyncRestoreMissing
|
|
}
|
|
|
|
// The two v2 subtree merges: destBase/hdd/<rel> ↔ liveNsRoot/<rel>;
|
|
// destBase/userdata/<rel> ↔ liveNsRoot/userdata/<rel>. Each missing-only, additive.
|
|
merges := []struct{ src, dst string }{
|
|
{filepath.Join(destBase, "hdd"), liveNsRoot},
|
|
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Tier-2 file restore for %s: %s (v2) → %s (additive-only)", stackName, destBase, liveNsRoot)
|
|
|
|
// Stop → copy → start → health (the standard restore shape; F17: errors surface, never swallowed).
|
|
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
|
|
m.logger.Printf("[WARN] [backup] could not stop %s before Tier-2 file restore: %v (continuing)", stackName, stopErr)
|
|
}
|
|
start := time.Now()
|
|
var copyErr error
|
|
for _, mg := range merges {
|
|
if _, err := os.Stat(mg.src); err != nil {
|
|
continue // that subtree is absent in this copy (e.g. no userdata legs) — skip
|
|
}
|
|
n, err := copier(mg.src, mg.dst)
|
|
filesRestored += n
|
|
if err != nil {
|
|
copyErr = err
|
|
break
|
|
}
|
|
}
|
|
startErr := m.stackProvider.StartStack(stackName)
|
|
if startErr != nil {
|
|
m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr)
|
|
}
|
|
if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil {
|
|
m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr)
|
|
}
|
|
|
|
if copyErr != nil {
|
|
return filesRestored, fmt.Errorf("fájlmásolás sikertelen: %w", copyErr)
|
|
}
|
|
if startErr != nil {
|
|
return filesRestored, fmt.Errorf("%d fájl visszaállítva, de az alkalmazás újraindítása sikertelen: %w", filesRestored, startErr)
|
|
}
|
|
// Privacy: count + duration only — customer file names never at INFO.
|
|
m.logger.Printf("[INFO] [backup] Tier-2 file restore completed for %s: %d file(s) restored (%s)",
|
|
stackName, filesRestored, time.Since(start).Round(time.Second))
|
|
return filesRestored, nil
|
|
}
|
|
|
|
// rsyncRestoreMissing copies the files MISSING from dst back from src, and nothing else:
|
|
// `rsync -a --ignore-existing` — existing dst files are never overwritten, and (unlike rsyncMirror,
|
|
// which carries --delete for the backup direction) nothing at dst is ever deleted. Returns the
|
|
// number of regular files transferred, counted from --itemize-changes output.
|
|
func rsyncRestoreMissing(src, dst string) (int, error) {
|
|
if err := os.MkdirAll(dst, 0755); err != nil {
|
|
return 0, fmt.Errorf("mkdir %s: %w", dst, err)
|
|
}
|
|
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
|
|
defer cancel()
|
|
// Trailing slashes: copy the CONTENTS of src into dst (same shape as rsyncMirror).
|
|
cmd := exec.CommandContext(ctx, "rsync", "-a", "--ignore-existing", "--itemize-changes",
|
|
strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/")
|
|
out, err := cmd.CombinedOutput()
|
|
if err != nil {
|
|
return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out)))
|
|
}
|
|
return countRestoredFiles(string(out)), nil
|
|
}
|
|
|
|
// countRestoredFiles counts the itemize-changes lines that mark a TRANSFERRED regular file (">f…").
|
|
// Created dirs ("cd…") and symlinks ("cL…") are not counted — the flash reports files. Pure
|
|
// (unit-tested without rsync).
|
|
func countRestoredFiles(itemizedOut string) int {
|
|
n := 0
|
|
for _, line := range strings.Split(itemizedOut, "\n") {
|
|
if strings.HasPrefix(line, ">f") {
|
|
n++
|
|
}
|
|
}
|
|
return n
|
|
}
|