5429d651ee
gates / gates (push) Successful in 11s
UnitRestoreDate also compared the package's date against the run's and flagged 'older'. A recovery unit is ALWAYS captured shortly before the run that mirrors it, so that comparison is true for every healthy app. Measured on demo-hp: bookstack, kimai, opengist and privatebin all had src and dest manifests at 12:03:49Z against a run at 12:14:24Z - perfectly healthy, and all four would have been told their package was stale. A warning that fires on everything is a warning nobody reads, which costs the same as the comforting lie it was meant to replace. The second return is now UnitLegPreserved and nothing else. TestR403_AHealthyAppIsNeverCalledStale pins it; red-proof: reinstate the comparison -> it fails.
470 lines
24 KiB
Go
470 lines
24 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// Tier-2 in-place file restore (TASK C2, closes drill finding F2): the customer-facing recovery for
|
|
// class-C data — HDD bind-mount user files under appdata/<stack>. Restores MISSING files from the
|
|
// recorded Tier-2 copy back into the live appdata dir, and touches NOTHING else:
|
|
//
|
|
// - a file that exists live is NEVER overwritten (a customer edit after the last Tier-2 run wins);
|
|
// - a live file absent from the backup is NEVER deleted (that is what rsyncMirror's --delete would
|
|
// do in this direction — the catastrophic trap this helper exists to avoid);
|
|
// - only files present in the copy and missing live are copied back (attrs preserved).
|
|
//
|
|
// This exactly serves the "I deleted my files" scenario. Corruption / point-in-time rollback stays
|
|
// with the offbox restore-to-verify + operator paths — deliberately out of scope.
|
|
|
|
// Refusal reasons (customer-readable — they surface verbatim in the flash message).
|
|
var (
|
|
errNoTier2Copy = errors.New("nincs másodlagos fájlmásolat ehhez az alkalmazáshoz")
|
|
errTier2DriveGone = errors.New("a másodlagos meghajtó nincs csatlakoztatva")
|
|
errLiveDriveGone = errors.New("az alkalmazás meghajtója nincs csatlakoztatva")
|
|
errLiveDriveDecommed = errors.New("az alkalmazás meghajtója le van szerelve")
|
|
// errTier2OldLayout (3b, §7-G2): the recorded copy predates the v2 relpath-mirroring layout (no
|
|
// marker). Refuse rather than read a flat layout we no longer understand — safe, because tier-2
|
|
// restore is missing-file recovery and the live data still exists in that scenario.
|
|
errTier2OldLayout = errors.New("A 2. mentés régi formátumú — futtass előbb egy új másodlagos mentést.")
|
|
// ErrTier2NoRestorableData (C9-F1) — this app HAS a Tier-2 copy, but that copy contains no subtree
|
|
// this restore can read: its data lives entirely in Docker named volumes, which are captured into
|
|
// recovery-unit/ (db-dumps + volume-dumps) and NEVER read by this path. 43 of the 53 catalog apps
|
|
// are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action
|
|
// that does work, instead of taking an outage and reporting "no missing files".
|
|
ErrTier2NoRestorableData = errors.New("ennek az alkalmazásnak az adatai nem ebből a másolatból állíthatók vissza")
|
|
// ErrTier2NoUnitInCopy (R-102) — the recorded Tier-2 copy holds no OPENABLE recovery unit: either
|
|
// recovery-unit/ is absent, or it is a directory without a readable manifest.json. Exported so the
|
|
// handler can refuse before beginning any op. FAIL CLOSED is the whole point of the second half:
|
|
// a directory that exists is not a package, and reading a half-copied mirror as if it were one is
|
|
// how a restore would overwrite live data with nothing.
|
|
ErrTier2NoUnitInCopy = errors.New("a másodlagos másolatban nincs megnyitható mentési egység ehhez az alkalmazáshoz")
|
|
)
|
|
|
|
// Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1
|
|
// is about. Computed from the RECORDED copy on disk, never guessed from the catalog, so an app whose
|
|
// template changed is judged by what its actual copy holds.
|
|
//
|
|
// The distinction that matters: Legs are the subtrees RestoreTier2Files reads (hdd/, userdata/);
|
|
// HasUnit means the copy ALSO holds a full recovery unit — the app's database dumps and named-volume
|
|
// tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in
|
|
// which case the restore is a guaranteed no-op no matter how much data was lost.
|
|
type Tier2Coverage struct {
|
|
Legs []string // subtrees the FILE restore reads and that exist in the copy: "hdd", "userdata"
|
|
HasUnit bool // recovery-unit/ present as a DIRECTORY — the disclosure fact, see below
|
|
|
|
// UnitRestorable (R-102) — the mirror is a real PACKAGE, not merely a directory: recovery-unit/
|
|
// exists AND carries a manifest.json that parses. This is the gate for the UNIT restore.
|
|
//
|
|
// It is a SECOND FIELD and not a widening of HasUnit, and the distinction is load-bearing in both
|
|
// directions. HasUnit answers "is there captured data this FILE restore is not looking at?" — the
|
|
// question tier2UnitNotCoveredMsg is appended for, and the honest answer for a half-copied mirror
|
|
// is still yes. UnitRestorable answers "can the unit restore open this?" — and for that same
|
|
// half-copied mirror the answer is no. Collapsing them would either silence a true disclosure or
|
|
// arm a restore over an unopenable package.
|
|
UnitRestorable bool
|
|
|
|
// CopyLastRun / CopyLastSuccess — WHEN the copy this restore would read was written, so the
|
|
// surface can name the date before it overwrites anything with it (Scenario E). Filled only by
|
|
// Tier2RestoreCoverage, which is the path that holds the settings; tier2CoverageAt is a pure
|
|
// filesystem inspection and leaves them empty. They are STRINGS in the recorded RFC3339 form,
|
|
// carried verbatim — no formatting decision is taken in this package.
|
|
//
|
|
// R-101 applies here exactly as it does on the backup card: CopyLastRun is the ATTEMPT clock and
|
|
// CopyLastSuccess is the only evidence a copy was actually made. A surface that shows one must
|
|
// not present it as the other.
|
|
CopyLastRun string
|
|
CopyLastSuccess string
|
|
|
|
// UnitPackageDate / UnitLegPreserved (R-403) — WHEN the package in this copy was actually
|
|
// captured, and whether the newest run PRESERVED it instead of refreshing it.
|
|
//
|
|
// They exist because after an R-403 skip, CopyLastRun and CopyLastSuccess stop describing the
|
|
// package: the run really did succeed and really is from today, and the package in the copy is
|
|
// from before it. A surface that names the run date as the package date would be trading a data
|
|
// loss for a comforting lie, which is the failure family this project keeps finding.
|
|
//
|
|
// UnitPackageDate is read from the MIRRORED UNIT'S OWN MANIFEST, not from the recorded status, so
|
|
// it is a fact about the artifact the restore will actually open. "" means UNKNOWN.
|
|
UnitPackageDate string
|
|
UnitLegPreserved bool
|
|
}
|
|
|
|
// CanRestore reports whether the FILE restore has any subtree to read at all.
|
|
//
|
|
// R-102/R-103: this answers exactly one question and must keep answering only that one. Widening it
|
|
// to include the unit is R-356 arriving a second time — there, ONE predicate meant both "has this app
|
|
// a drive?" and "is this app installed?", and it refused 40 running apps for months while telling
|
|
// their owners to reinstall them somewhere those apps never offer. Two questions, two predicates.
|
|
func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 }
|
|
|
|
// CanRestoreUnit reports whether the UNIT restore can run from this copy — the second predicate.
|
|
func (c Tier2Coverage) CanRestoreUnit() bool { return c.UnitRestorable }
|
|
|
|
// tier2UnitDir returns the recovery-unit directory inside a resolved Tier-2 copy. ONE expression of
|
|
// where Tier-2 puts the mirror; tier2.go writes it at the same relative name ("Unit leg (always)").
|
|
func tier2UnitDir(destBase string) string {
|
|
return filepath.Join(destBase, "recovery-unit")
|
|
}
|
|
|
|
// tier2UnitIsOpenable reports whether a mirrored unit directory is a PACKAGE and not just a
|
|
// directory. Fail-closed by construction: the manifest must be present AND parse (readManifest
|
|
// returns nil for both a missing file and malformed JSON), because an unopenable unit that armed a
|
|
// restore would stop the app, replay nothing, and rewrite its definition from an empty capture.
|
|
//
|
|
// This is the R-358 lesson one tier over — the off-site scratch marker — arriving on Tier-2.
|
|
func tier2UnitIsOpenable(unitDir string) bool {
|
|
if fi, err := os.Stat(unitDir); err != nil || !fi.IsDir() {
|
|
return false
|
|
}
|
|
return readManifest(UnitManifestFile(unitDir)) != nil
|
|
}
|
|
|
|
// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat/read — no side effects.
|
|
func tier2CoverageAt(destBase string) Tier2Coverage {
|
|
var c Tier2Coverage
|
|
for _, leg := range []string{"hdd", "userdata"} {
|
|
if fi, err := os.Stat(filepath.Join(destBase, leg)); err == nil && fi.IsDir() {
|
|
c.Legs = append(c.Legs, leg)
|
|
}
|
|
}
|
|
unitDir := tier2UnitDir(destBase)
|
|
if fi, err := os.Stat(unitDir); err == nil && fi.IsDir() {
|
|
c.HasUnit = true
|
|
}
|
|
c.UnitRestorable = tier2UnitIsOpenable(unitDir)
|
|
// R-403: ask the package itself when it was made. Reading the artifact rather than the status
|
|
// record is what makes this date impossible to overstate.
|
|
c.UnitPackageDate = unitPackageDate(unitDir)
|
|
return c
|
|
}
|
|
|
|
// Tier2RestoreCoverage resolves the app's RECORDED Tier-2 copy and reports what a restore could
|
|
// return from it. Errors are the same refusals RestoreTier2Files itself would raise, so the caller
|
|
// can surface them before starting anything — this is what lets the handler refuse without an outage.
|
|
func (m *Manager) Tier2RestoreCoverage(stackName string) (Tier2Coverage, error) {
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return Tier2Coverage{}, err
|
|
}
|
|
cov := tier2CoverageAt(destBase)
|
|
// R-102: carry WHEN the copy was written, so the surface can name the date on an action that
|
|
// overwrites live data with it. Read from the same recorded config tier2RecordedCopyDir just
|
|
// resolved the path from, so the date and the directory cannot describe different runs.
|
|
if m.settings != nil {
|
|
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil {
|
|
cov.CopyLastRun, cov.CopyLastSuccess = cfg.LastRun, cfg.LastSuccess
|
|
cov.UnitLegPreserved = cfg.UnitLegSkipped
|
|
}
|
|
}
|
|
return cov, nil
|
|
}
|
|
|
|
// RestoreTier2Unit runs the FULL recovery-unit restore from the app's Tier-2 copy on the SECOND
|
|
// DRIVE — R-102, and the reason this task exists.
|
|
//
|
|
// Tier-2 has mirrored each app's whole recovery unit to
|
|
// <dest>/backups/secondary/<app>/recovery-unit/ on every run for months, and no code path read it:
|
|
// every reader of a unit could only name a path under backups/primary/. So in the exact failure
|
|
// Tier-2 exists for — the primary drive is lost, and the primary unit with it — the surviving copy
|
|
// was unopenable by any customer action (07-backup-architecture §6.3, §7.2).
|
|
//
|
|
// It is NOT the additive file restore beside it. This one OVERWRITES: named volumes are recreated
|
|
// from the mirror's tars and the database is replayed from the mirror's dump. The surface must carry
|
|
// that difference; see tier2UnitConfirm in the web package.
|
|
//
|
|
// THE SINGLE-WRITER FLAG IS TAKEN INSIDE RestoreFromRecoveryUnitAt, exactly as on the primary path —
|
|
// do NOT add an acquireRunning() here. A second acquire would refuse the restore it is guarding.
|
|
func (m *Manager) RestoreTier2Unit(stackName string) (UnitRestoreResult, error) {
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return UnitRestoreResult{}, err
|
|
}
|
|
unitDir := tier2UnitDir(destBase)
|
|
if !tier2UnitIsOpenable(unitDir) {
|
|
m.logger.Printf("[WARN] [backup] Tier-2 unit restore refused for %s: no openable recovery unit in the recorded copy — the app was NOT stopped", stackName)
|
|
return UnitRestoreResult{}, ErrTier2NoUnitInCopy
|
|
}
|
|
// WARN, not INFO: this is the destructive one of the two Tier-2 restores. The unit directory is a
|
|
// path and never a secret, and naming it is what makes "the SECONDARY mirror was the source" a
|
|
// positive observable in the log rather than an absence to be argued from.
|
|
m.logger.Printf("[WARN] [backup] Tier-2 UNIT restore for %s from the secondary mirror %s — this OVERWRITES live app data", stackName, unitDir)
|
|
res, restoreErr := m.RestoreFromRecoveryUnitAt(stackName, unitDir)
|
|
|
|
// R-403, the CAUSE half. Refill the primary unit from the mirror we just restored from, INSIDE
|
|
// this call, before it returns.
|
|
//
|
|
// THE TIMING IS THE REQUIREMENT, NOT A DETAIL. On 2026-08-31 the hollow primary manifest was
|
|
// written TWO SECONDS after a restore of exactly this shape, by the 5-minute `backup-cache` job
|
|
// (`backup.go` → `captureAllRecoveryUnits`). Any follow-up job, scheduled refresh or goroutine
|
|
// races that capture and can lose. Doing it here is the only shape that cannot.
|
|
//
|
|
// The capture itself is NOT guarded and must not be: a capture that describes an empty drive as
|
|
// empty is CORRECT. With the primary refilled there is no hollow state left for it to describe,
|
|
// which is why the fix is here and not there. Guarding the capture would make the manifest lie.
|
|
m.rehydratePrimaryUnit(stackName, unitDir, restoreErr)
|
|
return res, restoreErr
|
|
}
|
|
|
|
// rehydratePrimaryUnit copies a mirrored recovery unit back onto the app's own drive when the primary
|
|
// unit is ABSENT or HOLLOW — the state a Tier-2 unit restore leaves behind, and the state that armed
|
|
// R-403's delete on the following night.
|
|
//
|
|
// Three refusals, each earned:
|
|
// - the restore FAILED → write nothing. A package written from a run that did not succeed is worse
|
|
// than no package: it would look like a backup and describe data that never landed.
|
|
// - the primary already CARRIES DATA → leave it byte-identical. It may be NEWER than the mirror
|
|
// (the customer restored while their own drive was fine), and overwriting it with an older copy
|
|
// is the very move this whole task exists to prevent, pointed the other way.
|
|
// - anything goes wrong copying → WARN and carry on. The restore itself succeeded; the app is back.
|
|
// Failing the restore because a convenience copy failed would report a success as a failure.
|
|
//
|
|
// It is best-effort by design and says so in the log either way, because an absent log line is not
|
|
// evidence that it ran.
|
|
func (m *Manager) rehydratePrimaryUnit(stackName, mirrorUnitDir string, restoreErr error) {
|
|
if restoreErr != nil {
|
|
m.logger.Printf("[INFO] [backup] %s: primary unit NOT refilled — the restore itself failed (R-403: a package from a failed run is worse than none)", stackName)
|
|
return
|
|
}
|
|
drivePath := m.GetAppDrivePath(stackName)
|
|
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
|
m.logger.Printf("[WARN] [backup] %s: primary unit NOT refilled — cannot resolve the app's drive", stackName)
|
|
return
|
|
}
|
|
primaryUnit := RecoveryUnitPath(m.namespaceRoot(drivePath), stackName)
|
|
if unitCarriesData(primaryUnit) {
|
|
m.logger.Printf("[INFO] [backup] %s: primary unit already carries data — left untouched (R-403 never overwrites a richer package with a poorer one)", stackName)
|
|
return
|
|
}
|
|
copier := m.unitRehydrate
|
|
if copier == nil {
|
|
copier = rsyncMirror
|
|
}
|
|
if err := copier(mirrorUnitDir, primaryUnit); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] %s: refilling the primary unit from the mirror FAILED: %v — the restore itself SUCCEEDED and the app is running; the local package stays incomplete until the next backup", stackName, err)
|
|
return
|
|
}
|
|
m.logger.Printf("[INFO] [backup] %s: primary unit refilled from the secondary mirror (R-403) — %d volume tar(s), %d database dump(s) now on the app's own drive",
|
|
stackName, countUnitFiles(UnitVolumeDumpDir(primaryUnit), ".tar"), countUnitFiles(UnitDBDumpDir(primaryUnit), ".sql"))
|
|
}
|
|
|
|
// countUnitFiles counts files with a suffix in a unit leg directory — for the log line only, so the
|
|
// refill states WHAT it put back rather than merely that it ran. Never a secret: counts, not names.
|
|
func countUnitFiles(dir, suffix string) int {
|
|
entries, err := os.ReadDir(dir)
|
|
if err != nil {
|
|
return 0
|
|
}
|
|
n := 0
|
|
for _, e := range entries {
|
|
if !e.IsDir() && strings.HasSuffix(e.Name(), suffix) {
|
|
n++
|
|
}
|
|
}
|
|
return n
|
|
}
|
|
|
|
// Tier2CopyDate returns the date the surface should name for this app's Tier-2 copy, preferring the
|
|
// last SUCCESS over the last ATTEMPT (R-101: a timestamp that records "we tried" cannot answer "did
|
|
// it work"), and reports whether the returned value is a proven success.
|
|
//
|
|
// It exists so the confirm text and the outcome sentence cannot disagree about which copy is being
|
|
// restored: one resolver, two readers.
|
|
func (c Tier2Coverage) Tier2CopyDate() (date string, proven bool) {
|
|
if c.CopyLastSuccess != "" {
|
|
return c.CopyLastSuccess, true
|
|
}
|
|
return c.CopyLastRun, false
|
|
}
|
|
|
|
// UnitRestoreDate returns the date of the PACKAGE the unit restore would actually open, and whether
|
|
// the newest run PRESERVED that package rather than refreshing it.
|
|
//
|
|
// R-403. `Tier2CopyDate` answers "when was this copy last written to" and is right for the file
|
|
// restore, whose legs really were refreshed by that run. It is the WRONG answer for the unit restore
|
|
// after a preserved leg, because the package is then from before the run that reports success. This
|
|
// asks the manifest first and falls back to the copy date only when the package cannot say.
|
|
//
|
|
// THE SECOND RETURN IS `UnitLegPreserved` AND NOTHING ELSE, and the first draft got this wrong in a
|
|
// way only the live run caught. It also compared the package's date against the run's and flagged
|
|
// "older" — but a unit is ALWAYS captured shortly before the run that mirrors it, so that comparison
|
|
// was true for every healthy app on the box and every one of them rendered the warning. Live on
|
|
// demo-hp 2026-08-31: bookstack, kimai, opengist and privatebin all had src and dest manifests at
|
|
// `12:03:49Z` against a run at `12:14:24Z` — perfectly healthy, and all four would have been told
|
|
// their package was stale. A warning that fires on everything is a warning nobody reads, which costs
|
|
// the same as the comforting lie it was meant to replace.
|
|
func (c Tier2Coverage) UnitRestoreDate() (date string, preserved bool) {
|
|
if c.UnitPackageDate == "" {
|
|
copyDate, _ := c.Tier2CopyDate()
|
|
return copyDate, c.UnitLegPreserved
|
|
}
|
|
return c.UnitPackageDate, c.UnitLegPreserved
|
|
}
|
|
|
|
// tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every
|
|
// source-side refusal in one place so the pre-flight check and the restore itself cannot drift.
|
|
func (m *Manager) tier2RecordedCopyDir(stackName string) (string, error) {
|
|
var destBase string
|
|
if m.settings != nil {
|
|
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil && cfg.LastRun != "" && cfg.DestinationPath != "" {
|
|
if m.settings.IsDisconnected(cfg.DestinationPath) {
|
|
return "", errTier2DriveGone
|
|
}
|
|
destBase = filepath.Join(cfg.DestinationPath, "backups", "secondary", stackName)
|
|
}
|
|
}
|
|
if destBase == "" {
|
|
return "", errNoTier2Copy
|
|
}
|
|
if _, statErr := os.Stat(destBase); statErr != nil {
|
|
return "", errNoTier2Copy // recorded but the copy dir is gone — same honest refusal
|
|
}
|
|
// §7-G2 marker gate: a pre-v2 (flat) copy has no marker → refuse rather than read a layout we no
|
|
// longer understand (live data still exists for missing-file recovery).
|
|
if _, mErr := os.Stat(filepath.Join(destBase, tier2LayoutMarker)); mErr != nil {
|
|
return "", errTier2OldLayout
|
|
}
|
|
return destBase, nil
|
|
}
|
|
|
|
// RestoreTier2Files restores the app's MISSING user files in place from its recorded Tier-2 copy
|
|
// (additive-only; see the package comment above). Returns how many regular files were copied back.
|
|
//
|
|
// The source is the RECORDED Tier-2 destination (settings.CrossDriveBackup.DestinationPath) — never
|
|
// a fresh selectTier2Target, which could re-pick a different (empty) drive and "restore" nothing.
|
|
// All refusals happen BEFORE the app is stopped. Stop-first is the locked consistency policy: the
|
|
// app must not be reorganizing its data dir mid-copy.
|
|
func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err error) {
|
|
if m.stackProvider == nil {
|
|
return 0, fmt.Errorf("stack provider not configured")
|
|
}
|
|
if err := m.acquireRunning(); err != nil {
|
|
return 0, err // shares the backup/restore single-flight — must not race a running backup
|
|
}
|
|
defer m.releaseRunning()
|
|
|
|
// Live side: the app's drive must be present and in service.
|
|
drive := m.GetAppDrivePath(stackName)
|
|
if drive == "" || !filepath.IsAbs(drive) {
|
|
return 0, fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
if m.settings != nil {
|
|
if m.settings.IsDisconnected(drive) {
|
|
return 0, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
|
|
}
|
|
if m.settings.IsDecommissioned(drive) {
|
|
return 0, fmt.Errorf("%w (%s)", errLiveDriveDecommed, drive)
|
|
}
|
|
}
|
|
// v2 relpath-mirroring: liveNsRoot == the app's HDD_PATH (Model A). The dest hdd/ and userdata/
|
|
// subtrees mirror the live relpath structure exactly, so restore is two whole-subtree merges (N>1
|
|
// dirs + nested binds handled natively — no per-appdata-dir resolution, no N>1 refusal).
|
|
liveNsRoot := m.namespaceRoot(drive)
|
|
|
|
// Source side: the RECORDED Tier-2 copy must exist, its drive connected, and it must be v2.
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
|
|
// C9-F1: refuse BEFORE the app is stopped if this copy holds nothing this path can read. Without
|
|
// this the app was stopped, zero files were copied, it was restarted, and the customer was told
|
|
// "Nincs hiányzó fájl — minden fájl megvan a helyén." — an outage plus a claim about data the
|
|
// restore never looked at. Placed with the other source-side refusals, all of which precede the
|
|
// stop, so the promise "all refusals happen BEFORE the app is stopped" stays true.
|
|
cov := tier2CoverageAt(destBase)
|
|
if !cov.CanRestore() {
|
|
m.logger.Printf("[WARN] [backup] Tier-2 file restore refused for %s: the recorded copy has no restorable subtree (unit_present=%v) — the app was NOT stopped",
|
|
stackName, cov.HasUnit)
|
|
return 0, ErrTier2NoRestorableData
|
|
}
|
|
|
|
copier := m.restoreFilesCopier
|
|
if copier == nil {
|
|
copier = rsyncRestoreMissing
|
|
}
|
|
|
|
// The two v2 subtree merges: destBase/hdd/<rel> ↔ liveNsRoot/<rel>;
|
|
// destBase/userdata/<rel> ↔ liveNsRoot/userdata/<rel>. Each missing-only, additive.
|
|
merges := []struct{ src, dst string }{
|
|
{filepath.Join(destBase, "hdd"), liveNsRoot},
|
|
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
|
|
}
|
|
m.logger.Printf("[INFO] [backup] Tier-2 file restore for %s: %s (v2) → %s (additive-only)", stackName, destBase, liveNsRoot)
|
|
|
|
// Stop → copy → start → health (the standard restore shape; F17: errors surface, never swallowed).
|
|
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
|
|
m.logger.Printf("[WARN] [backup] could not stop %s before Tier-2 file restore: %v (continuing)", stackName, stopErr)
|
|
}
|
|
start := time.Now()
|
|
var copyErr error
|
|
for _, mg := range merges {
|
|
if _, err := os.Stat(mg.src); err != nil {
|
|
continue // that subtree is absent in this copy (e.g. no userdata legs) — skip
|
|
}
|
|
n, err := copier(mg.src, mg.dst)
|
|
filesRestored += n
|
|
if err != nil {
|
|
copyErr = err
|
|
break
|
|
}
|
|
}
|
|
startErr := m.stackProvider.StartStack(stackName)
|
|
if startErr != nil {
|
|
m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr)
|
|
}
|
|
if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil {
|
|
m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr)
|
|
}
|
|
|
|
if copyErr != nil {
|
|
return filesRestored, fmt.Errorf("fájlmásolás sikertelen: %w", copyErr)
|
|
}
|
|
if startErr != nil {
|
|
return filesRestored, fmt.Errorf("%d fájl visszaállítva, de az alkalmazás újraindítása sikertelen: %w", filesRestored, startErr)
|
|
}
|
|
// Privacy: count + duration only — customer file names never at INFO.
|
|
m.logger.Printf("[INFO] [backup] Tier-2 file restore completed for %s: %d file(s) restored (%s)",
|
|
stackName, filesRestored, time.Since(start).Round(time.Second))
|
|
return filesRestored, nil
|
|
}
|
|
|
|
// rsyncRestoreMissing copies the files MISSING from dst back from src, and nothing else:
|
|
// `rsync -a --ignore-existing` — existing dst files are never overwritten, and (unlike rsyncMirror,
|
|
// which carries --delete for the backup direction) nothing at dst is ever deleted. Returns the
|
|
// number of regular files transferred, counted from --itemize-changes output.
|
|
func rsyncRestoreMissing(src, dst string) (int, error) {
|
|
if err := os.MkdirAll(dst, 0755); err != nil {
|
|
return 0, fmt.Errorf("mkdir %s: %w", dst, err)
|
|
}
|
|
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
|
|
defer cancel()
|
|
// Trailing slashes: copy the CONTENTS of src into dst (same shape as rsyncMirror).
|
|
cmd := exec.CommandContext(ctx, "rsync", "-a", "--ignore-existing", "--itemize-changes",
|
|
strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/")
|
|
out, err := cmd.CombinedOutput()
|
|
if err != nil {
|
|
return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out)))
|
|
}
|
|
return countRestoredFiles(string(out)), nil
|
|
}
|
|
|
|
// countRestoredFiles counts the itemize-changes lines that mark a TRANSFERRED regular file (">f…").
|
|
// Created dirs ("cd…") and symlinks ("cL…") are not counted — the flash reports files. Pure
|
|
// (unit-tested without rsync).
|
|
func countRestoredFiles(itemizedOut string) int {
|
|
n := 0
|
|
for _, line := range strings.Split(itemizedOut, "\n") {
|
|
if strings.HasPrefix(line, ">f") {
|
|
n++
|
|
}
|
|
}
|
|
return n
|
|
}
|