package backup import ( "context" "errors" "fmt" "os" "os/exec" "path/filepath" "strings" "time" ) // Tier-2 in-place file restore (TASK C2, closes drill finding F2): the customer-facing recovery for // class-C data — HDD bind-mount user files under appdata/. Restores MISSING files from the // recorded Tier-2 copy back into the live appdata dir, and touches NOTHING else: // // - a file that exists live is NEVER overwritten (a customer edit after the last Tier-2 run wins); // - a live file absent from the backup is NEVER deleted (that is what rsyncMirror's --delete would // do in this direction — the catastrophic trap this helper exists to avoid); // - only files present in the copy and missing live are copied back (attrs preserved). // // This exactly serves the "I deleted my files" scenario. Corruption / point-in-time rollback stays // with the offbox restore-to-verify + operator paths — deliberately out of scope. // Refusal reasons (customer-readable — they surface verbatim in the flash message). var ( errNoTier2Copy = errors.New("nincs másodlagos fájlmásolat ehhez az alkalmazáshoz") errTier2DriveGone = errors.New("a másodlagos meghajtó nincs csatlakoztatva") errLiveDriveGone = errors.New("az alkalmazás meghajtója nincs csatlakoztatva") errLiveDriveDecommed = errors.New("az alkalmazás meghajtója le van szerelve") // errTier2OldLayout (3b, §7-G2): the recorded copy predates the v2 relpath-mirroring layout (no // marker). Refuse rather than read a flat layout we no longer understand — safe, because tier-2 // restore is missing-file recovery and the live data still exists in that scenario. errTier2OldLayout = errors.New("A 2. mentés régi formátumú — futtass előbb egy új másodlagos mentést.") // ErrTier2NoRestorableData (C9-F1) — this app HAS a Tier-2 copy, but that copy contains no subtree // this restore can read: its data lives entirely in Docker named volumes, which are captured into // recovery-unit/ (db-dumps + volume-dumps) and NEVER read by this path. 43 of the 53 catalog apps // are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action // that does work, instead of taking an outage and reporting "no missing files". ErrTier2NoRestorableData = errors.New("ennek az alkalmazásnak az adatai nem ebből a másolatból állíthatók vissza") // ErrTier2NoUnitInCopy (R-102) — the recorded Tier-2 copy holds no OPENABLE recovery unit: either // recovery-unit/ is absent, or it is a directory without a readable manifest.json. Exported so the // handler can refuse before beginning any op. FAIL CLOSED is the whole point of the second half: // a directory that exists is not a package, and reading a half-copied mirror as if it were one is // how a restore would overwrite live data with nothing. ErrTier2NoUnitInCopy = errors.New("a másodlagos másolatban nincs megnyitható mentési egység ehhez az alkalmazáshoz") ) // Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1 // is about. Computed from the RECORDED copy on disk, never guessed from the catalog, so an app whose // template changed is judged by what its actual copy holds. // // The distinction that matters: Legs are the subtrees RestoreTier2Files reads (hdd/, userdata/); // HasUnit means the copy ALSO holds a full recovery unit — the app's database dumps and named-volume // tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in // which case the restore is a guaranteed no-op no matter how much data was lost. type Tier2Coverage struct { Legs []string // subtrees the FILE restore reads and that exist in the copy: "hdd", "userdata" HasUnit bool // recovery-unit/ present as a DIRECTORY — the disclosure fact, see below // UnitRestorable (R-102) — the mirror is a real PACKAGE, not merely a directory: recovery-unit/ // exists AND carries a manifest.json that parses. This is the gate for the UNIT restore. // // It is a SECOND FIELD and not a widening of HasUnit, and the distinction is load-bearing in both // directions. HasUnit answers "is there captured data this FILE restore is not looking at?" — the // question tier2UnitNotCoveredMsg is appended for, and the honest answer for a half-copied mirror // is still yes. UnitRestorable answers "can the unit restore open this?" — and for that same // half-copied mirror the answer is no. Collapsing them would either silence a true disclosure or // arm a restore over an unopenable package. UnitRestorable bool // CopyLastRun / CopyLastSuccess — WHEN the copy this restore would read was written, so the // surface can name the date before it overwrites anything with it (Scenario E). Filled only by // Tier2RestoreCoverage, which is the path that holds the settings; tier2CoverageAt is a pure // filesystem inspection and leaves them empty. They are STRINGS in the recorded RFC3339 form, // carried verbatim — no formatting decision is taken in this package. // // R-101 applies here exactly as it does on the backup card: CopyLastRun is the ATTEMPT clock and // CopyLastSuccess is the only evidence a copy was actually made. A surface that shows one must // not present it as the other. CopyLastRun string CopyLastSuccess string // UnitPackageDate / UnitLegPreserved (R-403) — WHEN the package in this copy was actually // captured, and whether the newest run PRESERVED it instead of refreshing it. // // They exist because after an R-403 skip, CopyLastRun and CopyLastSuccess stop describing the // package: the run really did succeed and really is from today, and the package in the copy is // from before it. A surface that names the run date as the package date would be trading a data // loss for a comforting lie, which is the failure family this project keeps finding. // // UnitPackageDate is read from the MIRRORED UNIT'S OWN MANIFEST, not from the recorded status, so // it is a fact about the artifact the restore will actually open. "" means UNKNOWN. UnitPackageDate string UnitLegPreserved bool // UnitDataDate (R-476) — the newest ARTIFACT in the mirrored unit: its dumps' mtime, or the // manifest's when nothing is newer. The manifest moves only when the app's DEFINITION changes // (checksum-skip), while the nightly dumps keep their names and their fresh bytes — so on // demo-hp a copy holding a dump written at 00:30Z was dated by a manifest from the day before. // RFC3339 UTC; "" when the unit is not readable. UnitDataDate string } // CanRestore reports whether the FILE restore has any subtree to read at all. // // R-102/R-103: this answers exactly one question and must keep answering only that one. Widening it // to include the unit is R-356 arriving a second time — there, ONE predicate meant both "has this app // a drive?" and "is this app installed?", and it refused 40 running apps for months while telling // their owners to reinstall them somewhere those apps never offer. Two questions, two predicates. func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 } // CanRestoreUnit reports whether the UNIT restore can run from this copy — the second predicate. func (c Tier2Coverage) CanRestoreUnit() bool { return c.UnitRestorable } // tier2UnitDir returns the recovery-unit directory inside a resolved Tier-2 copy. ONE expression of // where Tier-2 puts the mirror; tier2.go writes it at the same relative name ("Unit leg (always)"). func tier2UnitDir(destBase string) string { return filepath.Join(destBase, "recovery-unit") } // tier2UnitIsOpenable reports whether a mirrored unit directory is a PACKAGE and not just a // directory. Fail-closed by construction: the manifest must be present AND parse (readManifest // returns nil for both a missing file and malformed JSON), because an unopenable unit that armed a // restore would stop the app, replay nothing, and rewrite its definition from an empty capture. // // This is the R-358 lesson one tier over — the off-site scratch marker — arriving on Tier-2. func tier2UnitIsOpenable(unitDir string) bool { if fi, err := os.Stat(unitDir); err != nil || !fi.IsDir() { return false } return readManifest(UnitManifestFile(unitDir)) != nil } // tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat/read — no side effects. func tier2CoverageAt(destBase string) Tier2Coverage { var c Tier2Coverage for _, leg := range []string{"hdd", "userdata"} { if fi, err := os.Stat(filepath.Join(destBase, leg)); err == nil && fi.IsDir() { c.Legs = append(c.Legs, leg) } } unitDir := tier2UnitDir(destBase) if fi, err := os.Stat(unitDir); err == nil && fi.IsDir() { c.HasUnit = true } c.UnitRestorable = tier2UnitIsOpenable(unitDir) // R-403: ask the package itself when it was made. Reading the artifact rather than the status // record is what makes this date impossible to overstate. c.UnitPackageDate = unitPackageDate(unitDir) if newest, ok := unitNewestArtifact(unitDir); ok { c.UnitDataDate = newest.UTC().Format(time.RFC3339) } return c } // Tier2RestoreCoverage resolves the app's RECORDED Tier-2 copy and reports what a restore could // return from it. Errors are the same refusals RestoreTier2Files itself would raise, so the caller // can surface them before starting anything — this is what lets the handler refuse without an outage. func (m *Manager) Tier2RestoreCoverage(stackName string) (Tier2Coverage, error) { destBase, err := m.tier2RecordedCopyDir(stackName) if err != nil { return Tier2Coverage{}, err } cov := tier2CoverageAt(destBase) // R-102: carry WHEN the copy was written, so the surface can name the date on an action that // overwrites live data with it. Read from the same recorded config tier2RecordedCopyDir just // resolved the path from, so the date and the directory cannot describe different runs. if m.settings != nil { if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil { cov.CopyLastRun, cov.CopyLastSuccess = cfg.LastRun, cfg.LastSuccess cov.UnitLegPreserved = cfg.UnitLegSkipped } } return cov, nil } // RestoreTier2Unit runs the FULL recovery-unit restore from the app's Tier-2 copy on the SECOND // DRIVE — R-102, and the reason this task exists. // // Tier-2 has mirrored each app's whole recovery unit to // /backups/secondary//recovery-unit/ on every run for months, and no code path read it: // every reader of a unit could only name a path under backups/primary/. So in the exact failure // Tier-2 exists for — the primary drive is lost, and the primary unit with it — the surviving copy // was unopenable by any customer action (07-backup-architecture §6.3, §7.2). // // It is NOT the additive file restore beside it. This one OVERWRITES: named volumes are recreated // from the mirror's tars and the database is replayed from the mirror's dump. The surface must carry // that difference; see tier2UnitConfirm in the web package. // // THE SINGLE-WRITER FLAG IS TAKEN INSIDE RestoreFromRecoveryUnitAt, exactly as on the primary path — // do NOT add an acquireRunning() here. A second acquire would refuse the restore it is guarding. func (m *Manager) RestoreTier2Unit(stackName string) (UnitRestoreResult, error) { destBase, err := m.tier2RecordedCopyDir(stackName) if err != nil { return UnitRestoreResult{}, err } unitDir := tier2UnitDir(destBase) if !tier2UnitIsOpenable(unitDir) { m.logger.Printf("[WARN] [backup] Tier-2 unit restore refused for %s: no openable recovery unit in the recorded copy — the app was NOT stopped", stackName) return UnitRestoreResult{}, ErrTier2NoUnitInCopy } // WARN, not INFO: this is the destructive one of the two Tier-2 restores. The unit directory is a // path and never a secret, and naming it is what makes "the SECONDARY mirror was the source" a // positive observable in the log rather than an absence to be argued from. m.logger.Printf("[WARN] [backup] Tier-2 UNIT restore for %s from the secondary mirror %s — this OVERWRITES live app data", stackName, unitDir) res, restoreErr := m.RestoreFromRecoveryUnitAt(stackName, unitDir) // R-403, the CAUSE half. Refill the primary unit from the mirror we just restored from, INSIDE // this call, before it returns. // // THE TIMING IS THE REQUIREMENT, NOT A DETAIL. On 2026-08-31 the hollow primary manifest was // written TWO SECONDS after a restore of exactly this shape, by the 5-minute `backup-cache` job // (`backup.go` → `captureAllRecoveryUnits`). Any follow-up job, scheduled refresh or goroutine // races that capture and can lose. Doing it here is the only shape that cannot. // // The capture itself is NOT guarded and must not be: a capture that describes an empty drive as // empty is CORRECT. With the primary refilled there is no hollow state left for it to describe, // which is why the fix is here and not there. Guarding the capture would make the manifest lie. m.rehydratePrimaryUnit(stackName, unitDir, restoreErr) return res, restoreErr } // rehydratePrimaryUnit copies a mirrored recovery unit back onto the app's own drive when the primary // unit is ABSENT or HOLLOW — the state a Tier-2 unit restore leaves behind, and the state that armed // R-403's delete on the following night. // // Three refusals, each earned: // - the restore FAILED → write nothing. A package written from a run that did not succeed is worse // than no package: it would look like a backup and describe data that never landed. // - the primary already CARRIES DATA → leave it byte-identical. It may be NEWER than the mirror // (the customer restored while their own drive was fine), and overwriting it with an older copy // is the very move this whole task exists to prevent, pointed the other way. // - anything goes wrong copying → WARN and carry on. The restore itself succeeded; the app is back. // Failing the restore because a convenience copy failed would report a success as a failure. // // It is best-effort by design and says so in the log either way, because an absent log line is not // evidence that it ran. func (m *Manager) rehydratePrimaryUnit(stackName, mirrorUnitDir string, restoreErr error) { if restoreErr != nil { m.logger.Printf("[INFO] [backup] %s: primary unit NOT refilled — the restore itself failed (R-403: a package from a failed run is worse than none)", stackName) return } drivePath := m.GetAppDrivePath(stackName) if drivePath == "" || !filepath.IsAbs(drivePath) { m.logger.Printf("[WARN] [backup] %s: primary unit NOT refilled — cannot resolve the app's drive", stackName) return } primaryUnit := RecoveryUnitPath(m.namespaceRoot(drivePath), stackName) if unitCarriesData(primaryUnit) { m.logger.Printf("[INFO] [backup] %s: primary unit already carries data — left untouched (R-403 never overwrites a richer package with a poorer one)", stackName) return } copier := m.unitRehydrate if copier == nil { copier = rsyncMirror } if err := copier(mirrorUnitDir, primaryUnit); err != nil { m.logger.Printf("[ERROR] [backup] %s: refilling the primary unit from the mirror FAILED: %v — the restore itself SUCCEEDED and the app is running; the local package stays incomplete until the next backup", stackName, err) return } m.logger.Printf("[INFO] [backup] %s: primary unit refilled from the secondary mirror (R-403) — %d volume tar(s), %d database dump(s) now on the app's own drive", stackName, countUnitFiles(UnitVolumeDumpDir(primaryUnit), ".tar"), countUnitFiles(UnitDBDumpDir(primaryUnit), ".sql")) } // countUnitFiles counts files with a suffix in a unit leg directory — for the log line only, so the // refill states WHAT it put back rather than merely that it ran. Never a secret: counts, not names. func countUnitFiles(dir, suffix string) int { entries, err := os.ReadDir(dir) if err != nil { return 0 } n := 0 for _, e := range entries { if !e.IsDir() && strings.HasSuffix(e.Name(), suffix) { n++ } } return n } // Tier2CopyDate returns the date the surface should name for this app's Tier-2 copy, preferring the // last SUCCESS over the last ATTEMPT (R-101: a timestamp that records "we tried" cannot answer "did // it work"), and reports whether the returned value is a proven success. // // It exists so the confirm text and the outcome sentence cannot disagree about which copy is being // restored: one resolver, two readers. func (c Tier2Coverage) Tier2CopyDate() (date string, proven bool) { if c.CopyLastSuccess != "" { return c.CopyLastSuccess, true } return c.CopyLastRun, false } // UnitRestoreDate returns the date of the PACKAGE the unit restore would actually open, and whether // the newest run PRESERVED that package rather than refreshing it. // // R-403. `Tier2CopyDate` answers "when was this copy last written to" and is right for the file // restore, whose legs really were refreshed by that run. It is the WRONG answer for the unit restore // after a preserved leg, because the package is then from before the run that reports success. This // asks the manifest first and falls back to the copy date only when the package cannot say. // // THE SECOND RETURN IS `UnitLegPreserved` AND NOTHING ELSE, and the first draft got this wrong in a // way only the live run caught. It also compared the package's date against the run's and flagged // "older" — but a unit is ALWAYS captured shortly before the run that mirrors it, so that comparison // was true for every healthy app on the box and every one of them rendered the warning. Live on // demo-hp 2026-08-31: bookstack, kimai, opengist and privatebin all had src and dest manifests at // `12:03:49Z` against a run at `12:14:24Z` — perfectly healthy, and all four would have been told // their package was stale. A warning that fires on everything is a warning nobody reads, which costs // the same as the comforting lie it was meant to replace. // // R-476: when the leg was NOT preserved, the package's date is its DATA time — the newest dump in // the copy — never the manifest's, which moves only when the definition changes and so undersold a // fresh copy by a day. A PRESERVED package keeps the manifest date: nothing in it is newer, and the // R-403 rule that a preserved package is never shown as fresh is what this sits under. func (c Tier2Coverage) UnitRestoreDate() (date string, preserved bool) { if c.UnitPackageDate == "" { copyDate, _ := c.Tier2CopyDate() return copyDate, c.UnitLegPreserved } if !c.UnitLegPreserved && c.UnitDataDate != "" && c.UnitDataDate > c.UnitPackageDate { return c.UnitDataDate, false } return c.UnitPackageDate, c.UnitLegPreserved } // tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every // source-side refusal in one place so the pre-flight check and the restore itself cannot drift. func (m *Manager) tier2RecordedCopyDir(stackName string) (string, error) { var destBase string if m.settings != nil { if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil && cfg.LastRun != "" && cfg.DestinationPath != "" { if m.settings.IsDisconnected(cfg.DestinationPath) { return "", errTier2DriveGone } destBase = filepath.Join(cfg.DestinationPath, "backups", "secondary", stackName) } } if destBase == "" { return "", errNoTier2Copy } if _, statErr := os.Stat(destBase); statErr != nil { return "", errNoTier2Copy // recorded but the copy dir is gone — same honest refusal } // §7-G2 marker gate: a pre-v2 (flat) copy has no marker → refuse rather than read a layout we no // longer understand (live data still exists for missing-file recovery). if _, mErr := os.Stat(filepath.Join(destBase, tier2LayoutMarker)); mErr != nil { return "", errTier2OldLayout } return destBase, nil } // RestoreTier2Files restores the app's MISSING user files in place from its recorded Tier-2 copy // (additive-only; see the package comment above). Returns how many regular files were copied back. // // The source is the RECORDED Tier-2 destination (settings.CrossDriveBackup.DestinationPath) — never // a fresh selectTier2Target, which could re-pick a different (empty) drive and "restore" nothing. // All refusals happen BEFORE the app is stopped. Stop-first is the locked consistency policy: the // app must not be reorganizing its data dir mid-copy. func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err error) { if m.stackProvider == nil { return 0, fmt.Errorf("stack provider not configured") } if err := m.acquireRunning(); err != nil { return 0, err // shares the backup/restore single-flight — must not race a running backup } defer m.releaseRunning() // Live side: the app's drive must be present and in service. drive := m.GetAppDrivePath(stackName) if drive == "" || !filepath.IsAbs(drive) { return 0, fmt.Errorf("cannot determine drive path for %s", stackName) } if m.settings != nil { if m.settings.IsDisconnected(drive) { return 0, fmt.Errorf("%w (%s)", errLiveDriveGone, drive) } if m.settings.IsDecommissioned(drive) { return 0, fmt.Errorf("%w (%s)", errLiveDriveDecommed, drive) } } // v2 relpath-mirroring: liveNsRoot == the app's HDD_PATH (Model A). The dest hdd/ and userdata/ // subtrees mirror the live relpath structure exactly, so restore is two whole-subtree merges (N>1 // dirs + nested binds handled natively — no per-appdata-dir resolution, no N>1 refusal). liveNsRoot := m.namespaceRoot(drive) // Source side: the RECORDED Tier-2 copy must exist, its drive connected, and it must be v2. destBase, err := m.tier2RecordedCopyDir(stackName) if err != nil { return 0, err } // C9-F1: refuse BEFORE the app is stopped if this copy holds nothing this path can read. Without // this the app was stopped, zero files were copied, it was restarted, and the customer was told // "Nincs hiányzó fájl — minden fájl megvan a helyén." — an outage plus a claim about data the // restore never looked at. Placed with the other source-side refusals, all of which precede the // stop, so the promise "all refusals happen BEFORE the app is stopped" stays true. cov := tier2CoverageAt(destBase) if !cov.CanRestore() { m.logger.Printf("[WARN] [backup] Tier-2 file restore refused for %s: the recorded copy has no restorable subtree (unit_present=%v) — the app was NOT stopped", stackName, cov.HasUnit) return 0, ErrTier2NoRestorableData } copier := m.restoreFilesCopier if copier == nil { copier = rsyncRestoreMissing } // The two v2 subtree merges: destBase/hdd/ ↔ liveNsRoot/; // destBase/userdata/ ↔ liveNsRoot/userdata/. Each missing-only, additive. merges := []struct{ src, dst string }{ {filepath.Join(destBase, "hdd"), liveNsRoot}, {filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")}, } m.logger.Printf("[INFO] [backup] Tier-2 file restore for %s: %s (v2) → %s (additive-only)", stackName, destBase, liveNsRoot) // Stop → copy → start → health (the standard restore shape; F17: errors surface, never swallowed). if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil { m.logger.Printf("[WARN] [backup] could not stop %s before Tier-2 file restore: %v (continuing)", stackName, stopErr) } start := time.Now() var copyErr error for _, mg := range merges { if _, err := os.Stat(mg.src); err != nil { continue // that subtree is absent in this copy (e.g. no userdata legs) — skip } n, err := copier(mg.src, mg.dst) filesRestored += n if err != nil { copyErr = err break } } startErr := m.stackProvider.StartStack(stackName) if startErr != nil { m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr) } if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil { m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr) } if copyErr != nil { return filesRestored, fmt.Errorf("fájlmásolás sikertelen: %w", copyErr) } if startErr != nil { return filesRestored, fmt.Errorf("%d fájl visszaállítva, de az alkalmazás újraindítása sikertelen: %w", filesRestored, startErr) } // Privacy: count + duration only — customer file names never at INFO. m.logger.Printf("[INFO] [backup] Tier-2 file restore completed for %s: %d file(s) restored (%s)", stackName, filesRestored, time.Since(start).Round(time.Second)) return filesRestored, nil } // rsyncRestoreMissing copies the files MISSING from dst back from src, and nothing else: // `rsync -a --ignore-existing` — existing dst files are never overwritten, and (unlike rsyncMirror, // which carries --delete for the backup direction) nothing at dst is ever deleted. Returns the // number of regular files transferred, counted from --itemize-changes output. func rsyncRestoreMissing(src, dst string) (int, error) { if err := os.MkdirAll(dst, 0755); err != nil { return 0, fmt.Errorf("mkdir %s: %w", dst, err) } ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute) defer cancel() // Trailing slashes: copy the CONTENTS of src into dst (same shape as rsyncMirror). cmd := exec.CommandContext(ctx, "rsync", "-a", "--ignore-existing", "--itemize-changes", strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/") out, err := cmd.CombinedOutput() if err != nil { return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out))) } return countRestoredFiles(string(out)), nil } // countRestoredFiles counts the itemize-changes lines that mark a TRANSFERRED regular file (">f…"). // Created dirs ("cd…") and symlinks ("cL…") are not counted — the flash reports files. Pure // (unit-tested without rsync). func countRestoredFiles(itemizedOut string) int { n := 0 for _, line := range strings.Split(itemizedOut, "\n") { if strings.HasPrefix(line, ">f") { n++ } } return n }