Files
felhom-controller/controller/internal/backup/tier2_restore.go
T
admin 7c05b59708
gates / gates (push) Successful in 24s
v0.253.0 — errors carry the key of the sentence they are (R-557 slice 2 release B)
179 Hungarian sentences were built deep inside a package with fmt.Errorf and printed by
whoever caught them: too late to translate where they are shown, too early where they are
made. Every one now carries its key across that gap. ZERO Hungarian error literals remain.

util.MsgError does three things at once, each earned:
  - Error() is the Hungarian, byte for byte, so every un-converted printer is unchanged;
  - errors.Is answers for the kind AND for a wrapped cause (KindErrorf dropped the cause);
  - an error ARGUMENT renders recursively, so "formázás sikertelen: %w" translates whole.
A foreign error — restic, docker, ssh, the stdlib — prints verbatim. It is not ours.

76 display sites go through errText, and TestNoErrErrorInPageOutput convicts any that do
not. memoryVerdict returns an error rather than a sentence, so the deploy's 409 and the
household's language come from one value; UpdateRefusal gained a Cause to carry it.

Plurals, one rule, stated once: a key with .one/.other takes its COUNT first. Not a
per-call-site flag — the producer somebody forgot would read "3 app is not running". The
guard caught a real key collision (alert.deadapp.one) the day the rule landed.

TWO DEFECTS FOUND IN MY OWN TOOLING, recorded rather than quietly fixed. The bulk converter
silently dropped multi-line concatenations, damaging 7 producers — and the parity gate could
not see it, because every surviving fragment WAS a real base literal while the CALL had lost
text; two behaviour tests caught it. And the counting script was case-sensitive, so it said
"0 left" while five remained.

MinAgent: 0.131.0 (unchanged). No hub release needed.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-18 11:44:30 +02:00

487 lines
25 KiB
Go

package backup
import (
"context"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"os"
"os/exec"
"path/filepath"
"strings"
"time"
)
// Tier-2 in-place file restore (TASK C2, closes drill finding F2): the customer-facing recovery for
// class-C data — HDD bind-mount user files under appdata/<stack>. Restores MISSING files from the
// recorded Tier-2 copy back into the live appdata dir, and touches NOTHING else:
//
// - a file that exists live is NEVER overwritten (a customer edit after the last Tier-2 run wins);
// - a live file absent from the backup is NEVER deleted (that is what rsyncMirror's --delete would
// do in this direction — the catastrophic trap this helper exists to avoid);
// - only files present in the copy and missing live are copied back (attrs preserved).
//
// This exactly serves the "I deleted my files" scenario. Corruption / point-in-time rollback stays
// with the offbox restore-to-verify + operator paths — deliberately out of scope.
// Refusal reasons (customer-readable — they surface verbatim in the flash message).
var (
errNoTier2Copy = util.MsgError("err.backup.nincs_masodlagos_fajlmasolat_ehhez_az_alkalmazashoz")
errTier2DriveGone = util.MsgError("err.backup.a_masodlagos_meghajto_nincs_csatlakoztatva")
errLiveDriveGone = util.MsgError("err.backup.az_alkalmazas_meghajtoja_nincs_csatlakoztatva")
errLiveDriveDecommed = util.MsgError("err.backup.az_alkalmazas_meghajtoja_le_van_szerelve")
// errTier2OldLayout (3b, §7-G2): the recorded copy predates the v2 relpath-mirroring layout (no
// marker). Refuse rather than read a flat layout we no longer understand — safe, because tier-2
// restore is missing-file recovery and the live data still exists in that scenario.
errTier2OldLayout = util.MsgError("err.backup.a_2_mentes_regi_formatumu_futtass")
// ErrTier2NoRestorableData (C9-F1) — this app HAS a Tier-2 copy, but that copy contains no subtree
// this restore can read: its data lives entirely in Docker named volumes, which are captured into
// recovery-unit/ (db-dumps + volume-dumps) and NEVER read by this path. 43 of the 53 catalog apps
// are in this class. Exported so the handler can refuse BEFORE stopping the app and name the action
// that does work, instead of taking an outage and reporting "no missing files".
ErrTier2NoRestorableData = util.MsgError("err.backup.ennek_az_alkalmazasnak_az_adatai_nem")
// ErrTier2NoUnitInCopy (R-102) — the recorded Tier-2 copy holds no OPENABLE recovery unit: either
// recovery-unit/ is absent, or it is a directory without a readable manifest.json. Exported so the
// handler can refuse before beginning any op. FAIL CLOSED is the whole point of the second half:
// a directory that exists is not a package, and reading a half-copied mirror as if it were one is
// how a restore would overwrite live data with nothing.
ErrTier2NoUnitInCopy = util.MsgError("err.backup.a_masodlagos_masolatban_nincs_megnyithato_mentesi")
)
// Tier2Coverage says what a Tier-2 restore can and cannot return for one app — the asymmetry C9-F1
// is about. Computed from the RECORDED copy on disk, never guessed from the catalog, so an app whose
// template changed is judged by what its actual copy holds.
//
// The distinction that matters: Legs are the subtrees RestoreTier2Files reads (hdd/, userdata/);
// HasUnit means the copy ALSO holds a full recovery unit — the app's database dumps and named-volume
// tarballs — which this restore path never opens. An app can have HasUnit && no Legs (43 of 53), in
// which case the restore is a guaranteed no-op no matter how much data was lost.
type Tier2Coverage struct {
Legs []string // subtrees the FILE restore reads and that exist in the copy: "hdd", "userdata"
HasUnit bool // recovery-unit/ present as a DIRECTORY — the disclosure fact, see below
// UnitRestorable (R-102) — the mirror is a real PACKAGE, not merely a directory: recovery-unit/
// exists AND carries a manifest.json that parses. This is the gate for the UNIT restore.
//
// It is a SECOND FIELD and not a widening of HasUnit, and the distinction is load-bearing in both
// directions. HasUnit answers "is there captured data this FILE restore is not looking at?" — the
// question tier2UnitNotCoveredMsg is appended for, and the honest answer for a half-copied mirror
// is still yes. UnitRestorable answers "can the unit restore open this?" — and for that same
// half-copied mirror the answer is no. Collapsing them would either silence a true disclosure or
// arm a restore over an unopenable package.
UnitRestorable bool
// CopyLastRun / CopyLastSuccess — WHEN the copy this restore would read was written, so the
// surface can name the date before it overwrites anything with it (Scenario E). Filled only by
// Tier2RestoreCoverage, which is the path that holds the settings; tier2CoverageAt is a pure
// filesystem inspection and leaves them empty. They are STRINGS in the recorded RFC3339 form,
// carried verbatim — no formatting decision is taken in this package.
//
// R-101 applies here exactly as it does on the backup card: CopyLastRun is the ATTEMPT clock and
// CopyLastSuccess is the only evidence a copy was actually made. A surface that shows one must
// not present it as the other.
CopyLastRun string
CopyLastSuccess string
// UnitPackageDate / UnitLegPreserved (R-403) — WHEN the package in this copy was actually
// captured, and whether the newest run PRESERVED it instead of refreshing it.
//
// They exist because after an R-403 skip, CopyLastRun and CopyLastSuccess stop describing the
// package: the run really did succeed and really is from today, and the package in the copy is
// from before it. A surface that names the run date as the package date would be trading a data
// loss for a comforting lie, which is the failure family this project keeps finding.
//
// UnitPackageDate is read from the MIRRORED UNIT'S OWN MANIFEST, not from the recorded status, so
// it is a fact about the artifact the restore will actually open. "" means UNKNOWN.
UnitPackageDate string
UnitLegPreserved bool
// UnitDataDate (R-476) — the newest ARTIFACT in the mirrored unit: its dumps' mtime, or the
// manifest's when nothing is newer. The manifest moves only when the app's DEFINITION changes
// (checksum-skip), while the nightly dumps keep their names and their fresh bytes — so on
// demo-hp a copy holding a dump written at 00:30Z was dated by a manifest from the day before.
// RFC3339 UTC; "" when the unit is not readable.
UnitDataDate string
}
// CanRestore reports whether the FILE restore has any subtree to read at all.
//
// R-102/R-103: this answers exactly one question and must keep answering only that one. Widening it
// to include the unit is R-356 arriving a second time — there, ONE predicate meant both "has this app
// a drive?" and "is this app installed?", and it refused 40 running apps for months while telling
// their owners to reinstall them somewhere those apps never offer. Two questions, two predicates.
func (c Tier2Coverage) CanRestore() bool { return len(c.Legs) > 0 }
// CanRestoreUnit reports whether the UNIT restore can run from this copy — the second predicate.
func (c Tier2Coverage) CanRestoreUnit() bool { return c.UnitRestorable }
// tier2UnitDir returns the recovery-unit directory inside a resolved Tier-2 copy. ONE expression of
// where Tier-2 puts the mirror; tier2.go writes it at the same relative name ("Unit leg (always)").
func tier2UnitDir(destBase string) string {
return filepath.Join(destBase, "recovery-unit")
}
// tier2UnitIsOpenable reports whether a mirrored unit directory is a PACKAGE and not just a
// directory. Fail-closed by construction: the manifest must be present AND parse (readManifest
// returns nil for both a missing file and malformed JSON), because an unopenable unit that armed a
// restore would stop the app, replay nothing, and rewrite its definition from an empty capture.
//
// This is the R-358 lesson one tier over — the off-site scratch marker — arriving on Tier-2.
func tier2UnitIsOpenable(unitDir string) bool {
if fi, err := os.Stat(unitDir); err != nil || !fi.IsDir() {
return false
}
return readManifest(UnitManifestFile(unitDir)) != nil
}
// tier2CoverageAt inspects a resolved copy directory. Pure filesystem stat/read — no side effects.
func tier2CoverageAt(destBase string) Tier2Coverage {
var c Tier2Coverage
for _, leg := range []string{"hdd", "userdata"} {
if fi, err := os.Stat(filepath.Join(destBase, leg)); err == nil && fi.IsDir() {
c.Legs = append(c.Legs, leg)
}
}
unitDir := tier2UnitDir(destBase)
if fi, err := os.Stat(unitDir); err == nil && fi.IsDir() {
c.HasUnit = true
}
c.UnitRestorable = tier2UnitIsOpenable(unitDir)
// R-403: ask the package itself when it was made. Reading the artifact rather than the status
// record is what makes this date impossible to overstate.
c.UnitPackageDate = unitPackageDate(unitDir)
if newest, ok := unitNewestArtifact(unitDir); ok {
c.UnitDataDate = newest.UTC().Format(time.RFC3339)
}
return c
}
// Tier2RestoreCoverage resolves the app's RECORDED Tier-2 copy and reports what a restore could
// return from it. Errors are the same refusals RestoreTier2Files itself would raise, so the caller
// can surface them before starting anything — this is what lets the handler refuse without an outage.
func (m *Manager) Tier2RestoreCoverage(stackName string) (Tier2Coverage, error) {
destBase, err := m.tier2RecordedCopyDir(stackName)
if err != nil {
return Tier2Coverage{}, err
}
cov := tier2CoverageAt(destBase)
// R-102: carry WHEN the copy was written, so the surface can name the date on an action that
// overwrites live data with it. Read from the same recorded config tier2RecordedCopyDir just
// resolved the path from, so the date and the directory cannot describe different runs.
if m.settings != nil {
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil {
cov.CopyLastRun, cov.CopyLastSuccess = cfg.LastRun, cfg.LastSuccess
cov.UnitLegPreserved = cfg.UnitLegSkipped
}
}
return cov, nil
}
// RestoreTier2Unit runs the FULL recovery-unit restore from the app's Tier-2 copy on the SECOND
// DRIVE — R-102, and the reason this task exists.
//
// Tier-2 has mirrored each app's whole recovery unit to
// <dest>/backups/secondary/<app>/recovery-unit/ on every run for months, and no code path read it:
// every reader of a unit could only name a path under backups/primary/. So in the exact failure
// Tier-2 exists for — the primary drive is lost, and the primary unit with it — the surviving copy
// was unopenable by any customer action (07-backup-architecture §6.3, §7.2).
//
// It is NOT the additive file restore beside it. This one OVERWRITES: named volumes are recreated
// from the mirror's tars and the database is replayed from the mirror's dump. The surface must carry
// that difference; see tier2UnitConfirm in the web package.
//
// THE SINGLE-WRITER FLAG IS TAKEN INSIDE RestoreFromRecoveryUnitAt, exactly as on the primary path —
// do NOT add an acquireRunning() here. A second acquire would refuse the restore it is guarding.
func (m *Manager) RestoreTier2Unit(stackName string) (UnitRestoreResult, error) {
destBase, err := m.tier2RecordedCopyDir(stackName)
if err != nil {
return UnitRestoreResult{}, err
}
unitDir := tier2UnitDir(destBase)
if !tier2UnitIsOpenable(unitDir) {
m.logger.Printf("[WARN] [backup] Tier-2 unit restore refused for %s: no openable recovery unit in the recorded copy — the app was NOT stopped", stackName)
return UnitRestoreResult{}, ErrTier2NoUnitInCopy
}
// WARN, not INFO: this is the destructive one of the two Tier-2 restores. The unit directory is a
// path and never a secret, and naming it is what makes "the SECONDARY mirror was the source" a
// positive observable in the log rather than an absence to be argued from.
m.logger.Printf("[WARN] [backup] Tier-2 UNIT restore for %s from the secondary mirror %s — this OVERWRITES live app data", stackName, unitDir)
res, restoreErr := m.RestoreFromRecoveryUnitAt(stackName, unitDir)
// R-403, the CAUSE half. Refill the primary unit from the mirror we just restored from, INSIDE
// this call, before it returns.
//
// THE TIMING IS THE REQUIREMENT, NOT A DETAIL. On 2026-08-31 the hollow primary manifest was
// written TWO SECONDS after a restore of exactly this shape, by the 5-minute `backup-cache` job
// (`backup.go` → `captureAllRecoveryUnits`). Any follow-up job, scheduled refresh or goroutine
// races that capture and can lose. Doing it here is the only shape that cannot.
//
// The capture itself is NOT guarded and must not be: a capture that describes an empty drive as
// empty is CORRECT. With the primary refilled there is no hollow state left for it to describe,
// which is why the fix is here and not there. Guarding the capture would make the manifest lie.
m.rehydratePrimaryUnit(stackName, unitDir, restoreErr)
return res, restoreErr
}
// rehydratePrimaryUnit copies a mirrored recovery unit back onto the app's own drive when the primary
// unit is ABSENT or HOLLOW — the state a Tier-2 unit restore leaves behind, and the state that armed
// R-403's delete on the following night.
//
// Three refusals, each earned:
// - the restore FAILED → write nothing. A package written from a run that did not succeed is worse
// than no package: it would look like a backup and describe data that never landed.
// - the primary already CARRIES DATA → leave it byte-identical. It may be NEWER than the mirror
// (the customer restored while their own drive was fine), and overwriting it with an older copy
// is the very move this whole task exists to prevent, pointed the other way.
// - anything goes wrong copying → WARN and carry on. The restore itself succeeded; the app is back.
// Failing the restore because a convenience copy failed would report a success as a failure.
//
// It is best-effort by design and says so in the log either way, because an absent log line is not
// evidence that it ran.
func (m *Manager) rehydratePrimaryUnit(stackName, mirrorUnitDir string, restoreErr error) {
if restoreErr != nil {
m.logger.Printf("[INFO] [backup] %s: primary unit NOT refilled — the restore itself failed (R-403: a package from a failed run is worse than none)", stackName)
return
}
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" || !filepath.IsAbs(drivePath) {
m.logger.Printf("[WARN] [backup] %s: primary unit NOT refilled — cannot resolve the app's drive", stackName)
return
}
primaryUnit := RecoveryUnitPath(m.namespaceRoot(drivePath), stackName)
if unitCarriesData(primaryUnit) {
m.logger.Printf("[INFO] [backup] %s: primary unit already carries data — left untouched (R-403 never overwrites a richer package with a poorer one)", stackName)
return
}
copier := m.unitRehydrate
if copier == nil {
copier = rsyncMirror
}
if err := copier(mirrorUnitDir, primaryUnit); err != nil {
m.logger.Printf("[ERROR] [backup] %s: refilling the primary unit from the mirror FAILED: %v — the restore itself SUCCEEDED and the app is running; the local package stays incomplete until the next backup", stackName, err)
return
}
m.logger.Printf("[INFO] [backup] %s: primary unit refilled from the secondary mirror (R-403) — %d volume tar(s), %d database dump(s) now on the app's own drive",
stackName, countUnitFiles(UnitVolumeDumpDir(primaryUnit), ".tar"), countUnitFiles(UnitDBDumpDir(primaryUnit), ".sql"))
}
// countUnitFiles counts files with a suffix in a unit leg directory — for the log line only, so the
// refill states WHAT it put back rather than merely that it ran. Never a secret: counts, not names.
func countUnitFiles(dir, suffix string) int {
entries, err := os.ReadDir(dir)
if err != nil {
return 0
}
n := 0
for _, e := range entries {
if !e.IsDir() && strings.HasSuffix(e.Name(), suffix) {
n++
}
}
return n
}
// Tier2CopyDate returns the date the surface should name for this app's Tier-2 copy, preferring the
// last SUCCESS over the last ATTEMPT (R-101: a timestamp that records "we tried" cannot answer "did
// it work"), and reports whether the returned value is a proven success.
//
// It exists so the confirm text and the outcome sentence cannot disagree about which copy is being
// restored: one resolver, two readers.
func (c Tier2Coverage) Tier2CopyDate() (date string, proven bool) {
if c.CopyLastSuccess != "" {
return c.CopyLastSuccess, true
}
return c.CopyLastRun, false
}
// UnitRestoreDate returns the date of the PACKAGE the unit restore would actually open, and whether
// the newest run PRESERVED that package rather than refreshing it.
//
// R-403. `Tier2CopyDate` answers "when was this copy last written to" and is right for the file
// restore, whose legs really were refreshed by that run. It is the WRONG answer for the unit restore
// after a preserved leg, because the package is then from before the run that reports success. This
// asks the manifest first and falls back to the copy date only when the package cannot say.
//
// THE SECOND RETURN IS `UnitLegPreserved` AND NOTHING ELSE, and the first draft got this wrong in a
// way only the live run caught. It also compared the package's date against the run's and flagged
// "older" — but a unit is ALWAYS captured shortly before the run that mirrors it, so that comparison
// was true for every healthy app on the box and every one of them rendered the warning. Live on
// demo-hp 2026-08-31: bookstack, kimai, opengist and privatebin all had src and dest manifests at
// `12:03:49Z` against a run at `12:14:24Z` — perfectly healthy, and all four would have been told
// their package was stale. A warning that fires on everything is a warning nobody reads, which costs
// the same as the comforting lie it was meant to replace.
//
// R-476: when the leg was NOT preserved, the package's date is its DATA time — the newest dump in
// the copy — never the manifest's, which moves only when the definition changes and so undersold a
// fresh copy by a day. A PRESERVED package keeps the manifest date: nothing in it is newer, and the
// R-403 rule that a preserved package is never shown as fresh is what this sits under.
func (c Tier2Coverage) UnitRestoreDate() (date string, preserved bool) {
if c.UnitPackageDate == "" {
copyDate, _ := c.Tier2CopyDate()
return copyDate, c.UnitLegPreserved
}
if !c.UnitLegPreserved && c.UnitDataDate != "" && c.UnitDataDate > c.UnitPackageDate {
return c.UnitDataDate, false
}
return c.UnitPackageDate, c.UnitLegPreserved
}
// tier2RecordedCopyDir resolves the RECORDED Tier-2 copy dir for a stack, applying every
// source-side refusal in one place so the pre-flight check and the restore itself cannot drift.
func (m *Manager) tier2RecordedCopyDir(stackName string) (string, error) {
var destBase string
if m.settings != nil {
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil && cfg.LastRun != "" && cfg.DestinationPath != "" {
if m.settings.IsDisconnected(cfg.DestinationPath) {
return "", errTier2DriveGone
}
destBase = filepath.Join(cfg.DestinationPath, "backups", "secondary", stackName)
}
}
if destBase == "" {
return "", errNoTier2Copy
}
if _, statErr := os.Stat(destBase); statErr != nil {
return "", errNoTier2Copy // recorded but the copy dir is gone — same honest refusal
}
// §7-G2 marker gate: a pre-v2 (flat) copy has no marker → refuse rather than read a layout we no
// longer understand (live data still exists for missing-file recovery).
if _, mErr := os.Stat(filepath.Join(destBase, tier2LayoutMarker)); mErr != nil {
return "", errTier2OldLayout
}
return destBase, nil
}
// RestoreTier2Files restores the app's MISSING user files in place from its recorded Tier-2 copy
// (additive-only; see the package comment above). Returns how many regular files were copied back.
//
// The source is the RECORDED Tier-2 destination (settings.CrossDriveBackup.DestinationPath) — never
// a fresh selectTier2Target, which could re-pick a different (empty) drive and "restore" nothing.
// All refusals happen BEFORE the app is stopped. Stop-first is the locked consistency policy: the
// app must not be reorganizing its data dir mid-copy.
func (m *Manager) RestoreTier2Files(stackName string) (filesRestored int, err error) {
if m.stackProvider == nil {
return 0, fmt.Errorf("stack provider not configured")
}
if err := m.acquireRunning(); err != nil {
return 0, err // shares the backup/restore single-flight — must not race a running backup
}
defer m.releaseRunning()
// Live side: the app's drive must be present and in service.
drive := m.GetAppDrivePath(stackName)
if drive == "" || !filepath.IsAbs(drive) {
return 0, fmt.Errorf("cannot determine drive path for %s", stackName)
}
if m.settings != nil {
if m.settings.IsDisconnected(drive) {
return 0, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
}
if m.settings.IsDecommissioned(drive) {
return 0, fmt.Errorf("%w (%s)", errLiveDriveDecommed, drive)
}
}
// v2 relpath-mirroring: liveNsRoot == the app's HDD_PATH (Model A). The dest hdd/ and userdata/
// subtrees mirror the live relpath structure exactly, so restore is two whole-subtree merges (N>1
// dirs + nested binds handled natively — no per-appdata-dir resolution, no N>1 refusal).
liveNsRoot := m.namespaceRoot(drive)
// Source side: the RECORDED Tier-2 copy must exist, its drive connected, and it must be v2.
destBase, err := m.tier2RecordedCopyDir(stackName)
if err != nil {
return 0, err
}
// C9-F1: refuse BEFORE the app is stopped if this copy holds nothing this path can read. Without
// this the app was stopped, zero files were copied, it was restarted, and the customer was told
// "Nincs hiányzó fájl — minden fájl megvan a helyén." — an outage plus a claim about data the
// restore never looked at. Placed with the other source-side refusals, all of which precede the
// stop, so the promise "all refusals happen BEFORE the app is stopped" stays true.
cov := tier2CoverageAt(destBase)
if !cov.CanRestore() {
m.logger.Printf("[WARN] [backup] Tier-2 file restore refused for %s: the recorded copy has no restorable subtree (unit_present=%v) — the app was NOT stopped",
stackName, cov.HasUnit)
return 0, ErrTier2NoRestorableData
}
copier := m.restoreFilesCopier
if copier == nil {
copier = rsyncRestoreMissing
}
// The two v2 subtree merges: destBase/hdd/<rel> ↔ liveNsRoot/<rel>;
// destBase/userdata/<rel> ↔ liveNsRoot/userdata/<rel>. Each missing-only, additive.
merges := []struct{ src, dst string }{
{filepath.Join(destBase, "hdd"), liveNsRoot},
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
}
m.logger.Printf("[INFO] [backup] Tier-2 file restore for %s: %s (v2) → %s (additive-only)", stackName, destBase, liveNsRoot)
// Stop → copy → start → health (the standard restore shape; F17: errors surface, never swallowed).
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
m.logger.Printf("[WARN] [backup] could not stop %s before Tier-2 file restore: %v (continuing)", stackName, stopErr)
}
start := time.Now()
var copyErr error
for _, mg := range merges {
if _, err := os.Stat(mg.src); err != nil {
continue // that subtree is absent in this copy (e.g. no userdata legs) — skip
}
n, err := copier(mg.src, mg.dst)
filesRestored += n
if err != nil {
copyErr = err
break
}
}
startErr := m.stackProvider.StartStack(stackName)
if startErr != nil {
m.logger.Printf("[ERROR] [backup] failed to restart %s after Tier-2 file restore: %v", stackName, startErr)
}
if healthErr := m.waitForHealthy(stackName, 90*time.Second); healthErr != nil {
m.logger.Printf("[WARN] [backup] %s Tier-2 file restore done but health check failed: %v", stackName, healthErr)
}
if copyErr != nil {
return filesRestored, util.MsgError("err.backup.fajlmasolas_sikertelen", copyErr)
}
if startErr != nil {
return filesRestored, util.MsgError("err.backup.fajl_visszaallitva_de_az_alkalmazas_ujrainditasa", filesRestored, startErr)
}
// Privacy: count + duration only — customer file names never at INFO.
m.logger.Printf("[INFO] [backup] Tier-2 file restore completed for %s: %d file(s) restored (%s)",
stackName, filesRestored, time.Since(start).Round(time.Second))
return filesRestored, nil
}
// rsyncRestoreMissing copies the files MISSING from dst back from src, and nothing else:
// `rsync -a --ignore-existing` — existing dst files are never overwritten, and (unlike rsyncMirror,
// which carries --delete for the backup direction) nothing at dst is ever deleted. Returns the
// number of regular files transferred, counted from --itemize-changes output.
func rsyncRestoreMissing(src, dst string) (int, error) {
if err := os.MkdirAll(dst, 0755); err != nil {
return 0, fmt.Errorf("mkdir %s: %w", dst, err)
}
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
defer cancel()
// Trailing slashes: copy the CONTENTS of src into dst (same shape as rsyncMirror).
cmd := exec.CommandContext(ctx, "rsync", "-a", "--ignore-existing", "--itemize-changes",
strings.TrimRight(src, "/")+"/", strings.TrimRight(dst, "/")+"/")
out, err := cmd.CombinedOutput()
if err != nil {
return 0, fmt.Errorf("%v: %s", err, strings.TrimSpace(string(out)))
}
return countRestoredFiles(string(out)), nil
}
// countRestoredFiles counts the itemize-changes lines that mark a TRANSFERRED regular file (">f…").
// Created dirs ("cd…") and symlinks ("cL…") are not counted — the flash reports files. Pure
// (unit-tested without rsync).
func countRestoredFiles(itemizedOut string) int {
n := 0
for _, line := range strings.Split(itemizedOut, "\n") {
if strings.HasPrefix(line, ">f") {
n++
}
}
return n
}