package backup import ( "bytes" "crypto/sha256" "errors" "fmt" "io" "io/fs" "os" "path/filepath" "syscall" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/util" ) // ── The WHOLE restore from the second drive (`09` §3 decision 26, R-661; controller v0.269.0) ──── // // WHY IT EXISTS. For an app whose files live on the data drive (DeclaredDriveFileLegs — nextcloud, // immich, paperless-ngx, calibre-web) the Tier-2 mirror holds BOTH halves: the recovery unit (settings + // database) and the drive files (hdd/, userdata/). Until v0.269.0 no single action brought // the app back from it: the unit restore refuses such an app (R-538 — it will not replay a database over // files it cannot also restore), and the file restore only ADDS missing files and replays no database. // Measured from source 2026-09-24. So after a failed update and a failed undo, a box with a second drive // but no off-site tier still stranded the app (R-659's case). The operator ruled (decision 26): one // action brings the app back WHOLE from the second drive. R-538's guard STAYS — this is a new action // beside it; it passes AcceptMissingFiles only because it has just put the files back itself. // // THE FILE RULES, in this order of importance (the operator's brief, verbatim in spirit): // // 1. NEVER DELETE a live file. Nothing here removes anything the household has. // 2. NEVER OVERWRITE a file whose live copy is NEWER than the mirror's (mtime). // 3. BRING BACK every file that is missing. // 4. A file that DIFFERS and is OLDER (or equally old) on the live side is replaced from the mirror, and // the live copy is KEPT BESIDE it as `.felhom-`. // // rsyncMirror carries --delete and is NEVER used here; rsyncRestoreMissing is additive but cannot do // rule 4 or count it, so the merge is done file by file, and every decision is counted. // WholeRestoreCounts is what the whole restore did to the drive files, per rule. type WholeRestoreCounts struct { Restored int // rule 3: missing live, copied back Replaced int // rule 4: live older + different → mirror copy in place, live kept beside KeptNewer int // rule 2: live newer → left alone Unchanged int // same content → nothing to do BytesTotal int64 } // wholeRestoreSuffix names the copy rule 4 keeps beside a replaced file. func wholeRestoreSuffix(ts time.Time) string { return ".felhom-" + ts.UTC().Format("20060102T150405Z") } func sameContent(a, b string) (bool, error) { ha, err := fileSHA(a) if err != nil { return false, err } hb, err := fileSHA(b) if err != nil { return false, err } return bytes.Equal(ha, hb), nil } func fileSHA(p string) ([]byte, error) { f, err := os.Open(p) if err != nil { return nil, err } defer f.Close() h := sha256.New() if _, err := io.Copy(h, f); err != nil { return nil, err } return h.Sum(nil), nil } // copyFilePreserving copies src to dst (which must not exist), keeping mode, mtime and — when possible — // the owner, so a restored file is readable by the app exactly as the mirrored one was. func copyFilePreserving(src, dst string, fi fs.FileInfo) error { in, err := os.Open(src) if err != nil { return err } defer in.Close() tmp := dst + ".felhom-restoring" out, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_EXCL, fi.Mode().Perm()) if err != nil { return err } if _, err := io.Copy(out, in); err != nil { out.Close() os.Remove(tmp) return err } if err := out.Sync(); err != nil { out.Close() os.Remove(tmp) return err } if err := out.Close(); err != nil { os.Remove(tmp) return err } if st, ok := fi.Sys().(*syscall.Stat_t); ok { _ = os.Lchown(tmp, int(st.Uid), int(st.Gid)) } _ = os.Chmod(tmp, fi.Mode().Perm()) _ = os.Chtimes(tmp, fi.ModTime(), fi.ModTime()) // Link, not rename: rename would silently REPLACE a dst that appeared meanwhile (rule 1/2). if err := os.Link(tmp, dst); err != nil { os.Remove(tmp) return err } return os.Remove(tmp) } // mergeRestoreFiles applies the four rules from the mirror subtree src onto the live subtree dst. It // never deletes and never follows a symlink out of either tree. Directories missing live are created // with the mirror's mode and owner. func mergeRestoreFiles(src, dst string, ts time.Time) (WholeRestoreCounts, error) { var c WholeRestoreCounts suffix := wholeRestoreSuffix(ts) err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error { if walkErr != nil { return walkErr } rel, err := filepath.Rel(src, p) if err != nil { return err } target := filepath.Join(dst, rel) fi, err := d.Info() if err != nil { return err } switch { case d.IsDir(): if _, err := os.Lstat(target); errors.Is(err, fs.ErrNotExist) { if err := os.MkdirAll(target, fi.Mode().Perm()); err != nil { return err } if st, ok := fi.Sys().(*syscall.Stat_t); ok { _ = os.Lchown(target, int(st.Uid), int(st.Gid)) } } return nil case !fi.Mode().IsRegular(): return nil // symlinks, sockets, devices: never materialised by a restore } c.BytesTotal += fi.Size() live, err := os.Lstat(target) if errors.Is(err, fs.ErrNotExist) { if err := copyFilePreserving(p, target, fi); err != nil { return fmt.Errorf("restoring %s: %w", rel, err) } c.Restored++ return nil } if err != nil { return err } if !live.Mode().IsRegular() { c.KeptNewer++ // something else lives there now (a dir, a link): never touched (rule 1) return nil } if live.Size() == fi.Size() { same, err := sameContent(p, target) if err != nil { return err } if same { c.Unchanged++ return nil } } if live.ModTime().After(fi.ModTime()) { c.KeptNewer++ // rule 2: the household's newer copy wins return nil } // rule 4: keep the live copy beside, then put the mirror's in place. kept := target + suffix if err := os.Link(target, kept); err != nil { return fmt.Errorf("keeping the live copy of %s: %w", rel, err) } if err := os.Remove(target); err != nil { return fmt.Errorf("moving the live copy of %s aside: %w", rel, err) } if err := copyFilePreserving(p, target, fi); err != nil { // put the live copy back where it was; the kept link stays either way _ = os.Link(kept, target) return fmt.Errorf("replacing %s: %w", rel, err) } c.Replaced++ return nil }) return c, err } // planWholeRestoreBytes is how many bytes the file half would ADD to the live drive (rule 3 and rule 4 // copies — rule 4 keeps the old copy, so it adds its full size). Read-only. func planWholeRestoreBytes(src, dst string) (int64, error) { var need int64 err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error { if walkErr != nil || d.IsDir() { return walkErr } fi, err := d.Info() if err != nil || !fi.Mode().IsRegular() { return err } rel, _ := filepath.Rel(src, p) live, err := os.Lstat(filepath.Join(dst, rel)) if err != nil || live.Size() != fi.Size() || !live.ModTime().After(fi.ModTime()) { need += fi.Size() } return nil }) return need, err } // wholeRestoreFloorBytes is the free space the live drive must keep after the file half. The same 2 GB // the update keeps on the Docker root (stacks.updateDiskFloorGiB). const wholeRestoreFloorBytes = int64(2) << 30 var ( // ErrWholeRestoreNotFileApp — the app keeps no files on the drive; its whole restore is the unit // restore itself („Teljes visszaállítás"). ErrWholeRestoreNotFileApp = util.MsgError("err.backup.whole_restore_not_file_app") // ErrWholeRestoreNoProvenCopy — the mirror's unit is not a proven, openable package, or the mirror // carries no file leg: nothing moves. ErrWholeRestoreNoProvenCopy = util.MsgError("err.backup.whole_restore_no_proven_copy") ) // wholeRestoreSpace is the refusal when the file half would leave less than the floor free. type wholeRestoreSpace struct{ need, free int64 } func (e *wholeRestoreSpace) Error() string { return util.Text("hu", "err.backup.whole_restore_space", float64(e.need)/(1<<30), float64(e.free)/(1<<30)) } // WholeRestoreResult is what a whole restore returns. type WholeRestoreResult struct { Files WholeRestoreCounts Unit UnitRestoreResult } // RestoreTier2Whole brings a FILE app back whole from its recorded Tier-2 copy: the drive files by the // four rules, then the unit (settings + database) — with the app stopped throughout. Every refusal // happens before anything moves. func (m *Manager) RestoreTier2Whole(stackName string) (WholeRestoreResult, error) { var res WholeRestoreResult if m.stackProvider == nil { return res, fmt.Errorf("stack provider not configured") } if !m.HasDriveFileLegs(stackName) { return res, ErrWholeRestoreNotFileApp } destBase, err := m.tier2RecordedCopyDir(stackName) if err != nil { return res, err } cov := tier2CoverageAt(destBase) rp, rerr := m.Tier2UnitRestorePoint(stackName) _, proven := rp.ProvenCopyTime() if rerr != nil || !cov.CanRestoreUnit() || !cov.CanRestore() || !proven { m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: unit_openable=%v legs=%v proven=%v (%v) — nothing moved", stackName, cov.CanRestoreUnit(), cov.Legs, proven, rerr) return res, ErrWholeRestoreNoProvenCopy } drive := m.GetAppDrivePath(stackName) if drive == "" || !filepath.IsAbs(drive) { return res, fmt.Errorf("cannot determine drive path for %s", stackName) } if m.settings != nil && (m.settings.IsDisconnected(drive) || m.settings.IsDecommissioned(drive)) { return res, fmt.Errorf("%w (%s)", errLiveDriveGone, drive) } liveNsRoot := m.namespaceRoot(drive) merges := []struct{ src, dst string }{ {filepath.Join(destBase, "hdd"), liveNsRoot}, {filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")}, } var need int64 for _, mg := range merges { if _, err := os.Stat(mg.src); err != nil { continue } n, err := planWholeRestoreBytes(mg.src, mg.dst) if err != nil { return res, fmt.Errorf("planning the file half: %w", err) } need += n } free := m.freeBytes(liveNsRoot) if free >= 0 && free-need < wholeRestoreFloorBytes { m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: the file half needs %d B and %d B is free (floor %d B) — nothing moved", stackName, need, free, wholeRestoreFloorBytes) return res, &wholeRestoreSpace{need: need, free: free} } // THE FILE HALF, under the backup single-flight, with the app stopped. if err := m.acquireRunning(); err != nil { return res, err } m.logger.Printf("[WARN] [backup] WHOLE restore for %s from the second drive %s: files by the four rules (never delete, never overwrite newer), then the unit", stackName, destBase) if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil { m.logger.Printf("[WARN] [backup] could not stop %s before the whole restore: %v (continuing)", stackName, stopErr) } ts := time.Now() var mergeErr error for _, mg := range merges { if _, err := os.Stat(mg.src); err != nil { continue } c, err := mergeRestoreFiles(mg.src, mg.dst, ts) res.Files.Restored += c.Restored res.Files.Replaced += c.Replaced res.Files.KeptNewer += c.KeptNewer res.Files.Unchanged += c.Unchanged res.Files.BytesTotal += c.BytesTotal if err != nil { mergeErr = err break } } m.releaseRunning() m.logger.Printf("[INFO] [backup] whole restore %s: files restored=%d replaced=%d (live kept beside as *%s) kept-newer=%d unchanged=%d", stackName, res.Files.Restored, res.Files.Replaced, wholeRestoreSuffix(ts), res.Files.KeptNewer, res.Files.Unchanged) if mergeErr != nil { // The unit is NOT replayed over a half-merged file tree; nothing was deleted, the app is restarted. if startErr := m.stackProvider.StartStack(stackName); startErr != nil { m.logger.Printf("[ERROR] [backup] restarting %s after a failed file half also failed: %v", stackName, startErr) } return res, util.MsgError("err.backup.fajlmasolas_sikertelen", mergeErr) } // THE UNIT HALF — the existing restore, told that the files are handled (R-538 stays for every other caller). unitDir := tier2UnitDir(destBase) res.Unit, err = m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{AcceptMissingFiles: true}) m.rehydratePrimaryUnit(stackName, unitDir, err) return res, err } // freeBytes is the free space under p, -1 when unknown. A seam for tests. func (m *Manager) freeBytes(p string) int64 { if m.freeBytesFn != nil { return m.freeBytesFn(p) } var st syscall.Statfs_t if err := syscall.Statfs(p, &st); err != nil { return -1 } return int64(st.Bavail) * int64(st.Bsize) }