3c6b49b31c
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
354 lines
12 KiB
Go
354 lines
12 KiB
Go
package backup
|
|
|
|
import (
|
|
"bytes"
|
|
"crypto/sha256"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"io/fs"
|
|
"os"
|
|
"path/filepath"
|
|
"syscall"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
|
)
|
|
|
|
// ── The WHOLE restore from the second drive (`09` §3 decision 26, R-661; controller v0.269.0) ────
|
|
//
|
|
// WHY IT EXISTS. For an app whose files live on the data drive (DeclaredDriveFileLegs — nextcloud,
|
|
// immich, paperless-ngx, calibre-web) the Tier-2 mirror holds BOTH halves: the recovery unit (settings +
|
|
// database) and the drive files (hdd/<rel>, userdata/<rel>). Until v0.269.0 no single action brought
|
|
// the app back from it: the unit restore refuses such an app (R-538 — it will not replay a database over
|
|
// files it cannot also restore), and the file restore only ADDS missing files and replays no database.
|
|
// Measured from source 2026-09-24. So after a failed update and a failed undo, a box with a second drive
|
|
// but no off-site tier still stranded the app (R-659's case). The operator ruled (decision 26): one
|
|
// action brings the app back WHOLE from the second drive. R-538's guard STAYS — this is a new action
|
|
// beside it; it passes AcceptMissingFiles only because it has just put the files back itself.
|
|
//
|
|
// THE FILE RULES, in this order of importance (the operator's brief, verbatim in spirit):
|
|
//
|
|
// 1. NEVER DELETE a live file. Nothing here removes anything the household has.
|
|
// 2. NEVER OVERWRITE a file whose live copy is NEWER than the mirror's (mtime).
|
|
// 3. BRING BACK every file that is missing.
|
|
// 4. A file that DIFFERS and is OLDER (or equally old) on the live side is replaced from the mirror, and
|
|
// the live copy is KEPT BESIDE it as `<name>.felhom-<ts>`.
|
|
//
|
|
// rsyncMirror carries --delete and is NEVER used here; rsyncRestoreMissing is additive but cannot do
|
|
// rule 4 or count it, so the merge is done file by file, and every decision is counted.
|
|
|
|
// WholeRestoreCounts is what the whole restore did to the drive files, per rule.
|
|
type WholeRestoreCounts struct {
|
|
Restored int // rule 3: missing live, copied back
|
|
Replaced int // rule 4: live older + different → mirror copy in place, live kept beside
|
|
KeptNewer int // rule 2: live newer → left alone
|
|
Unchanged int // same content → nothing to do
|
|
BytesTotal int64
|
|
}
|
|
|
|
// wholeRestoreSuffix names the copy rule 4 keeps beside a replaced file.
|
|
func wholeRestoreSuffix(ts time.Time) string { return ".felhom-" + ts.UTC().Format("20060102T150405Z") }
|
|
|
|
func sameContent(a, b string) (bool, error) {
|
|
ha, err := fileSHA(a)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
hb, err := fileSHA(b)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
return bytes.Equal(ha, hb), nil
|
|
}
|
|
|
|
func fileSHA(p string) ([]byte, error) {
|
|
f, err := os.Open(p)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer f.Close()
|
|
h := sha256.New()
|
|
if _, err := io.Copy(h, f); err != nil {
|
|
return nil, err
|
|
}
|
|
return h.Sum(nil), nil
|
|
}
|
|
|
|
// copyFilePreserving copies src to dst (which must not exist), keeping mode, mtime and — when possible —
|
|
// the owner, so a restored file is readable by the app exactly as the mirrored one was.
|
|
func copyFilePreserving(src, dst string, fi fs.FileInfo) error {
|
|
in, err := os.Open(src)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer in.Close()
|
|
tmp := dst + ".felhom-restoring"
|
|
out, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_EXCL, fi.Mode().Perm())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if _, err := io.Copy(out, in); err != nil {
|
|
out.Close()
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if err := out.Sync(); err != nil {
|
|
out.Close()
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if err := out.Close(); err != nil {
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
|
_ = os.Lchown(tmp, int(st.Uid), int(st.Gid))
|
|
}
|
|
_ = os.Chmod(tmp, fi.Mode().Perm())
|
|
_ = os.Chtimes(tmp, fi.ModTime(), fi.ModTime())
|
|
// Link, not rename: rename would silently REPLACE a dst that appeared meanwhile (rule 1/2).
|
|
if err := os.Link(tmp, dst); err != nil {
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
return os.Remove(tmp)
|
|
}
|
|
|
|
// mergeRestoreFiles applies the four rules from the mirror subtree src onto the live subtree dst. It
|
|
// never deletes and never follows a symlink out of either tree. Directories missing live are created
|
|
// with the mirror's mode and owner.
|
|
func mergeRestoreFiles(src, dst string, ts time.Time) (WholeRestoreCounts, error) {
|
|
var c WholeRestoreCounts
|
|
suffix := wholeRestoreSuffix(ts)
|
|
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
|
|
if walkErr != nil {
|
|
return walkErr
|
|
}
|
|
rel, err := filepath.Rel(src, p)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
target := filepath.Join(dst, rel)
|
|
fi, err := d.Info()
|
|
if err != nil {
|
|
return err
|
|
}
|
|
switch {
|
|
case d.IsDir():
|
|
if _, err := os.Lstat(target); errors.Is(err, fs.ErrNotExist) {
|
|
if err := os.MkdirAll(target, fi.Mode().Perm()); err != nil {
|
|
return err
|
|
}
|
|
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
|
_ = os.Lchown(target, int(st.Uid), int(st.Gid))
|
|
}
|
|
}
|
|
return nil
|
|
case !fi.Mode().IsRegular():
|
|
return nil // symlinks, sockets, devices: never materialised by a restore
|
|
}
|
|
c.BytesTotal += fi.Size()
|
|
live, err := os.Lstat(target)
|
|
if errors.Is(err, fs.ErrNotExist) {
|
|
if err := copyFilePreserving(p, target, fi); err != nil {
|
|
return fmt.Errorf("restoring %s: %w", rel, err)
|
|
}
|
|
c.Restored++
|
|
return nil
|
|
}
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if !live.Mode().IsRegular() {
|
|
c.KeptNewer++ // something else lives there now (a dir, a link): never touched (rule 1)
|
|
return nil
|
|
}
|
|
if live.Size() == fi.Size() {
|
|
same, err := sameContent(p, target)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if same {
|
|
c.Unchanged++
|
|
return nil
|
|
}
|
|
}
|
|
if live.ModTime().After(fi.ModTime()) {
|
|
c.KeptNewer++ // rule 2: the household's newer copy wins
|
|
return nil
|
|
}
|
|
// rule 4: keep the live copy beside, then put the mirror's in place.
|
|
kept := target + suffix
|
|
if err := os.Link(target, kept); err != nil {
|
|
return fmt.Errorf("keeping the live copy of %s: %w", rel, err)
|
|
}
|
|
if err := os.Remove(target); err != nil {
|
|
return fmt.Errorf("moving the live copy of %s aside: %w", rel, err)
|
|
}
|
|
if err := copyFilePreserving(p, target, fi); err != nil {
|
|
// put the live copy back where it was; the kept link stays either way
|
|
_ = os.Link(kept, target)
|
|
return fmt.Errorf("replacing %s: %w", rel, err)
|
|
}
|
|
c.Replaced++
|
|
return nil
|
|
})
|
|
return c, err
|
|
}
|
|
|
|
// planWholeRestoreBytes is how many bytes the file half would ADD to the live drive (rule 3 and rule 4
|
|
// copies — rule 4 keeps the old copy, so it adds its full size). Read-only.
|
|
func planWholeRestoreBytes(src, dst string) (int64, error) {
|
|
var need int64
|
|
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
|
|
if walkErr != nil || d.IsDir() {
|
|
return walkErr
|
|
}
|
|
fi, err := d.Info()
|
|
if err != nil || !fi.Mode().IsRegular() {
|
|
return err
|
|
}
|
|
rel, _ := filepath.Rel(src, p)
|
|
live, err := os.Lstat(filepath.Join(dst, rel))
|
|
if err != nil || live.Size() != fi.Size() || !live.ModTime().After(fi.ModTime()) {
|
|
need += fi.Size()
|
|
}
|
|
return nil
|
|
})
|
|
return need, err
|
|
}
|
|
|
|
// wholeRestoreFloorBytes is the free space the live drive must keep after the file half. The same 2 GB
|
|
// the update keeps on the Docker root (stacks.updateDiskFloorGiB).
|
|
const wholeRestoreFloorBytes = int64(2) << 30
|
|
|
|
var (
|
|
// ErrWholeRestoreNotFileApp — the app keeps no files on the drive; its whole restore is the unit
|
|
// restore itself („Teljes visszaállítás").
|
|
ErrWholeRestoreNotFileApp = util.MsgError("err.backup.whole_restore_not_file_app")
|
|
// ErrWholeRestoreNoProvenCopy — the mirror's unit is not a proven, openable package, or the mirror
|
|
// carries no file leg: nothing moves.
|
|
ErrWholeRestoreNoProvenCopy = util.MsgError("err.backup.whole_restore_no_proven_copy")
|
|
)
|
|
|
|
// wholeRestoreSpace is the refusal when the file half would leave less than the floor free.
|
|
type wholeRestoreSpace struct{ need, free int64 }
|
|
|
|
func (e *wholeRestoreSpace) Error() string {
|
|
return util.Text("hu", "err.backup.whole_restore_space", float64(e.need)/(1<<30), float64(e.free)/(1<<30))
|
|
}
|
|
|
|
// WholeRestoreResult is what a whole restore returns.
|
|
type WholeRestoreResult struct {
|
|
Files WholeRestoreCounts
|
|
Unit UnitRestoreResult
|
|
}
|
|
|
|
// RestoreTier2Whole brings a FILE app back whole from its recorded Tier-2 copy: the drive files by the
|
|
// four rules, then the unit (settings + database) — with the app stopped throughout. Every refusal
|
|
// happens before anything moves.
|
|
func (m *Manager) RestoreTier2Whole(stackName string) (WholeRestoreResult, error) {
|
|
var res WholeRestoreResult
|
|
if m.stackProvider == nil {
|
|
return res, fmt.Errorf("stack provider not configured")
|
|
}
|
|
if !m.HasDriveFileLegs(stackName) {
|
|
return res, ErrWholeRestoreNotFileApp
|
|
}
|
|
destBase, err := m.tier2RecordedCopyDir(stackName)
|
|
if err != nil {
|
|
return res, err
|
|
}
|
|
cov := tier2CoverageAt(destBase)
|
|
rp, rerr := m.Tier2UnitRestorePoint(stackName)
|
|
_, proven := rp.ProvenCopyTime()
|
|
if rerr != nil || !cov.CanRestoreUnit() || !cov.CanRestore() || !proven {
|
|
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: unit_openable=%v legs=%v proven=%v (%v) — nothing moved",
|
|
stackName, cov.CanRestoreUnit(), cov.Legs, proven, rerr)
|
|
return res, ErrWholeRestoreNoProvenCopy
|
|
}
|
|
drive := m.GetAppDrivePath(stackName)
|
|
if drive == "" || !filepath.IsAbs(drive) {
|
|
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
if m.settings != nil && (m.settings.IsDisconnected(drive) || m.settings.IsDecommissioned(drive)) {
|
|
return res, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
|
|
}
|
|
liveNsRoot := m.namespaceRoot(drive)
|
|
merges := []struct{ src, dst string }{
|
|
{filepath.Join(destBase, "hdd"), liveNsRoot},
|
|
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
|
|
}
|
|
var need int64
|
|
for _, mg := range merges {
|
|
if _, err := os.Stat(mg.src); err != nil {
|
|
continue
|
|
}
|
|
n, err := planWholeRestoreBytes(mg.src, mg.dst)
|
|
if err != nil {
|
|
return res, fmt.Errorf("planning the file half: %w", err)
|
|
}
|
|
need += n
|
|
}
|
|
free := m.freeBytes(liveNsRoot)
|
|
if free >= 0 && free-need < wholeRestoreFloorBytes {
|
|
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: the file half needs %d B and %d B is free (floor %d B) — nothing moved", stackName, need, free, wholeRestoreFloorBytes)
|
|
return res, &wholeRestoreSpace{need: need, free: free}
|
|
}
|
|
|
|
// THE FILE HALF, under the backup single-flight, with the app stopped.
|
|
if err := m.acquireRunning(); err != nil {
|
|
return res, err
|
|
}
|
|
m.logger.Printf("[WARN] [backup] WHOLE restore for %s from the second drive %s: files by the four rules (never delete, never overwrite newer), then the unit", stackName, destBase)
|
|
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
|
|
m.logger.Printf("[WARN] [backup] could not stop %s before the whole restore: %v (continuing)", stackName, stopErr)
|
|
}
|
|
ts := time.Now()
|
|
var mergeErr error
|
|
for _, mg := range merges {
|
|
if _, err := os.Stat(mg.src); err != nil {
|
|
continue
|
|
}
|
|
c, err := mergeRestoreFiles(mg.src, mg.dst, ts)
|
|
res.Files.Restored += c.Restored
|
|
res.Files.Replaced += c.Replaced
|
|
res.Files.KeptNewer += c.KeptNewer
|
|
res.Files.Unchanged += c.Unchanged
|
|
res.Files.BytesTotal += c.BytesTotal
|
|
if err != nil {
|
|
mergeErr = err
|
|
break
|
|
}
|
|
}
|
|
m.releaseRunning()
|
|
m.logger.Printf("[INFO] [backup] whole restore %s: files restored=%d replaced=%d (live kept beside as *%s) kept-newer=%d unchanged=%d",
|
|
stackName, res.Files.Restored, res.Files.Replaced, wholeRestoreSuffix(ts), res.Files.KeptNewer, res.Files.Unchanged)
|
|
if mergeErr != nil {
|
|
// The unit is NOT replayed over a half-merged file tree; nothing was deleted, the app is restarted.
|
|
if startErr := m.stackProvider.StartStack(stackName); startErr != nil {
|
|
m.logger.Printf("[ERROR] [backup] restarting %s after a failed file half also failed: %v", stackName, startErr)
|
|
}
|
|
return res, util.MsgError("err.backup.fajlmasolas_sikertelen", mergeErr)
|
|
}
|
|
|
|
// THE UNIT HALF — the existing restore, told that the files are handled (R-538 stays for every other caller).
|
|
unitDir := tier2UnitDir(destBase)
|
|
res.Unit, err = m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{AcceptMissingFiles: true})
|
|
m.rehydratePrimaryUnit(stackName, unitDir, err)
|
|
return res, err
|
|
}
|
|
|
|
// freeBytes is the free space under p, -1 when unknown. A seam for tests.
|
|
func (m *Manager) freeBytes(p string) int64 {
|
|
if m.freeBytesFn != nil {
|
|
return m.freeBytesFn(p)
|
|
}
|
|
var st syscall.Statfs_t
|
|
if err := syscall.Statfs(p, &st); err != nil {
|
|
return -1
|
|
}
|
|
return int64(st.Bavail) * int64(st.Bsize)
|
|
}
|