Files
felhom-controller/controller/internal/backup/tier2_whole.go
T

354 lines
12 KiB
Go

package backup
import (
"bytes"
"crypto/sha256"
"errors"
"fmt"
"io"
"io/fs"
"os"
"path/filepath"
"syscall"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
)
// ── The WHOLE restore from the second drive (`09` §3 decision 26, R-661; controller v0.269.0) ────
//
// WHY IT EXISTS. For an app whose files live on the data drive (DeclaredDriveFileLegs — nextcloud,
// immich, paperless-ngx, calibre-web) the Tier-2 mirror holds BOTH halves: the recovery unit (settings +
// database) and the drive files (hdd/<rel>, userdata/<rel>). Until v0.269.0 no single action brought
// the app back from it: the unit restore refuses such an app (R-538 — it will not replay a database over
// files it cannot also restore), and the file restore only ADDS missing files and replays no database.
// Measured from source 2026-09-24. So after a failed update and a failed undo, a box with a second drive
// but no off-site tier still stranded the app (R-659's case). The operator ruled (decision 26): one
// action brings the app back WHOLE from the second drive. R-538's guard STAYS — this is a new action
// beside it; it passes AcceptMissingFiles only because it has just put the files back itself.
//
// THE FILE RULES, in this order of importance (the operator's brief, verbatim in spirit):
//
// 1. NEVER DELETE a live file. Nothing here removes anything the household has.
// 2. NEVER OVERWRITE a file whose live copy is NEWER than the mirror's (mtime).
// 3. BRING BACK every file that is missing.
// 4. A file that DIFFERS and is OLDER (or equally old) on the live side is replaced from the mirror, and
// the live copy is KEPT BESIDE it as `<name>.felhom-<ts>`.
//
// rsyncMirror carries --delete and is NEVER used here; rsyncRestoreMissing is additive but cannot do
// rule 4 or count it, so the merge is done file by file, and every decision is counted.
// WholeRestoreCounts is what the whole restore did to the drive files, per rule.
type WholeRestoreCounts struct {
Restored int // rule 3: missing live, copied back
Replaced int // rule 4: live older + different → mirror copy in place, live kept beside
KeptNewer int // rule 2: live newer → left alone
Unchanged int // same content → nothing to do
BytesTotal int64
}
// wholeRestoreSuffix names the copy rule 4 keeps beside a replaced file.
func wholeRestoreSuffix(ts time.Time) string { return ".felhom-" + ts.UTC().Format("20060102T150405Z") }
func sameContent(a, b string) (bool, error) {
ha, err := fileSHA(a)
if err != nil {
return false, err
}
hb, err := fileSHA(b)
if err != nil {
return false, err
}
return bytes.Equal(ha, hb), nil
}
func fileSHA(p string) ([]byte, error) {
f, err := os.Open(p)
if err != nil {
return nil, err
}
defer f.Close()
h := sha256.New()
if _, err := io.Copy(h, f); err != nil {
return nil, err
}
return h.Sum(nil), nil
}
// copyFilePreserving copies src to dst (which must not exist), keeping mode, mtime and — when possible —
// the owner, so a restored file is readable by the app exactly as the mirrored one was.
func copyFilePreserving(src, dst string, fi fs.FileInfo) error {
in, err := os.Open(src)
if err != nil {
return err
}
defer in.Close()
tmp := dst + ".felhom-restoring"
out, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_EXCL, fi.Mode().Perm())
if err != nil {
return err
}
if _, err := io.Copy(out, in); err != nil {
out.Close()
os.Remove(tmp)
return err
}
if err := out.Sync(); err != nil {
out.Close()
os.Remove(tmp)
return err
}
if err := out.Close(); err != nil {
os.Remove(tmp)
return err
}
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
_ = os.Lchown(tmp, int(st.Uid), int(st.Gid))
}
_ = os.Chmod(tmp, fi.Mode().Perm())
_ = os.Chtimes(tmp, fi.ModTime(), fi.ModTime())
// Link, not rename: rename would silently REPLACE a dst that appeared meanwhile (rule 1/2).
if err := os.Link(tmp, dst); err != nil {
os.Remove(tmp)
return err
}
return os.Remove(tmp)
}
// mergeRestoreFiles applies the four rules from the mirror subtree src onto the live subtree dst. It
// never deletes and never follows a symlink out of either tree. Directories missing live are created
// with the mirror's mode and owner.
func mergeRestoreFiles(src, dst string, ts time.Time) (WholeRestoreCounts, error) {
var c WholeRestoreCounts
suffix := wholeRestoreSuffix(ts)
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
if walkErr != nil {
return walkErr
}
rel, err := filepath.Rel(src, p)
if err != nil {
return err
}
target := filepath.Join(dst, rel)
fi, err := d.Info()
if err != nil {
return err
}
switch {
case d.IsDir():
if _, err := os.Lstat(target); errors.Is(err, fs.ErrNotExist) {
if err := os.MkdirAll(target, fi.Mode().Perm()); err != nil {
return err
}
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
_ = os.Lchown(target, int(st.Uid), int(st.Gid))
}
}
return nil
case !fi.Mode().IsRegular():
return nil // symlinks, sockets, devices: never materialised by a restore
}
c.BytesTotal += fi.Size()
live, err := os.Lstat(target)
if errors.Is(err, fs.ErrNotExist) {
if err := copyFilePreserving(p, target, fi); err != nil {
return fmt.Errorf("restoring %s: %w", rel, err)
}
c.Restored++
return nil
}
if err != nil {
return err
}
if !live.Mode().IsRegular() {
c.KeptNewer++ // something else lives there now (a dir, a link): never touched (rule 1)
return nil
}
if live.Size() == fi.Size() {
same, err := sameContent(p, target)
if err != nil {
return err
}
if same {
c.Unchanged++
return nil
}
}
if live.ModTime().After(fi.ModTime()) {
c.KeptNewer++ // rule 2: the household's newer copy wins
return nil
}
// rule 4: keep the live copy beside, then put the mirror's in place.
kept := target + suffix
if err := os.Link(target, kept); err != nil {
return fmt.Errorf("keeping the live copy of %s: %w", rel, err)
}
if err := os.Remove(target); err != nil {
return fmt.Errorf("moving the live copy of %s aside: %w", rel, err)
}
if err := copyFilePreserving(p, target, fi); err != nil {
// put the live copy back where it was; the kept link stays either way
_ = os.Link(kept, target)
return fmt.Errorf("replacing %s: %w", rel, err)
}
c.Replaced++
return nil
})
return c, err
}
// planWholeRestoreBytes is how many bytes the file half would ADD to the live drive (rule 3 and rule 4
// copies — rule 4 keeps the old copy, so it adds its full size). Read-only.
func planWholeRestoreBytes(src, dst string) (int64, error) {
var need int64
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
if walkErr != nil || d.IsDir() {
return walkErr
}
fi, err := d.Info()
if err != nil || !fi.Mode().IsRegular() {
return err
}
rel, _ := filepath.Rel(src, p)
live, err := os.Lstat(filepath.Join(dst, rel))
if err != nil || live.Size() != fi.Size() || !live.ModTime().After(fi.ModTime()) {
need += fi.Size()
}
return nil
})
return need, err
}
// wholeRestoreFloorBytes is the free space the live drive must keep after the file half. The same 2 GB
// the update keeps on the Docker root (stacks.updateDiskFloorGiB).
const wholeRestoreFloorBytes = int64(2) << 30
var (
// ErrWholeRestoreNotFileApp — the app keeps no files on the drive; its whole restore is the unit
// restore itself („Teljes visszaállítás").
ErrWholeRestoreNotFileApp = util.MsgError("err.backup.whole_restore_not_file_app")
// ErrWholeRestoreNoProvenCopy — the mirror's unit is not a proven, openable package, or the mirror
// carries no file leg: nothing moves.
ErrWholeRestoreNoProvenCopy = util.MsgError("err.backup.whole_restore_no_proven_copy")
)
// wholeRestoreSpace is the refusal when the file half would leave less than the floor free.
type wholeRestoreSpace struct{ need, free int64 }
func (e *wholeRestoreSpace) Error() string {
return util.Text("hu", "err.backup.whole_restore_space", float64(e.need)/(1<<30), float64(e.free)/(1<<30))
}
// WholeRestoreResult is what a whole restore returns.
type WholeRestoreResult struct {
Files WholeRestoreCounts
Unit UnitRestoreResult
}
// RestoreTier2Whole brings a FILE app back whole from its recorded Tier-2 copy: the drive files by the
// four rules, then the unit (settings + database) — with the app stopped throughout. Every refusal
// happens before anything moves.
func (m *Manager) RestoreTier2Whole(stackName string) (WholeRestoreResult, error) {
var res WholeRestoreResult
if m.stackProvider == nil {
return res, fmt.Errorf("stack provider not configured")
}
if !m.HasDriveFileLegs(stackName) {
return res, ErrWholeRestoreNotFileApp
}
destBase, err := m.tier2RecordedCopyDir(stackName)
if err != nil {
return res, err
}
cov := tier2CoverageAt(destBase)
rp, rerr := m.Tier2UnitRestorePoint(stackName)
_, proven := rp.ProvenCopyTime()
if rerr != nil || !cov.CanRestoreUnit() || !cov.CanRestore() || !proven {
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: unit_openable=%v legs=%v proven=%v (%v) — nothing moved",
stackName, cov.CanRestoreUnit(), cov.Legs, proven, rerr)
return res, ErrWholeRestoreNoProvenCopy
}
drive := m.GetAppDrivePath(stackName)
if drive == "" || !filepath.IsAbs(drive) {
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
}
if m.settings != nil && (m.settings.IsDisconnected(drive) || m.settings.IsDecommissioned(drive)) {
return res, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
}
liveNsRoot := m.namespaceRoot(drive)
merges := []struct{ src, dst string }{
{filepath.Join(destBase, "hdd"), liveNsRoot},
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
}
var need int64
for _, mg := range merges {
if _, err := os.Stat(mg.src); err != nil {
continue
}
n, err := planWholeRestoreBytes(mg.src, mg.dst)
if err != nil {
return res, fmt.Errorf("planning the file half: %w", err)
}
need += n
}
free := m.freeBytes(liveNsRoot)
if free >= 0 && free-need < wholeRestoreFloorBytes {
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: the file half needs %d B and %d B is free (floor %d B) — nothing moved", stackName, need, free, wholeRestoreFloorBytes)
return res, &wholeRestoreSpace{need: need, free: free}
}
// THE FILE HALF, under the backup single-flight, with the app stopped.
if err := m.acquireRunning(); err != nil {
return res, err
}
m.logger.Printf("[WARN] [backup] WHOLE restore for %s from the second drive %s: files by the four rules (never delete, never overwrite newer), then the unit", stackName, destBase)
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
m.logger.Printf("[WARN] [backup] could not stop %s before the whole restore: %v (continuing)", stackName, stopErr)
}
ts := time.Now()
var mergeErr error
for _, mg := range merges {
if _, err := os.Stat(mg.src); err != nil {
continue
}
c, err := mergeRestoreFiles(mg.src, mg.dst, ts)
res.Files.Restored += c.Restored
res.Files.Replaced += c.Replaced
res.Files.KeptNewer += c.KeptNewer
res.Files.Unchanged += c.Unchanged
res.Files.BytesTotal += c.BytesTotal
if err != nil {
mergeErr = err
break
}
}
m.releaseRunning()
m.logger.Printf("[INFO] [backup] whole restore %s: files restored=%d replaced=%d (live kept beside as *%s) kept-newer=%d unchanged=%d",
stackName, res.Files.Restored, res.Files.Replaced, wholeRestoreSuffix(ts), res.Files.KeptNewer, res.Files.Unchanged)
if mergeErr != nil {
// The unit is NOT replayed over a half-merged file tree; nothing was deleted, the app is restarted.
if startErr := m.stackProvider.StartStack(stackName); startErr != nil {
m.logger.Printf("[ERROR] [backup] restarting %s after a failed file half also failed: %v", stackName, startErr)
}
return res, util.MsgError("err.backup.fajlmasolas_sikertelen", mergeErr)
}
// THE UNIT HALF — the existing restore, told that the files are handled (R-538 stays for every other caller).
unitDir := tier2UnitDir(destBase)
res.Unit, err = m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{AcceptMissingFiles: true})
m.rehydratePrimaryUnit(stackName, unitDir, err)
return res, err
}
// freeBytes is the free space under p, -1 when unknown. A seam for tests.
func (m *Manager) freeBytes(p string) int64 {
if m.freeBytesFn != nil {
return m.freeBytesFn(p)
}
var st syscall.Statfs_t
if err := syscall.Statfs(p, &st); err != nil {
return -1
}
return int64(st.Bavail) * int64(st.Bsize)
}