controller v0.269.0: whole restore from the second drive; crash loops stopped; exact image digests; steps judged by their own .felhom.yml (decisions 26-28, R-661 R-666 R-667 R-668 R-664 R-665 R-662, 09 6.4 part 6)
gates / gates (push) Successful in 27s
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,353 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
)
|
||||
|
||||
// ── The WHOLE restore from the second drive (`09` §3 decision 26, R-661; controller v0.269.0) ────
|
||||
//
|
||||
// WHY IT EXISTS. For an app whose files live on the data drive (DeclaredDriveFileLegs — nextcloud,
|
||||
// immich, paperless-ngx, calibre-web) the Tier-2 mirror holds BOTH halves: the recovery unit (settings +
|
||||
// database) and the drive files (hdd/<rel>, userdata/<rel>). Until v0.269.0 no single action brought
|
||||
// the app back from it: the unit restore refuses such an app (R-538 — it will not replay a database over
|
||||
// files it cannot also restore), and the file restore only ADDS missing files and replays no database.
|
||||
// Measured from source 2026-09-24. So after a failed update and a failed undo, a box with a second drive
|
||||
// but no off-site tier still stranded the app (R-659's case). The operator ruled (decision 26): one
|
||||
// action brings the app back WHOLE from the second drive. R-538's guard STAYS — this is a new action
|
||||
// beside it; it passes AcceptMissingFiles only because it has just put the files back itself.
|
||||
//
|
||||
// THE FILE RULES, in this order of importance (the operator's brief, verbatim in spirit):
|
||||
//
|
||||
// 1. NEVER DELETE a live file. Nothing here removes anything the household has.
|
||||
// 2. NEVER OVERWRITE a file whose live copy is NEWER than the mirror's (mtime).
|
||||
// 3. BRING BACK every file that is missing.
|
||||
// 4. A file that DIFFERS and is OLDER (or equally old) on the live side is replaced from the mirror, and
|
||||
// the live copy is KEPT BESIDE it as `<name>.felhom-<ts>`.
|
||||
//
|
||||
// rsyncMirror carries --delete and is NEVER used here; rsyncRestoreMissing is additive but cannot do
|
||||
// rule 4 or count it, so the merge is done file by file, and every decision is counted.
|
||||
|
||||
// WholeRestoreCounts is what the whole restore did to the drive files, per rule.
|
||||
type WholeRestoreCounts struct {
|
||||
Restored int // rule 3: missing live, copied back
|
||||
Replaced int // rule 4: live older + different → mirror copy in place, live kept beside
|
||||
KeptNewer int // rule 2: live newer → left alone
|
||||
Unchanged int // same content → nothing to do
|
||||
BytesTotal int64
|
||||
}
|
||||
|
||||
// wholeRestoreSuffix names the copy rule 4 keeps beside a replaced file.
|
||||
func wholeRestoreSuffix(ts time.Time) string { return ".felhom-" + ts.UTC().Format("20060102T150405Z") }
|
||||
|
||||
func sameContent(a, b string) (bool, error) {
|
||||
ha, err := fileSHA(a)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
hb, err := fileSHA(b)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return bytes.Equal(ha, hb), nil
|
||||
}
|
||||
|
||||
func fileSHA(p string) ([]byte, error) {
|
||||
f, err := os.Open(p)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
h := sha256.New()
|
||||
if _, err := io.Copy(h, f); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return h.Sum(nil), nil
|
||||
}
|
||||
|
||||
// copyFilePreserving copies src to dst (which must not exist), keeping mode, mtime and — when possible —
|
||||
// the owner, so a restored file is readable by the app exactly as the mirrored one was.
|
||||
func copyFilePreserving(src, dst string, fi fs.FileInfo) error {
|
||||
in, err := os.Open(src)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer in.Close()
|
||||
tmp := dst + ".felhom-restoring"
|
||||
out, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_EXCL, fi.Mode().Perm())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := io.Copy(out, in); err != nil {
|
||||
out.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := out.Sync(); err != nil {
|
||||
out.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := out.Close(); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
||||
_ = os.Lchown(tmp, int(st.Uid), int(st.Gid))
|
||||
}
|
||||
_ = os.Chmod(tmp, fi.Mode().Perm())
|
||||
_ = os.Chtimes(tmp, fi.ModTime(), fi.ModTime())
|
||||
// Link, not rename: rename would silently REPLACE a dst that appeared meanwhile (rule 1/2).
|
||||
if err := os.Link(tmp, dst); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Remove(tmp)
|
||||
}
|
||||
|
||||
// mergeRestoreFiles applies the four rules from the mirror subtree src onto the live subtree dst. It
|
||||
// never deletes and never follows a symlink out of either tree. Directories missing live are created
|
||||
// with the mirror's mode and owner.
|
||||
func mergeRestoreFiles(src, dst string, ts time.Time) (WholeRestoreCounts, error) {
|
||||
var c WholeRestoreCounts
|
||||
suffix := wholeRestoreSuffix(ts)
|
||||
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
|
||||
if walkErr != nil {
|
||||
return walkErr
|
||||
}
|
||||
rel, err := filepath.Rel(src, p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
target := filepath.Join(dst, rel)
|
||||
fi, err := d.Info()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch {
|
||||
case d.IsDir():
|
||||
if _, err := os.Lstat(target); errors.Is(err, fs.ErrNotExist) {
|
||||
if err := os.MkdirAll(target, fi.Mode().Perm()); err != nil {
|
||||
return err
|
||||
}
|
||||
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
||||
_ = os.Lchown(target, int(st.Uid), int(st.Gid))
|
||||
}
|
||||
}
|
||||
return nil
|
||||
case !fi.Mode().IsRegular():
|
||||
return nil // symlinks, sockets, devices: never materialised by a restore
|
||||
}
|
||||
c.BytesTotal += fi.Size()
|
||||
live, err := os.Lstat(target)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
if err := copyFilePreserving(p, target, fi); err != nil {
|
||||
return fmt.Errorf("restoring %s: %w", rel, err)
|
||||
}
|
||||
c.Restored++
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !live.Mode().IsRegular() {
|
||||
c.KeptNewer++ // something else lives there now (a dir, a link): never touched (rule 1)
|
||||
return nil
|
||||
}
|
||||
if live.Size() == fi.Size() {
|
||||
same, err := sameContent(p, target)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if same {
|
||||
c.Unchanged++
|
||||
return nil
|
||||
}
|
||||
}
|
||||
if live.ModTime().After(fi.ModTime()) {
|
||||
c.KeptNewer++ // rule 2: the household's newer copy wins
|
||||
return nil
|
||||
}
|
||||
// rule 4: keep the live copy beside, then put the mirror's in place.
|
||||
kept := target + suffix
|
||||
if err := os.Link(target, kept); err != nil {
|
||||
return fmt.Errorf("keeping the live copy of %s: %w", rel, err)
|
||||
}
|
||||
if err := os.Remove(target); err != nil {
|
||||
return fmt.Errorf("moving the live copy of %s aside: %w", rel, err)
|
||||
}
|
||||
if err := copyFilePreserving(p, target, fi); err != nil {
|
||||
// put the live copy back where it was; the kept link stays either way
|
||||
_ = os.Link(kept, target)
|
||||
return fmt.Errorf("replacing %s: %w", rel, err)
|
||||
}
|
||||
c.Replaced++
|
||||
return nil
|
||||
})
|
||||
return c, err
|
||||
}
|
||||
|
||||
// planWholeRestoreBytes is how many bytes the file half would ADD to the live drive (rule 3 and rule 4
|
||||
// copies — rule 4 keeps the old copy, so it adds its full size). Read-only.
|
||||
func planWholeRestoreBytes(src, dst string) (int64, error) {
|
||||
var need int64
|
||||
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
|
||||
if walkErr != nil || d.IsDir() {
|
||||
return walkErr
|
||||
}
|
||||
fi, err := d.Info()
|
||||
if err != nil || !fi.Mode().IsRegular() {
|
||||
return err
|
||||
}
|
||||
rel, _ := filepath.Rel(src, p)
|
||||
live, err := os.Lstat(filepath.Join(dst, rel))
|
||||
if err != nil || live.Size() != fi.Size() || !live.ModTime().After(fi.ModTime()) {
|
||||
need += fi.Size()
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return need, err
|
||||
}
|
||||
|
||||
// wholeRestoreFloorBytes is the free space the live drive must keep after the file half. The same 2 GB
|
||||
// the update keeps on the Docker root (stacks.updateDiskFloorGiB).
|
||||
const wholeRestoreFloorBytes = int64(2) << 30
|
||||
|
||||
var (
|
||||
// ErrWholeRestoreNotFileApp — the app keeps no files on the drive; its whole restore is the unit
|
||||
// restore itself („Teljes visszaállítás").
|
||||
ErrWholeRestoreNotFileApp = util.MsgError("err.backup.whole_restore_not_file_app")
|
||||
// ErrWholeRestoreNoProvenCopy — the mirror's unit is not a proven, openable package, or the mirror
|
||||
// carries no file leg: nothing moves.
|
||||
ErrWholeRestoreNoProvenCopy = util.MsgError("err.backup.whole_restore_no_proven_copy")
|
||||
)
|
||||
|
||||
// wholeRestoreSpace is the refusal when the file half would leave less than the floor free.
|
||||
type wholeRestoreSpace struct{ need, free int64 }
|
||||
|
||||
func (e *wholeRestoreSpace) Error() string {
|
||||
return util.Text("hu", "err.backup.whole_restore_space", float64(e.need)/(1<<30), float64(e.free)/(1<<30))
|
||||
}
|
||||
|
||||
// WholeRestoreResult is what a whole restore returns.
|
||||
type WholeRestoreResult struct {
|
||||
Files WholeRestoreCounts
|
||||
Unit UnitRestoreResult
|
||||
}
|
||||
|
||||
// RestoreTier2Whole brings a FILE app back whole from its recorded Tier-2 copy: the drive files by the
|
||||
// four rules, then the unit (settings + database) — with the app stopped throughout. Every refusal
|
||||
// happens before anything moves.
|
||||
func (m *Manager) RestoreTier2Whole(stackName string) (WholeRestoreResult, error) {
|
||||
var res WholeRestoreResult
|
||||
if m.stackProvider == nil {
|
||||
return res, fmt.Errorf("stack provider not configured")
|
||||
}
|
||||
if !m.HasDriveFileLegs(stackName) {
|
||||
return res, ErrWholeRestoreNotFileApp
|
||||
}
|
||||
destBase, err := m.tier2RecordedCopyDir(stackName)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
cov := tier2CoverageAt(destBase)
|
||||
rp, rerr := m.Tier2UnitRestorePoint(stackName)
|
||||
_, proven := rp.ProvenCopyTime()
|
||||
if rerr != nil || !cov.CanRestoreUnit() || !cov.CanRestore() || !proven {
|
||||
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: unit_openable=%v legs=%v proven=%v (%v) — nothing moved",
|
||||
stackName, cov.CanRestoreUnit(), cov.Legs, proven, rerr)
|
||||
return res, ErrWholeRestoreNoProvenCopy
|
||||
}
|
||||
drive := m.GetAppDrivePath(stackName)
|
||||
if drive == "" || !filepath.IsAbs(drive) {
|
||||
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
||||
}
|
||||
if m.settings != nil && (m.settings.IsDisconnected(drive) || m.settings.IsDecommissioned(drive)) {
|
||||
return res, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
|
||||
}
|
||||
liveNsRoot := m.namespaceRoot(drive)
|
||||
merges := []struct{ src, dst string }{
|
||||
{filepath.Join(destBase, "hdd"), liveNsRoot},
|
||||
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
|
||||
}
|
||||
var need int64
|
||||
for _, mg := range merges {
|
||||
if _, err := os.Stat(mg.src); err != nil {
|
||||
continue
|
||||
}
|
||||
n, err := planWholeRestoreBytes(mg.src, mg.dst)
|
||||
if err != nil {
|
||||
return res, fmt.Errorf("planning the file half: %w", err)
|
||||
}
|
||||
need += n
|
||||
}
|
||||
free := m.freeBytes(liveNsRoot)
|
||||
if free >= 0 && free-need < wholeRestoreFloorBytes {
|
||||
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: the file half needs %d B and %d B is free (floor %d B) — nothing moved", stackName, need, free, wholeRestoreFloorBytes)
|
||||
return res, &wholeRestoreSpace{need: need, free: free}
|
||||
}
|
||||
|
||||
// THE FILE HALF, under the backup single-flight, with the app stopped.
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
return res, err
|
||||
}
|
||||
m.logger.Printf("[WARN] [backup] WHOLE restore for %s from the second drive %s: files by the four rules (never delete, never overwrite newer), then the unit", stackName, destBase)
|
||||
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
|
||||
m.logger.Printf("[WARN] [backup] could not stop %s before the whole restore: %v (continuing)", stackName, stopErr)
|
||||
}
|
||||
ts := time.Now()
|
||||
var mergeErr error
|
||||
for _, mg := range merges {
|
||||
if _, err := os.Stat(mg.src); err != nil {
|
||||
continue
|
||||
}
|
||||
c, err := mergeRestoreFiles(mg.src, mg.dst, ts)
|
||||
res.Files.Restored += c.Restored
|
||||
res.Files.Replaced += c.Replaced
|
||||
res.Files.KeptNewer += c.KeptNewer
|
||||
res.Files.Unchanged += c.Unchanged
|
||||
res.Files.BytesTotal += c.BytesTotal
|
||||
if err != nil {
|
||||
mergeErr = err
|
||||
break
|
||||
}
|
||||
}
|
||||
m.releaseRunning()
|
||||
m.logger.Printf("[INFO] [backup] whole restore %s: files restored=%d replaced=%d (live kept beside as *%s) kept-newer=%d unchanged=%d",
|
||||
stackName, res.Files.Restored, res.Files.Replaced, wholeRestoreSuffix(ts), res.Files.KeptNewer, res.Files.Unchanged)
|
||||
if mergeErr != nil {
|
||||
// The unit is NOT replayed over a half-merged file tree; nothing was deleted, the app is restarted.
|
||||
if startErr := m.stackProvider.StartStack(stackName); startErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] restarting %s after a failed file half also failed: %v", stackName, startErr)
|
||||
}
|
||||
return res, util.MsgError("err.backup.fajlmasolas_sikertelen", mergeErr)
|
||||
}
|
||||
|
||||
// THE UNIT HALF — the existing restore, told that the files are handled (R-538 stays for every other caller).
|
||||
unitDir := tier2UnitDir(destBase)
|
||||
res.Unit, err = m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{AcceptMissingFiles: true})
|
||||
m.rehydratePrimaryUnit(stackName, unitDir, err)
|
||||
return res, err
|
||||
}
|
||||
|
||||
// freeBytes is the free space under p, -1 when unknown. A seam for tests.
|
||||
func (m *Manager) freeBytes(p string) int64 {
|
||||
if m.freeBytesFn != nil {
|
||||
return m.freeBytesFn(p)
|
||||
}
|
||||
var st syscall.Statfs_t
|
||||
if err := syscall.Statfs(p, &st); err != nil {
|
||||
return -1
|
||||
}
|
||||
return int64(st.Bavail) * int64(st.Bsize)
|
||||
}
|
||||
Reference in New Issue
Block a user