controller v0.269.0: whole restore from the second drive; crash loops stopped; exact image digests; steps judged by their own .felhom.yml (decisions 26-28, R-661 R-666 R-667 R-668 R-664 R-665 R-662, 09 6.4 part 6)
gates / gates (push) Successful in 27s
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -181,6 +181,8 @@ type Manager struct {
|
||||
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
|
||||
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
|
||||
updateOffsiteTimesFn func(ctx context.Context) (map[string]time.Time, error)
|
||||
// freeBytesFn (v0.269.0, decision 26): the whole restore's disk-floor check; nil → statfs.
|
||||
freeBytesFn func(path string) int64
|
||||
|
||||
// R-354 volume-REPLAY seam — the mirror of the F17 DB seams above, so the off-site path's new
|
||||
// volume leg is unit-testable without Docker. Nil → the real restoreDockerVolumesFrom.
|
||||
@@ -1264,6 +1266,18 @@ func (m *Manager) sameDevice(a, b string) bool {
|
||||
if m.samePhysicalDevice != nil {
|
||||
return m.samePhysicalDevice(a, b)
|
||||
}
|
||||
// R-668 (v0.269.0): FAIL CLOSED. system.SamePhysicalDevice answers "different" when either path cannot
|
||||
// be stat'd, and a Tier-2 target chosen on that answer is written wherever the path gets re-created —
|
||||
// measured on 9202 2026-09-24: a removed folder on nextcloud's OWN disk became its "second drive". A
|
||||
// path we cannot read is treated as the SAME disk: no second copy is claimed on it.
|
||||
if _, err := os.Stat(a); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] same-disk check: %s cannot be read (%v) — treated as the same disk", a, err)
|
||||
return true
|
||||
}
|
||||
if _, err := os.Stat(b); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] same-disk check: %s cannot be read (%v) — treated as the same disk", b, err)
|
||||
return true
|
||||
}
|
||||
return system.SamePhysicalDevice(a, b)
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// Decision 28: the stop is a hold (same store — no start path revives it), Start lifts ONLY that kind,
|
||||
// and a second stop within 24 h says support is informed.
|
||||
func TestD28_HoldTripsAndLift(t *testing.T) {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
t0 := time.Date(2026, 9, 24, 22, 0, 0, 0, time.UTC)
|
||||
trip, err := m.HoldUnhealthy("gokapi", "crash_loop", t0)
|
||||
if err != nil || trip != 1 {
|
||||
t.Fatalf("trip=%d err=%v", trip, err)
|
||||
}
|
||||
held, why := m.RestoreHoldForLang("gokapi", "hu")
|
||||
if !held || !strings.Contains(why, "Indítás gombbal") || strings.Contains(why, "gyfélszolgálat") {
|
||||
t.Fatalf("first stop: held=%v %q", held, why)
|
||||
}
|
||||
if !m.UpdateHeldStacks()["gokapi"] {
|
||||
t.Fatal("an unhealthy stop is a product stop — it must be in the app-down suppression set")
|
||||
}
|
||||
if m.HoldKind("gokapi") != settings.HoldReasonUnhealthyStop || !m.LiftUnhealthyStop("gokapi") {
|
||||
t.Fatal("Start could not lift the unhealthy stop")
|
||||
}
|
||||
if held, _ := m.RestoreHoldFor("gokapi"); held {
|
||||
t.Fatal("still held after Start")
|
||||
}
|
||||
trip, _ = m.HoldUnhealthy("gokapi", "oom_storm", t0.Add(3*time.Hour))
|
||||
_, why = m.RestoreHoldForLang("gokapi", "en")
|
||||
if trip != 2 || !strings.Contains(why, "support has been told") {
|
||||
t.Fatalf("second stop within 24 h: trip=%d %q", trip, why)
|
||||
}
|
||||
m.LiftUnhealthyStop("gokapi")
|
||||
if trip, _ = m.HoldUnhealthy("gokapi", "crash_loop", t0.Add(30*time.Hour)); trip != 1 {
|
||||
t.Fatalf("a stop 27 h after the last is a first stop again, got trip %d", trip)
|
||||
}
|
||||
// Start never lifts another kind.
|
||||
_ = sett.SetRestoreHold(settings.RestoreHold{Stack: "nextcloud", At: t0.Format(time.RFC3339), Reason: settings.HoldReasonUpdateFailed})
|
||||
if m.LiftUnhealthyStop("nextcloud") {
|
||||
t.Fatal("Start lifted an UPDATE hold")
|
||||
}
|
||||
}
|
||||
@@ -346,6 +346,12 @@ func (m *Manager) RestoreHoldForLang(stack, lang string) (bool, string) {
|
||||
}
|
||||
// Slice 4: one storage, two reasons. An update hold names the copy it can be restored from; a
|
||||
// restore hold names nothing, because the restore it refers to already consumed the copy.
|
||||
if h.Reason == settings.HoldReasonUnhealthyStop { // v0.269.0, decision 28
|
||||
if h.Trip >= 2 {
|
||||
return true, util.Text(lang, "hold.unhealthy_stop.repeat")
|
||||
}
|
||||
return true, util.Text(lang, "hold.unhealthy_stop")
|
||||
}
|
||||
if h.Reason == settings.HoldReasonUpdateFailed {
|
||||
if h.NoWholeCopy { // R-659: the sentence carries the undo's failure itself and names no copy
|
||||
return true, util.Text(lang, "hold.update.no_whole_copy", stack)
|
||||
|
||||
@@ -51,6 +51,10 @@ func r102Tier2Fixture(t *testing.T, volTars []string, dbDump string) *r102T2 {
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// A registered drive is a directory that EXISTS (R-668, v0.269.0: a missing one is no candidate).
|
||||
if err := os.MkdirAll(dest, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: dest, Label: "flash", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
@@ -123,11 +123,11 @@ func TestR659_SentenceBothLanguages(t *testing.T) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, en := m.RestoreHoldForLang("nextcloud", "en")
|
||||
if en != "The update of nextcloud did not work, and neither did the automatic undo. This box has no copy that can bring the app back together with its files. Felhom support has been told — until then, do not restart or remove the app." {
|
||||
if en != "The update of nextcloud did not work, and neither did the automatic undo. This box has no copy that can bring the app back together with its files. Felhom support has been told — until then, do not restart the app or delete its data." {
|
||||
t.Fatalf("en = %q", en)
|
||||
}
|
||||
_, hu := m.RestoreHoldForLang("nextcloud", "hu")
|
||||
if hu != "A(z) nextcloud frissítése nem sikerült, és az automatikus visszaállítás sem. Ezen a dobozon nincs olyan másolat, amely az alkalmazást a fájljaival együtt vissza tudná hozni. A Felhom ügyfélszolgálatát értesítettük — kérjük, addig ne indítsa újra és ne törölje az alkalmazást." {
|
||||
if hu != "A(z) nextcloud frissítése nem sikerült, és az automatikus visszaállítás sem. Ezen a dobozon nincs olyan másolat, amely az alkalmazást a fájljaival együtt vissza tudná hozni. A Felhom ügyfélszolgálatát értesítettük — addig ne indítsd újra, és ne töröld az adatait." {
|
||||
t.Fatalf("hu = %q", hu)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-668 (v0.269.0) — a registered storage path whose folder does not exist is NOT a second drive.
|
||||
// Measured on 9202 2026-09-24: `scratch_hdd/userdata/romm` (removed that morning) was chosen as
|
||||
// nextcloud's Tier-2 target — the stat in the same-disk check failed, which read as "another disk" — and
|
||||
// the copy was written into a re-created folder on nextcloud's OWN disk. Runs the PRODUCTION same-disk
|
||||
// check (no seam).
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT): remove the fail-closed stats in Manager.sameDevice — this test then fails
|
||||
// at "chose the missing path".
|
||||
func TestR668_MissingStoragePathIsNotASecondDrive(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
live := filepath.Join(tmp, "drive")
|
||||
missing := filepath.Join(tmp, "drive-gone")
|
||||
if err := os.MkdirAll(live, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sett, err := settings.Load(filepath.Join(tmp, "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: missing, Label: "gone", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett, systemDataPath: filepath.Join(tmp, "no-sys")}
|
||||
tgt, err := m.selectTier2TargetFrom("app", live, 1, 1)
|
||||
if err == nil && tgt != nil && tgt.NamespaceRoot == NamespaceRoot(missing, true) {
|
||||
t.Fatalf("chose the missing path %s as the second drive", missing)
|
||||
}
|
||||
if !m.sameDevice(live, missing) {
|
||||
t.Fatal("an unreadable path must count as the SAME disk")
|
||||
}
|
||||
}
|
||||
@@ -21,7 +21,12 @@ func newTestManager(t *testing.T, systemDataPath string) (*Manager, *settings.Se
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.SystemDataPath = systemDataPath
|
||||
return NewManager(cfg, sett, logger), sett
|
||||
m := NewManager(cfg, sett, logger)
|
||||
// These tests name paths that do not exist (/mnt/a, /srv/sys). Until v0.269.0 they passed only because
|
||||
// the production same-disk check answered "different" on a failed stat — the R-668 defect. They now
|
||||
// say what they mean: every distinct path is its own disk.
|
||||
m.samePhysicalDevice = func(a, b string) bool { return a == b }
|
||||
return m, sett
|
||||
}
|
||||
|
||||
// A customer-pinned PreferredTarget must win over the auto-pick (which would take the first
|
||||
|
||||
@@ -0,0 +1,353 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
)
|
||||
|
||||
// ── The WHOLE restore from the second drive (`09` §3 decision 26, R-661; controller v0.269.0) ────
|
||||
//
|
||||
// WHY IT EXISTS. For an app whose files live on the data drive (DeclaredDriveFileLegs — nextcloud,
|
||||
// immich, paperless-ngx, calibre-web) the Tier-2 mirror holds BOTH halves: the recovery unit (settings +
|
||||
// database) and the drive files (hdd/<rel>, userdata/<rel>). Until v0.269.0 no single action brought
|
||||
// the app back from it: the unit restore refuses such an app (R-538 — it will not replay a database over
|
||||
// files it cannot also restore), and the file restore only ADDS missing files and replays no database.
|
||||
// Measured from source 2026-09-24. So after a failed update and a failed undo, a box with a second drive
|
||||
// but no off-site tier still stranded the app (R-659's case). The operator ruled (decision 26): one
|
||||
// action brings the app back WHOLE from the second drive. R-538's guard STAYS — this is a new action
|
||||
// beside it; it passes AcceptMissingFiles only because it has just put the files back itself.
|
||||
//
|
||||
// THE FILE RULES, in this order of importance (the operator's brief, verbatim in spirit):
|
||||
//
|
||||
// 1. NEVER DELETE a live file. Nothing here removes anything the household has.
|
||||
// 2. NEVER OVERWRITE a file whose live copy is NEWER than the mirror's (mtime).
|
||||
// 3. BRING BACK every file that is missing.
|
||||
// 4. A file that DIFFERS and is OLDER (or equally old) on the live side is replaced from the mirror, and
|
||||
// the live copy is KEPT BESIDE it as `<name>.felhom-<ts>`.
|
||||
//
|
||||
// rsyncMirror carries --delete and is NEVER used here; rsyncRestoreMissing is additive but cannot do
|
||||
// rule 4 or count it, so the merge is done file by file, and every decision is counted.
|
||||
|
||||
// WholeRestoreCounts is what the whole restore did to the drive files, per rule.
|
||||
type WholeRestoreCounts struct {
|
||||
Restored int // rule 3: missing live, copied back
|
||||
Replaced int // rule 4: live older + different → mirror copy in place, live kept beside
|
||||
KeptNewer int // rule 2: live newer → left alone
|
||||
Unchanged int // same content → nothing to do
|
||||
BytesTotal int64
|
||||
}
|
||||
|
||||
// wholeRestoreSuffix names the copy rule 4 keeps beside a replaced file.
|
||||
func wholeRestoreSuffix(ts time.Time) string { return ".felhom-" + ts.UTC().Format("20060102T150405Z") }
|
||||
|
||||
func sameContent(a, b string) (bool, error) {
|
||||
ha, err := fileSHA(a)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
hb, err := fileSHA(b)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return bytes.Equal(ha, hb), nil
|
||||
}
|
||||
|
||||
func fileSHA(p string) ([]byte, error) {
|
||||
f, err := os.Open(p)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
h := sha256.New()
|
||||
if _, err := io.Copy(h, f); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return h.Sum(nil), nil
|
||||
}
|
||||
|
||||
// copyFilePreserving copies src to dst (which must not exist), keeping mode, mtime and — when possible —
|
||||
// the owner, so a restored file is readable by the app exactly as the mirrored one was.
|
||||
func copyFilePreserving(src, dst string, fi fs.FileInfo) error {
|
||||
in, err := os.Open(src)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer in.Close()
|
||||
tmp := dst + ".felhom-restoring"
|
||||
out, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_EXCL, fi.Mode().Perm())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := io.Copy(out, in); err != nil {
|
||||
out.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := out.Sync(); err != nil {
|
||||
out.Close()
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if err := out.Close(); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
||||
_ = os.Lchown(tmp, int(st.Uid), int(st.Gid))
|
||||
}
|
||||
_ = os.Chmod(tmp, fi.Mode().Perm())
|
||||
_ = os.Chtimes(tmp, fi.ModTime(), fi.ModTime())
|
||||
// Link, not rename: rename would silently REPLACE a dst that appeared meanwhile (rule 1/2).
|
||||
if err := os.Link(tmp, dst); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Remove(tmp)
|
||||
}
|
||||
|
||||
// mergeRestoreFiles applies the four rules from the mirror subtree src onto the live subtree dst. It
|
||||
// never deletes and never follows a symlink out of either tree. Directories missing live are created
|
||||
// with the mirror's mode and owner.
|
||||
func mergeRestoreFiles(src, dst string, ts time.Time) (WholeRestoreCounts, error) {
|
||||
var c WholeRestoreCounts
|
||||
suffix := wholeRestoreSuffix(ts)
|
||||
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
|
||||
if walkErr != nil {
|
||||
return walkErr
|
||||
}
|
||||
rel, err := filepath.Rel(src, p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
target := filepath.Join(dst, rel)
|
||||
fi, err := d.Info()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
switch {
|
||||
case d.IsDir():
|
||||
if _, err := os.Lstat(target); errors.Is(err, fs.ErrNotExist) {
|
||||
if err := os.MkdirAll(target, fi.Mode().Perm()); err != nil {
|
||||
return err
|
||||
}
|
||||
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
|
||||
_ = os.Lchown(target, int(st.Uid), int(st.Gid))
|
||||
}
|
||||
}
|
||||
return nil
|
||||
case !fi.Mode().IsRegular():
|
||||
return nil // symlinks, sockets, devices: never materialised by a restore
|
||||
}
|
||||
c.BytesTotal += fi.Size()
|
||||
live, err := os.Lstat(target)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
if err := copyFilePreserving(p, target, fi); err != nil {
|
||||
return fmt.Errorf("restoring %s: %w", rel, err)
|
||||
}
|
||||
c.Restored++
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !live.Mode().IsRegular() {
|
||||
c.KeptNewer++ // something else lives there now (a dir, a link): never touched (rule 1)
|
||||
return nil
|
||||
}
|
||||
if live.Size() == fi.Size() {
|
||||
same, err := sameContent(p, target)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if same {
|
||||
c.Unchanged++
|
||||
return nil
|
||||
}
|
||||
}
|
||||
if live.ModTime().After(fi.ModTime()) {
|
||||
c.KeptNewer++ // rule 2: the household's newer copy wins
|
||||
return nil
|
||||
}
|
||||
// rule 4: keep the live copy beside, then put the mirror's in place.
|
||||
kept := target + suffix
|
||||
if err := os.Link(target, kept); err != nil {
|
||||
return fmt.Errorf("keeping the live copy of %s: %w", rel, err)
|
||||
}
|
||||
if err := os.Remove(target); err != nil {
|
||||
return fmt.Errorf("moving the live copy of %s aside: %w", rel, err)
|
||||
}
|
||||
if err := copyFilePreserving(p, target, fi); err != nil {
|
||||
// put the live copy back where it was; the kept link stays either way
|
||||
_ = os.Link(kept, target)
|
||||
return fmt.Errorf("replacing %s: %w", rel, err)
|
||||
}
|
||||
c.Replaced++
|
||||
return nil
|
||||
})
|
||||
return c, err
|
||||
}
|
||||
|
||||
// planWholeRestoreBytes is how many bytes the file half would ADD to the live drive (rule 3 and rule 4
|
||||
// copies — rule 4 keeps the old copy, so it adds its full size). Read-only.
|
||||
func planWholeRestoreBytes(src, dst string) (int64, error) {
|
||||
var need int64
|
||||
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, walkErr error) error {
|
||||
if walkErr != nil || d.IsDir() {
|
||||
return walkErr
|
||||
}
|
||||
fi, err := d.Info()
|
||||
if err != nil || !fi.Mode().IsRegular() {
|
||||
return err
|
||||
}
|
||||
rel, _ := filepath.Rel(src, p)
|
||||
live, err := os.Lstat(filepath.Join(dst, rel))
|
||||
if err != nil || live.Size() != fi.Size() || !live.ModTime().After(fi.ModTime()) {
|
||||
need += fi.Size()
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return need, err
|
||||
}
|
||||
|
||||
// wholeRestoreFloorBytes is the free space the live drive must keep after the file half. The same 2 GB
|
||||
// the update keeps on the Docker root (stacks.updateDiskFloorGiB).
|
||||
const wholeRestoreFloorBytes = int64(2) << 30
|
||||
|
||||
var (
|
||||
// ErrWholeRestoreNotFileApp — the app keeps no files on the drive; its whole restore is the unit
|
||||
// restore itself („Teljes visszaállítás").
|
||||
ErrWholeRestoreNotFileApp = util.MsgError("err.backup.whole_restore_not_file_app")
|
||||
// ErrWholeRestoreNoProvenCopy — the mirror's unit is not a proven, openable package, or the mirror
|
||||
// carries no file leg: nothing moves.
|
||||
ErrWholeRestoreNoProvenCopy = util.MsgError("err.backup.whole_restore_no_proven_copy")
|
||||
)
|
||||
|
||||
// wholeRestoreSpace is the refusal when the file half would leave less than the floor free.
|
||||
type wholeRestoreSpace struct{ need, free int64 }
|
||||
|
||||
func (e *wholeRestoreSpace) Error() string {
|
||||
return util.Text("hu", "err.backup.whole_restore_space", float64(e.need)/(1<<30), float64(e.free)/(1<<30))
|
||||
}
|
||||
|
||||
// WholeRestoreResult is what a whole restore returns.
|
||||
type WholeRestoreResult struct {
|
||||
Files WholeRestoreCounts
|
||||
Unit UnitRestoreResult
|
||||
}
|
||||
|
||||
// RestoreTier2Whole brings a FILE app back whole from its recorded Tier-2 copy: the drive files by the
|
||||
// four rules, then the unit (settings + database) — with the app stopped throughout. Every refusal
|
||||
// happens before anything moves.
|
||||
func (m *Manager) RestoreTier2Whole(stackName string) (WholeRestoreResult, error) {
|
||||
var res WholeRestoreResult
|
||||
if m.stackProvider == nil {
|
||||
return res, fmt.Errorf("stack provider not configured")
|
||||
}
|
||||
if !m.HasDriveFileLegs(stackName) {
|
||||
return res, ErrWholeRestoreNotFileApp
|
||||
}
|
||||
destBase, err := m.tier2RecordedCopyDir(stackName)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
cov := tier2CoverageAt(destBase)
|
||||
rp, rerr := m.Tier2UnitRestorePoint(stackName)
|
||||
_, proven := rp.ProvenCopyTime()
|
||||
if rerr != nil || !cov.CanRestoreUnit() || !cov.CanRestore() || !proven {
|
||||
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: unit_openable=%v legs=%v proven=%v (%v) — nothing moved",
|
||||
stackName, cov.CanRestoreUnit(), cov.Legs, proven, rerr)
|
||||
return res, ErrWholeRestoreNoProvenCopy
|
||||
}
|
||||
drive := m.GetAppDrivePath(stackName)
|
||||
if drive == "" || !filepath.IsAbs(drive) {
|
||||
return res, fmt.Errorf("cannot determine drive path for %s", stackName)
|
||||
}
|
||||
if m.settings != nil && (m.settings.IsDisconnected(drive) || m.settings.IsDecommissioned(drive)) {
|
||||
return res, fmt.Errorf("%w (%s)", errLiveDriveGone, drive)
|
||||
}
|
||||
liveNsRoot := m.namespaceRoot(drive)
|
||||
merges := []struct{ src, dst string }{
|
||||
{filepath.Join(destBase, "hdd"), liveNsRoot},
|
||||
{filepath.Join(destBase, "userdata"), filepath.Join(liveNsRoot, "userdata")},
|
||||
}
|
||||
var need int64
|
||||
for _, mg := range merges {
|
||||
if _, err := os.Stat(mg.src); err != nil {
|
||||
continue
|
||||
}
|
||||
n, err := planWholeRestoreBytes(mg.src, mg.dst)
|
||||
if err != nil {
|
||||
return res, fmt.Errorf("planning the file half: %w", err)
|
||||
}
|
||||
need += n
|
||||
}
|
||||
free := m.freeBytes(liveNsRoot)
|
||||
if free >= 0 && free-need < wholeRestoreFloorBytes {
|
||||
m.logger.Printf("[WARN] [backup] whole restore REFUSED for %s: the file half needs %d B and %d B is free (floor %d B) — nothing moved", stackName, need, free, wholeRestoreFloorBytes)
|
||||
return res, &wholeRestoreSpace{need: need, free: free}
|
||||
}
|
||||
|
||||
// THE FILE HALF, under the backup single-flight, with the app stopped.
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
return res, err
|
||||
}
|
||||
m.logger.Printf("[WARN] [backup] WHOLE restore for %s from the second drive %s: files by the four rules (never delete, never overwrite newer), then the unit", stackName, destBase)
|
||||
if stopErr := m.stackProvider.StopStack(stackName); stopErr != nil {
|
||||
m.logger.Printf("[WARN] [backup] could not stop %s before the whole restore: %v (continuing)", stackName, stopErr)
|
||||
}
|
||||
ts := time.Now()
|
||||
var mergeErr error
|
||||
for _, mg := range merges {
|
||||
if _, err := os.Stat(mg.src); err != nil {
|
||||
continue
|
||||
}
|
||||
c, err := mergeRestoreFiles(mg.src, mg.dst, ts)
|
||||
res.Files.Restored += c.Restored
|
||||
res.Files.Replaced += c.Replaced
|
||||
res.Files.KeptNewer += c.KeptNewer
|
||||
res.Files.Unchanged += c.Unchanged
|
||||
res.Files.BytesTotal += c.BytesTotal
|
||||
if err != nil {
|
||||
mergeErr = err
|
||||
break
|
||||
}
|
||||
}
|
||||
m.releaseRunning()
|
||||
m.logger.Printf("[INFO] [backup] whole restore %s: files restored=%d replaced=%d (live kept beside as *%s) kept-newer=%d unchanged=%d",
|
||||
stackName, res.Files.Restored, res.Files.Replaced, wholeRestoreSuffix(ts), res.Files.KeptNewer, res.Files.Unchanged)
|
||||
if mergeErr != nil {
|
||||
// The unit is NOT replayed over a half-merged file tree; nothing was deleted, the app is restarted.
|
||||
if startErr := m.stackProvider.StartStack(stackName); startErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] restarting %s after a failed file half also failed: %v", stackName, startErr)
|
||||
}
|
||||
return res, util.MsgError("err.backup.fajlmasolas_sikertelen", mergeErr)
|
||||
}
|
||||
|
||||
// THE UNIT HALF — the existing restore, told that the files are handled (R-538 stays for every other caller).
|
||||
unitDir := tier2UnitDir(destBase)
|
||||
res.Unit, err = m.RestoreFromRecoveryUnitAtWith(stackName, unitDir, UnitRestoreOptions{AcceptMissingFiles: true})
|
||||
m.rehydratePrimaryUnit(stackName, unitDir, err)
|
||||
return res, err
|
||||
}
|
||||
|
||||
// freeBytes is the free space under p, -1 when unknown. A seam for tests.
|
||||
func (m *Manager) freeBytes(p string) int64 {
|
||||
if m.freeBytesFn != nil {
|
||||
return m.freeBytesFn(p)
|
||||
}
|
||||
var st syscall.Statfs_t
|
||||
if err := syscall.Statfs(p, &st); err != nil {
|
||||
return -1
|
||||
}
|
||||
return int64(st.Bavail) * int64(st.Bsize)
|
||||
}
|
||||
@@ -0,0 +1,233 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// v0.269.0 — the WHOLE restore from the second drive (`09` §3 decision 26, R-661). The file rules are
|
||||
// asserted as CONSEQUENCES over the whole tree: every live file is fingerprinted before and after, and
|
||||
// nothing the household had may disappear or be silently overwritten.
|
||||
|
||||
func writeAt(t *testing.T, p, body string, mt time.Time) {
|
||||
t.Helper()
|
||||
mustWrite(t, p, body)
|
||||
if err := os.Chtimes(p, mt, mt); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func readFile(t *testing.T, p string) string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile(p)
|
||||
if err != nil {
|
||||
t.Fatalf("read %s: %v", p, err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// fingerprint maps every regular file under root to its content.
|
||||
func fingerprint(t *testing.T, root string) map[string]string {
|
||||
t.Helper()
|
||||
out := map[string]string{}
|
||||
_ = filepath.Walk(root, func(p string, fi os.FileInfo, err error) error {
|
||||
if err == nil && fi.Mode().IsRegular() {
|
||||
rel, _ := filepath.Rel(root, p)
|
||||
out[rel] = readFile(t, p)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return out
|
||||
}
|
||||
|
||||
// TestWhole_TheFourFileRules — one mirror, one live tree, every rule exercised at once.
|
||||
//
|
||||
// COMPANION RED-PROOFS (REPORT):
|
||||
// - rule 2 removed (no newer-check): "b.txt, the household's NEWER copy, was overwritten";
|
||||
// - rule 4 without keeping the live copy: "the older live copy of c.txt is gone";
|
||||
// - rule 3 removed: "a.txt, missing live, was not brought back".
|
||||
func TestWhole_TheFourFileRules(t *testing.T) {
|
||||
tmp := t.TempDir()
|
||||
mirror, live := filepath.Join(tmp, "mirror"), filepath.Join(tmp, "live")
|
||||
old := time.Date(2026, 9, 20, 10, 0, 0, 0, time.UTC)
|
||||
mid := time.Date(2026, 9, 22, 10, 0, 0, 0, time.UTC)
|
||||
newer := time.Date(2026, 9, 24, 10, 0, 0, 0, time.UTC)
|
||||
|
||||
writeAt(t, filepath.Join(mirror, "photos", "a.txt"), "A from the copy", mid) // live: missing
|
||||
writeAt(t, filepath.Join(mirror, "photos", "b.txt"), "B from the copy", mid) // live: newer
|
||||
writeAt(t, filepath.Join(mirror, "photos", "c.txt"), "C from the copy", mid) // live: older, different
|
||||
writeAt(t, filepath.Join(mirror, "photos", "d.txt"), "D same", mid) // live: same
|
||||
writeAt(t, filepath.Join(mirror, "docs", "deep", "e.txt"), "E from the copy", mid) // live: whole dir missing
|
||||
writeAt(t, filepath.Join(live, "photos", "b.txt"), "B edited by the household", newer) // newer live
|
||||
writeAt(t, filepath.Join(live, "photos", "c.txt"), "C broken by the update", old) // older live
|
||||
writeAt(t, filepath.Join(live, "photos", "d.txt"), "D same", old) // same content
|
||||
writeAt(t, filepath.Join(live, "photos", "only-live.txt"), "not in the copy", old) // must survive
|
||||
before := fingerprint(t, live)
|
||||
|
||||
ts := time.Date(2026, 9, 24, 21, 0, 0, 0, time.UTC)
|
||||
c, err := mergeRestoreFiles(mirror, live, ts)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
after := fingerprint(t, live)
|
||||
|
||||
// Rule 1 — nothing the household had is gone: every file present before still exists, or (c.txt)
|
||||
// its content survives beside it.
|
||||
kept := "photos/c.txt" + wholeRestoreSuffix(ts)
|
||||
for rel, body := range before {
|
||||
if got, ok := after[rel]; ok && (got == body || rel == "photos/c.txt") {
|
||||
continue
|
||||
}
|
||||
t.Fatalf("live file %s (%q) is gone or changed — rule 1", rel, body)
|
||||
}
|
||||
if after[kept] != "C broken by the update" {
|
||||
t.Fatalf("the older live copy of c.txt is gone — rule 4 must keep it beside as %s; files: %v", kept, keys(after))
|
||||
}
|
||||
// Rule 2
|
||||
if after["photos/b.txt"] != "B edited by the household" {
|
||||
t.Fatalf("b.txt, the household's NEWER copy, was overwritten: %q", after["photos/b.txt"])
|
||||
}
|
||||
// Rule 3
|
||||
if after["photos/a.txt"] != "A from the copy" || after["docs/deep/e.txt"] != "E from the copy" {
|
||||
t.Fatalf("a.txt / e.txt, missing live, were not brought back: %q %q", after["photos/a.txt"], after["docs/deep/e.txt"])
|
||||
}
|
||||
// Rule 4
|
||||
if after["photos/c.txt"] != "C from the copy" {
|
||||
t.Fatalf("c.txt (older, different) was not replaced from the copy: %q", after["photos/c.txt"])
|
||||
}
|
||||
if after["photos/only-live.txt"] != "not in the copy" {
|
||||
t.Fatal("a live file absent from the copy was touched")
|
||||
}
|
||||
if c.Restored != 2 || c.Replaced != 1 || c.KeptNewer != 1 || c.Unchanged != 1 {
|
||||
t.Fatalf("counts = %+v, want restored 2, replaced 1, kept-newer 1, unchanged 1", c)
|
||||
}
|
||||
// mtime of a restored file is the copy's, so a later restore's rule 2 compares like with like.
|
||||
if fi, _ := os.Stat(filepath.Join(live, "photos", "a.txt")); !fi.ModTime().Equal(mid) {
|
||||
t.Fatalf("a.txt mtime %v, want the copy's %v", fi.ModTime(), mid)
|
||||
}
|
||||
// Idempotent: a second run changes nothing and deletes nothing.
|
||||
c2, err := mergeRestoreFiles(mirror, live, ts.Add(time.Hour))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if c2.Restored != 0 || c2.Replaced != 0 {
|
||||
t.Fatalf("a second run moved files again: %+v", c2)
|
||||
}
|
||||
}
|
||||
|
||||
func keys(m map[string]string) []string {
|
||||
var out []string
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// fileAppProvider makes the R-102 fixture's app a FILE app: a mandatory drive leg.
|
||||
type fileAppProvider struct{ *fakeRecoveryProvider }
|
||||
|
||||
func (f fileAppProvider) GetStackClassifiedBinds(string) ([]ClassifiedBind, bool) {
|
||||
return []ClassifiedBind{mandatoryHDD("appdata/app")}, true
|
||||
}
|
||||
|
||||
// wholeFixture: the R-102 fixture (a real RunTier2 mirror + unit restore seams), turned into a file app
|
||||
// with photos in the mirror's hdd leg and a damaged live tree.
|
||||
func wholeFixture(t *testing.T) (*r102T2, string, string) {
|
||||
f := r102Tier2Fixture(t, []string{"app_data.tar"}, pgDump(1))
|
||||
f.m.stackProvider = fileAppProvider{f.fake}
|
||||
mid := time.Date(2026, 9, 22, 10, 0, 0, 0, time.UTC)
|
||||
mirrorLeg := filepath.Join(f.destBase, "hdd", "appdata", "app")
|
||||
liveLeg := filepath.Join(f.liveDrive, "appdata", "app")
|
||||
writeAt(t, filepath.Join(mirrorLeg, "p1.jpg"), "P1", mid)
|
||||
writeAt(t, filepath.Join(mirrorLeg, "p2.jpg"), "P2", mid)
|
||||
writeAt(t, filepath.Join(liveLeg, "p2.jpg"), "P2 newer", mid.Add(48*time.Hour))
|
||||
return f, mirrorLeg, liveLeg
|
||||
}
|
||||
|
||||
// TestWhole_BringsTheFilesAndTheUnitBack — the whole action: files merged by the rules, then the unit
|
||||
// restore FROM THE MIRROR (not the primary), with the app stopped throughout.
|
||||
func TestWhole_BringsTheFilesAndTheUnitBack(t *testing.T) {
|
||||
f, _, liveLeg := wholeFixture(t)
|
||||
res, err := f.m.RestoreTier2Whole("app")
|
||||
if err != nil {
|
||||
t.Fatalf("whole restore: %v", err)
|
||||
}
|
||||
if readFile(t, filepath.Join(liveLeg, "p1.jpg")) != "P1" || readFile(t, filepath.Join(liveLeg, "p2.jpg")) != "P2 newer" {
|
||||
t.Fatal("the file half did not follow the rules")
|
||||
}
|
||||
if res.Files.Restored != 1 || res.Files.KeptNewer != 1 {
|
||||
t.Fatalf("file counts %+v", res.Files)
|
||||
}
|
||||
if len(*f.dbPaths) == 0 || !strings.HasPrefix((*f.dbPaths)[0], f.destBase) {
|
||||
t.Fatalf("the database was not replayed from the MIRROR: %v", *f.dbPaths)
|
||||
}
|
||||
if !f.fake.stopped {
|
||||
t.Fatal("the app was not stopped")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWhole_RefusesBeforeAnythingMoves — no proven copy, no room, not a file app: refused, the app never
|
||||
// stopped, not one file written.
|
||||
func TestWhole_RefusesBeforeAnythingMoves(t *testing.T) {
|
||||
t.Run("no room", func(t *testing.T) {
|
||||
f, _, liveLeg := wholeFixture(t)
|
||||
f.m.freeBytesFn = func(string) int64 { return 1 << 20 }
|
||||
_, err := f.m.RestoreTier2Whole("app")
|
||||
var sp *wholeRestoreSpace
|
||||
if !errors.As(err, &sp) || f.fake.stopped {
|
||||
t.Fatalf("err=%v stopped=%v — want the space refusal before the stop", err, f.fake.stopped)
|
||||
}
|
||||
if _, e := os.Stat(filepath.Join(liveLeg, "p1.jpg")); e == nil {
|
||||
t.Fatal("a file was written by a refused restore")
|
||||
}
|
||||
})
|
||||
t.Run("unit not openable", func(t *testing.T) {
|
||||
f, _, _ := wholeFixture(t)
|
||||
if err := os.Remove(UnitManifestFile(tier2UnitDir(f.destBase))); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := f.m.RestoreTier2Whole("app"); !errors.Is(err, ErrWholeRestoreNoProvenCopy) || f.fake.stopped {
|
||||
t.Fatalf("err=%v stopped=%v", err, f.fake.stopped)
|
||||
}
|
||||
})
|
||||
t.Run("not a file app", func(t *testing.T) {
|
||||
f := r102Tier2Fixture(t, []string{"app_data.tar"}, pgDump(1))
|
||||
if _, err := f.m.RestoreTier2Whole("app"); !errors.Is(err, ErrWholeRestoreNotFileApp) || f.fake.stopped {
|
||||
t.Fatalf("err=%v stopped=%v", err, f.fake.stopped)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestWhole_R538StillRefusesTheUnitOnlyRestore — decision 26 is a NEW action beside R-538, not its
|
||||
// removal: the unit-only restore of a file app (own unit AND the second drive's unit) still refuses.
|
||||
func TestWhole_R538StillRefusesTheUnitOnlyRestore(t *testing.T) {
|
||||
f, _, _ := wholeFixture(t)
|
||||
_, err := f.m.RestoreTier2Unit("app")
|
||||
var refusal *ErrUnitLacksFileLegs
|
||||
if !errors.As(err, &refusal) {
|
||||
t.Fatalf("the unit-only restore of a file app must still refuse (R-538): %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWhole_SecondDriveIsWholeForAFileAppWithAMirror — decision 25's table gains the second drive.
|
||||
func TestWhole_SecondDriveIsWholeForAFileAppWithAMirror(t *testing.T) {
|
||||
f, _, _ := wholeFixture(t)
|
||||
if !f.m.WholeOnTier("app", UpdateTierSecondDrive) {
|
||||
t.Fatal("a file app with a mirrored unit AND file legs must be whole on the second drive")
|
||||
}
|
||||
if f.m.WholeOnTier("app", UpdateTierLocal) {
|
||||
t.Fatal("the own unit is still NOT whole for a file app")
|
||||
}
|
||||
if err := os.RemoveAll(filepath.Join(f.destBase, "hdd")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if f.m.WholeOnTier("app", UpdateTierSecondDrive) {
|
||||
t.Fatal("a mirror without the file leg is not whole")
|
||||
}
|
||||
}
|
||||
@@ -620,7 +620,9 @@ func (m *Manager) UpdateHeldStacks() map[string]bool {
|
||||
}
|
||||
var out map[string]bool
|
||||
for _, h := range m.settings.ListRestoreHolds() {
|
||||
if h.Reason != settings.HoldReasonUpdateFailed {
|
||||
// v0.269.0 (decision 28): an app the box stopped for a crash loop / OOM storm is stopped BY THE
|
||||
// PRODUCT and has its own event (app_stopped_unhealthy) — the same class as an update hold.
|
||||
if h.Reason != settings.HoldReasonUpdateFailed && h.Reason != settings.HoldReasonUnhealthyStop {
|
||||
continue
|
||||
}
|
||||
if out == nil {
|
||||
@@ -659,8 +661,16 @@ func (m *Manager) WholeOnTier(stackName string, tier int) bool {
|
||||
switch tier {
|
||||
case UpdateTierOffsite:
|
||||
return true
|
||||
case UpdateTierLocal, UpdateTierSecondDrive:
|
||||
case UpdateTierLocal:
|
||||
return !m.HasDriveFileLegs(stackName)
|
||||
case UpdateTierSecondDrive:
|
||||
if !m.HasDriveFileLegs(stackName) {
|
||||
return true
|
||||
}
|
||||
// v0.269.0 (decision 26): a file app is whole on the second drive when the mirror holds BOTH an
|
||||
// openable unit and its file legs — RestoreTier2Whole brings back both.
|
||||
cov, err := m.Tier2RestoreCoverage(stackName)
|
||||
return err == nil && cov.CanRestoreUnit() && cov.CanRestore()
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -737,3 +747,60 @@ func (m *Manager) UpdateHold(stackName string) (settings.RestoreHold, bool) {
|
||||
}
|
||||
return h, true
|
||||
}
|
||||
|
||||
// ── Decision 28 (v0.269.0): the box stops an app in a crash loop or an out-of-memory storm ─────────
|
||||
|
||||
// UnhealthyRepeatWindow is how soon a second stop counts as a repeat: the sentence then says support is
|
||||
// informed.
|
||||
const UnhealthyRepeatWindow = 24 * time.Hour
|
||||
|
||||
// HoldUnhealthy records that the box stopped `stack` (kind "crash_loop" or "oom_storm") and returns the
|
||||
// trip number: 1, or 2 when the previous stop was less than 24 h ago. Same store as every hold, so no
|
||||
// start path — the boot sweep, the drive gate, the nightly legs — revives it silently.
|
||||
func (m *Manager) HoldUnhealthy(stack, kind string, at time.Time) (int, error) {
|
||||
if m == nil || m.settings == nil {
|
||||
return 0, fmt.Errorf("no settings wired — the unhealthy stop of %s cannot be recorded", stack)
|
||||
}
|
||||
trip := 1
|
||||
if last, ok := m.settings.LastUnhealthyStop(stack); ok && at.Sub(last) < UnhealthyRepeatWindow {
|
||||
trip = 2
|
||||
}
|
||||
h := settings.RestoreHold{Stack: stack, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUnhealthyStop,
|
||||
UnhealthyKind: kind, Trip: trip}
|
||||
if err := m.settings.SetRestoreHold(h); err != nil {
|
||||
return trip, fmt.Errorf("persisting the unhealthy stop of %s: %w", stack, err)
|
||||
}
|
||||
if err := m.settings.RecordUnhealthyStop(stack, at); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] %s: recording the stop time failed: %v", stack, err)
|
||||
}
|
||||
m.logger.Printf("[WARN] [backup] %s is STOPPED by the box: %s (trip %d within %s) — Start gives it one more try (decision 28)", stack, kind, trip, UnhealthyRepeatWindow)
|
||||
return trip, nil
|
||||
}
|
||||
|
||||
// LiftUnhealthyStop is the Start button's half: an unhealthy-stop hold is lifted, any other kind stays.
|
||||
func (m *Manager) LiftUnhealthyStop(stack string) bool {
|
||||
if m == nil || m.settings == nil {
|
||||
return false
|
||||
}
|
||||
ok, err := m.settings.ClearUnhealthyStopHold(stack)
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] lifting the unhealthy stop of %s failed: %v", stack, err)
|
||||
return false
|
||||
}
|
||||
return ok
|
||||
}
|
||||
|
||||
// HoldKind names the kind of hold in force ("" when none): update_failed, unhealthy_stop, or "restore".
|
||||
func (m *Manager) HoldKind(stack string) string {
|
||||
if m == nil || m.settings == nil {
|
||||
return ""
|
||||
}
|
||||
h, ok := m.settings.GetRestoreHold(stack)
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
if h.Reason == "" {
|
||||
return "restore"
|
||||
}
|
||||
return h.Reason
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user