package backup import ( "context" "encoding/json" "fmt" "os" "os/exec" "path/filepath" "strings" "time" ) // Offsite restore rework (Task 3a §7). With mandatory userdata now in snapshots, restore needs three // changes over the old dump-to-rootfs-scratch: // 1. scratch relocated off the ~8 GB guest rootfs onto a data drive, behind a headroom gate (F-A1); // 2. a unit-only DEFAULT restore (`--include `, SP-3.2) — full is a deliberate, // size-gated second action; // 3. place-to-live = a missing-only merge (never --delete) so the SQ3 immich case is restorable // from offsite alone. // ID-first everywhere (§3): `restic stats --tag` is UNPROVEN on 0.14.0, so the size lookup resolves the // snapshot ID via `snapshots latest --tag` and calls `stats `. const ( // offboxUnitOnlyFreeFloor — a unit-only restore needs at least this much free on the scratch drive. // Catalog recovery units are MB–1 GB (SQ4); 2 GiB is a safe floor without a per-snapshot size probe. offboxUnitOnlyFreeFloor = int64(2) << 30 ) // unitOnlyHeadroom is THE unit-only free-space gate, shared by the customer's scratch restore and the // R-87 nightly proof so the two can never disagree about how much room a unit restore needs or about // what the customer is told when there is not enough (REUSE.md's `offsiteNoSpaceMsgFmt` rule). // // FAIL-CLOSED ON AN UNMEASURABLE PROBE. `offboxFree` returns 0 when it cannot read the filesystem, // and `0 < floor` is true, so an unreadable drive REFUSES rather than sailing through — the inverse // of the R-357 shape where `free < need` with `need == 0` made a gate inert. func unitOnlyHeadroom(free int64) error { if free < offboxUnitOnlyFreeFloor { return fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(offboxUnitOnlyFreeFloor), humanizeBytes(free)) } return nil } // SetOffboxFreeFn overrides the restore free-space probe (tests; the Windows go-test host has no df). func (m *Manager) SetOffboxFreeFn(fn func(path string) int64) { m.offboxFreeFn = fn } // WriteScratchMarkerForTest exposes the marker writer to the web package's flow test. Test-only by // name so a production caller reads as obviously wrong: only RestoreOffboxScratch may certify a // scratch, because only it knows whether the download finished. func (m *Manager) WriteScratchMarkerForTest(scratch, snapshotID string, full bool) error { return m.writeScratchMarker(scratch, snapshotID, full) } // SetOffboxLatestSnapshotFn overrides the restic snapshot lookup (tests; no restic needed). See the // field comment on Manager.offboxLatestSnapFn for why this seam exists rather than a code-reading // argument that the R-357 gate sits early enough. func (m *Manager) SetOffboxLatestSnapshotFn(fn func(ctx context.Context, stack string) (string, []string, error)) { m.offboxLatestSnapFn = fn } // SetOffboxFullPlaceCopier overrides the FULL-restore overwrite copier (tests; no rsync needed). func (m *Manager) SetOffboxFullPlaceCopier(fn func(src, dst string) (int, error)) { m.offboxFullPlaceCopier = fn } // SetRollbackImportFn overrides the ROLLBACK's ImportDump (tests; no Docker needed). Separate from // the replay's own import seam on purpose — see the field comment on Manager.rollbackImport. func (m *Manager) SetRollbackImportFn(fn func(ctx context.Context, db DiscoveredDB, dumpPath string) error) { m.rollbackImport = fn } // SetSafetyDumpFn overrides the pre-restore safety dump (tests; no Docker needed). func (m *Manager) SetSafetyDumpFn(fn func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult) { m.safetyDumpFn = fn } // offboxFree returns the free-space probe (nil seam → the real diskFreeBytes). func (m *Manager) offboxFree() func(string) int64 { if m.offboxFreeFn != nil { return m.offboxFreeFn } return diskFreeBytes } // diskFreeBytes returns available bytes on the filesystem holding path (0 on any error). Mirrors // appexport.DiskFree; kept local so the backup package needs no cross-package dependency. func diskFreeBytes(path string) int64 { ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) defer cancel() out, err := exec.CommandContext(ctx, "df", "--output=avail", "-B1", path).Output() if err != nil { return 0 } lines := strings.Split(strings.TrimSpace(string(out)), "\n") if len(lines) < 2 { return 0 } var size int64 fmt.Sscanf(strings.TrimSpace(lines[1]), "%d", &size) return size } // offboxUnitPathOf returns the snapshot path that is the recovery unit for stack (suffix // backups/primary/), or "" if none is present. func offboxUnitPathOf(paths []string, stack string) string { suffix := "/backups/primary/" + stack for _, p := range paths { if strings.HasSuffix(p, suffix) { return p } } return "" } // offboxLatestSnapshot resolves the newest snapshot for stack: its short ID + captured paths, via // `snapshots latest --tag --json`. When the tag spans more than one group (old unit-only shape // + new enlarged shape), it returns the newest by time. func (m *Manager) offboxLatestSnapshot(ctx context.Context, stack string) (id string, paths []string, err error) { if m.offboxLatestSnapFn != nil { return m.offboxLatestSnapFn(ctx, stack) } t := m.settings.GetOffboxTarget() base, env := m.offboxBaseArgs(t) sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout) defer cancel() out, serr := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "latest", "--tag", stack, "--json")...) if serr != nil { return "", nil, fmt.Errorf("offbox snapshots %s: %w: %s", stack, serr, truncate(out)) } var snaps []struct { ShortID string `json:"short_id"` ID string `json:"id"` Time time.Time `json:"time"` Paths []string `json:"paths"` } if json.Unmarshal(out, &snaps) != nil || len(snaps) == 0 { return "", nil, fmt.Errorf("offbox: nincs pillanatkép a(z) %s alkalmazáshoz", stack) } best := 0 for i := 1; i < len(snaps); i++ { if snaps[i].Time.After(snaps[best].Time) { best = i } } id = snaps[best].ShortID if id == "" { id = snaps[best].ID } return id, snaps[best].Paths, nil } // offboxSnapshotSize returns the restore-size (logical bytes) of ONE snapshot via `stats --json` // (default mode — for a single snapshot ID this is exactly that snapshot's on-disk-when-restored size, // the correct headroom meaning; SP-1). ID-first: never `stats --tag` (unproven on 0.14.0). func (m *Manager) offboxSnapshotSize(ctx context.Context, id string) (int64, error) { t := m.settings.GetOffboxTarget() base, env := m.offboxBaseArgs(t) sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout) defer cancel() out, err := m.runner()(sctx, env, append(append([]string{}, base...), "stats", id, "--json")...) if err != nil { return 0, fmt.Errorf("offbox stats %s: %w: %s", id, err, truncate(out)) } var st struct { TotalSize int64 `json:"total_size"` } if json.Unmarshal(out, &st) != nil || st.TotalSize <= 0 { return 0, fmt.Errorf("offbox: a(z) %s pillanatkép mérete ismeretlen", id) } return st.TotalSize, nil } // offboxRestoreScratchDir returns the on-DATA-DRIVE scratch dir for an app's offsite restore // (/backups/offsite-restore/) plus the namespace root (an existing dir, for the free-space // probe). NEVER cfg.Paths.DataDir (the rootfs — the F-A1 filler). App's HDD drive first; else the first // schedulable storage path; else a Hungarian refusal. func (m *Manager) offboxRestoreScratchDir(stack string) (scratch, nsRoot string, err error) { // offsiteRestoreRootFor is THE place `backups/offsite-restore` is spelled (offbox_verify_copies.go) // — the listing/delete surface must resolve byte-identical paths to the ones written here. // unitOnly=false: the CUSTOMER's scratch can hold a full restore (bulk userdata), so it must NOT // fall back to the state-only system disk — see offboxScratchDirIn. return m.offboxScratchDirIn(stack, m.offsiteRestoreRootFor, false) } // offboxProofScratchDir is the R-87 nightly proof's scratch, resolved by the SAME drive-preference // rules and into a DIFFERENT root (`backups/offsite-proof`). // // THE SEPARATE ROOT IS NOT TIDINESS, IT PREVENTS A DELETE. The proof removes its scratch on every // path, including failure. Sharing `backups/offsite-restore/` would mean a nightly background // job deleting the verification copy a CUSTOMER made and is looking at — a poorer actor destroying a // richer one, which is R-403's shape wearing different clothes. A separate root also keeps the proof // copy invisible to `DeleteOffsiteRestoreCopy`, the copy listing and `OffboxFullScratchReady`, so it // can never be offered for placement into a live app. func (m *Manager) offboxProofScratchDir(stack string) (scratch, nsRoot string, err error) { // unitOnly=true: the proof restores ONE recovery unit (`--include `) and deletes it. For a // driveless app that unit already lives permanently on the system data path, so a scratch there is // at most a second copy of something already present — see offboxScratchDirIn. return m.offboxScratchDirIn(stack, m.offsiteProofRootFor, true) } // offboxScratchDirIn holds the drive-preference rules once. `rootFor` chooses WHICH root under the // namespace the scratch lands in; everything else — the network-storage refusal, the ordering, the // R-252 wording — is shared, so the proof path can never drift from the customer path on the parts // that must not differ. // R-414 — THE SYSTEM-DATA FALLBACK, AND WHY IT IS SCOPED BY WHAT IS BEING RESTORED. // // THE GAP. `demo-felhom` has ZERO registered storage paths, so steps (1)-(3) all miss and this // refused. The nightly proof therefore could not run AT ALL on that box — every night, with only a // WARN — and because its error path reaches no verdict, `last_proof_result` stayed ABSENT, which is // also what a controller too old to have the feature sends. The hub could not tell them apart. // // WAS IT MISSED OR DELIBERATE? Established from R-356's own commit (`08eb1a6`, 2026-08-22), whose test // comments say the scratch resolver *"still resolves to the registered storage path … only the // DESTINATION moves"* — i.e. it was OUT OF SCOPE for that change, which was about where restored data // LANDS. It was never ruled out on state-only grounds: the one comment about a `systemDataPath` // fallback belonged to `PlaceOffsiteRestore` and concerned merging bulk USERDATA onto the SSD, and // R-356 deliberately overruled even that. This function's own documented exclusion is // `cfg.Paths.DataDir` — the ROOTFS — which is a different filesystem entirely. // // SO §6.3's [DESIGN] RULE APPLIES, AND IT NOW HAS A FOURTH CONSUMER: "the restore destination is // resolved by the same rule as the capture destination — the drive if the app declares one, the system // data path otherwise." // // BUT THE TWO CALLERS ASK DIFFERENT QUESTIONS, and answering both with one predicate is the R-356 // defect itself. So the fallback is scoped: // // - unitOnly=true (the R-87 proof): may fall back. `07` §7 records as [FACT] that a driveless app's // recovery unit ALREADY sits on `systemDataPath` indefinitely — "the SSD-only system-data // fallback" — and that the same-device placement is "intended, not a defect". The scratch is // bounded by that unit's own size and is deleted on every path. // - unitOnly=false (the customer's scratch): must NOT. A full restore pulls the app's bulk userdata, // and `07` §2.2 makes the internal SSD a STATE-ONLY tier. This is exactly the case the deleted // `PlaceOffsiteRestore` comment worried about, and the R-252 refusal below stays correct for it. // // The headroom gate still applies on the fallback path — it is the caller's `unitOnlyHeadroom`, which // refuses when the floor is not met, so a small system disk is protected by the same floor as a drive. func (m *Manager) offboxScratchDirIn(stack string, rootFor func(string) string, unitOnly bool) (scratch, nsRoot string, err error) { scratchFor := func(root string) (string, string) { return filepath.Join(rootFor(root), stack), m.namespaceRoot(root) } isNet := func(path string) bool { return m.settings != nil && m.settings.IsNetworkStoragePath(path) } // (1) the app's own drive — preferred, but ONLY if it is not NETWORK storage (F-3afix-1). restic // restores uid/gid/setgid fully onto a LOCAL fs (SP-3.3); a squashed network scratch would feed // PlaceOffsiteRestore wrong-owner files — the F-6C-1 silently-broken-restore class, offsite-side. if m.stackProvider != nil { if hdd := strings.TrimSpace(m.stackProvider.GetStackHDDPath(stack)); hdd != "" && !isNet(hdd) { s, nr := scratchFor(hdd) return s, nr, nil } } // (2) the first NON-network schedulable path. if m.settings != nil { for _, sp := range m.settings.GetSchedulableStoragePaths() { if strings.TrimSpace(sp.Path) != "" && !sp.IsNetwork() { s, nr := scratchFor(sp.Path) return s, nr, nil } } // (3) last resort ONLY: any schedulable path, with a loud WARN — a network scratch cannot // guarantee ownership fidelity under root_squash. for _, sp := range m.settings.GetSchedulableStoragePaths() { if strings.TrimSpace(sp.Path) != "" { m.logger.Printf("[WARN] [offbox] %s: restore scratch on network storage %s — ownership fidelity not guaranteed under squash", stack, sp.Path) s, nr := scratchFor(sp.Path) return s, nr, nil } } } // (4) R-414: a UNIT-ONLY restore falls back to the system data path, which is where a driveless // app's unit already lives. Deliberately AFTER the network last-resort: a registered drive, // even a network one, is still a better scratch for ownership fidelity than the system disk. if unitOnly { if sysPath := strings.TrimSpace(m.cfg.Paths.SystemDataPath); sysPath != "" { s, nr := scratchFor(sysPath) m.logger.Printf("[INFO] [offbox] %s: no registered data drive — unit-only scratch falls back to the system data path %s (R-414; the unit already lives there)", stack, sysPath) return s, nr, nil } } // R-252: name the reason AND the way to act on it. This refusal is what a rebuilt box hits — the // drives are physically fine and still mounted, it is their REGISTRATION that the destroyed guest // took with it — and until v0.207.0 it said only that a drive was missing, which reads like data // loss and offers nothing to do. return "", "", fmt.Errorf("nincs regisztrált adatmeghajtó, ezért nincs hová visszaállítani — " + "a meghajtók megvannak, csak újra kell csatolni őket a Tárhely → Meghajtók oldalon, utána " + "ez a visszaállítás működni fog") } // HasRestoreDestination reports whether an offsite restore has anywhere on this box to write. // // R-252: the restore PAGE asks this question through the same helper the resolver answers it with, // so the notice cannot appear on a box that would restore fine (Scenario E) nor stay hidden on one // that would refuse. A second copy of the predicate is exactly how a page ends up promising what the // handler then refuses — which is the neighbouring defect, R-253. // // It mirrors the resolver's BOX-level branches (2) and (3) — the schedulable storage paths. Branch // (1), the app's own HDD path, is deliberately not consulted: an installed app's HDD path IS a // registered storage path, so the two cannot disagree in practice, and where they could, erring // toward showing the notice is erring toward telling the customer something true. func (m *Manager) HasRestoreDestination() bool { if m.settings == nil { return false } for _, sp := range m.settings.GetSchedulableStoragePaths() { if strings.TrimSpace(sp.Path) != "" { return true } } return false } // RestoreOffboxScratch restores an app's latest offsite snapshot to an on-data-drive scratch dir // (non-destructive — never overwrites live data). full=false (the default) restores the recovery UNIT // only (`--include `, SP-3.2); full=true restores the whole snapshot (unit + // mandatory userdata) behind a size×1.1 headroom gate. Fail-closed: an unknown snapshot size refuses a // full restore. func (m *Manager) RestoreOffboxScratch(ctx context.Context, stack string, full bool) error { if !m.OffboxConfigured() { return fmt.Errorf("off-box backup not configured") } // R-411/R-408 — THE SINGLE-WRITER FLAG, and it must be taken HERE, before anything touches the // repository. // // WHAT IT COSTS TO OMIT IT, measured on demo-hp 2026-08-31 and not reasoned about: this function // runs `offboxSnapshotSize` for a full restore, which shells `restic stats` — and **`stats` TAKES // A REPOSITORY LOCK** (clean-room test: nothing else running, four invocations, the sampler reads // `locks=1`). Without this flag the integrity check is not blocked, starts, meets that lock, and // `resticStep` escalates to `unlock --remove-all` — the argv sampler caught `restore …` and // `unlock --remove-all` in the SAME sample at 20:50:51 — while logging *"a stale exclusive lock // left by a previous crash"*. There was no crash. `resticStep`'s own safety argument is that the // in-process mutex proves no sibling is live; this is the caller that made that false. // // BEFORE the snapshot lookup and the size probe, deliberately: a flag taken after the probe // protects nothing, because the probe is what takes the lock. // // The refusal shape matches the five siblings, so the handler's Hungarian wording is unchanged and // `restoreOpBlocked()` still refuses a second press exactly as it does today. if err := m.acquireRunning(); err != nil { return err } defer m.releaseRunning() if !isSafeStackName(stack) { return fmt.Errorf("invalid stack name") } id, paths, err := m.offboxLatestSnapshot(ctx, stack) if err != nil { return err } unitPath := offboxUnitPathOf(paths, stack) if unitPath == "" { return fmt.Errorf("a(z) %s pillanatképében nincs mentési egység — a visszaállítás nem indítható", stack) } scratch, nsRoot, err := m.offboxRestoreScratchDir(stack) if err != nil { return err } // Headroom gate (F-A1) — probed on the namespace root (an existing dir). free := m.offboxFree()(nsRoot) if full { size, serr := m.offboxSnapshotSize(ctx, id) if serr != nil { // SizeUnknown never renders as fits — fail closed. return fmt.Errorf(offsiteSizeUnknownMsg) } need := size + size/10 // ×1.1 if free < need { return fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free)) } } else if herr := unitOnlyHeadroom(free); herr != nil { return herr } // F-A1 hygiene: drop the legacy rootfs scratch (DataDir/offbox-restore/) best-effort. legacy := filepath.Join(m.cfg.Paths.DataDir, "offbox-restore", stack) if _, sErr := os.Stat(legacy); sErr == nil { if rmErr := os.RemoveAll(legacy); rmErr != nil { m.logger.Printf("[WARN] [offbox] could not remove legacy rootfs restore scratch %s: %v", legacy, rmErr) } else { m.logger.Printf("[INFO] [offbox] removed legacy rootfs restore scratch %s", legacy) } } if err := os.MkdirAll(scratch, 0o755); err != nil { return fmt.Errorf("restore dir: %w", err) } // R-358: a marker from a PREVIOUS run must never certify this one. Cleared here, before restic // touches anything, so the window in which a stale certificate could vouch for a part-copy does not // exist. If this run fails, the scratch is left with files and NO marker — which is precisely the // state OffboxFullScratchReady must read as "not ready". m.clearScratchMarker(scratch) t := m.settings.GetOffboxTarget() base, env := m.offboxBaseArgs(t) rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout) defer cancel() m.unlockStale(rctx, base, env) // pre-restore hygiene args := []string{"restore", id, "--target", scratch} if !full { args = append(args, "--include", unitPath) // SP-3.2: absolute snapshot unit path = unit-only } out, rerr := m.resticStep(rctx, env, base, "restore:"+stack, args...) if rerr != nil { return fmt.Errorf("offbox restore %s: %w: %s", stack, rerr, truncate(out)) } m.logger.Printf("[INFO] [offbox] restored %s (%s, full=%v) → %s", stack, id, full, scratch) // R-358: the completion certificate, written ONLY now — after restic returned nil. Writing it // earlier would certify a download that has not happened, which is the defect with an extra step. // Written for full=false runs too: the `full` field inside it, not its presence, is what // distinguishes a unit-only scratch from a complete one. if err := m.writeScratchMarker(scratch, id, full); err != nil { // The restore itself succeeded, so this is not an error to fail the operation on — but it is // NOT silent, and the consequence is stated: without the marker the scratch reads as not-ready, // which is the fail-closed direction. Better a re-run than a placement over an uncertified copy. m.logger.Printf("[ERROR] [offbox] %s: restore succeeded but the completion marker could not be written: %v — the scratch will read as NOT ready and the download must be re-run", stack, err) } return nil } // --- R-358: the scratch completion marker ------------------------------------------------------ // // THE DEFECT. `OffboxFullScratchReady` used to answer "the directory exists and is non-empty". A restic // download that failed part-way leaves exactly that: a directory with files in it. So the product // offered „Teljes visszaállítás indítása" over a part-copy, and pressing it reported success — // observed on demo-hp 2026-08-21. A non-empty directory is evidence that something was written, never // that everything was. // // The marker is the missing fact: not "are there files" but "did the run that wrote them FINISH, and // was it the full one". Only the run itself can know that, so only the run writes it. // // It lives at the scratch ROOT, which is safe from placement for a reason worth stating rather than // assuming: `mapOffsiteRestorePaths` builds placements from the SNAPSHOT's own path list, not from a // directory walk, so a file that exists only locally is invisible to it. That is pinned by // TestR358_MarkerIsNeverPlaced rather than left as a comment. const scratchMarkerName = ".felhom-restore-complete.json" // scratchMarker is the on-disk completion certificate. `Schema` is carried so a future format change // is a refusal rather than a misreading — an unrecognised schema fails closed like every other // unreadable marker. type scratchMarker struct { Schema int `json:"schema"` SnapshotID string `json:"snapshot_id"` Full bool `json:"full"` FinishedAt string `json:"finished_at"` } const scratchMarkerSchema = 1 // clearScratchMarker removes any existing marker, best-effort. A failure to remove is logged and NOT // returned: the caller is about to overwrite the scratch anyway, and refusing a restore because a stale // certificate would not delete trades a real capability for a bookkeeping problem. func (m *Manager) clearScratchMarker(scratch string) { if err := os.Remove(filepath.Join(scratch, scratchMarkerName)); err != nil && !os.IsNotExist(err) { m.logger.Printf("[WARN] [offbox] could not clear the stale scratch marker in %s: %v", scratch, err) } } // writeScratchMarker writes the certificate atomically (tmp + fsync + rename) at mode 0600. Atomic // because a torn marker read as valid is the one failure this whole mechanism cannot tolerate — it // would certify a part-copy, which is the original defect wearing a new hat. func (m *Manager) writeScratchMarker(scratch, snapshotID string, full bool) error { data, err := json.Marshal(scratchMarker{ Schema: scratchMarkerSchema, SnapshotID: snapshotID, Full: full, FinishedAt: time.Now().UTC().Format(time.RFC3339), }) if err != nil { return err } final := filepath.Join(scratch, scratchMarkerName) tmp := final + ".tmp" f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600) if err != nil { return err } if _, err := f.Write(data); err != nil { f.Close() os.Remove(tmp) return err } if err := f.Sync(); err != nil { f.Close() os.Remove(tmp) return err } if err := f.Close(); err != nil { os.Remove(tmp) return err } return os.Rename(tmp, final) } // OffboxRestorePrepareFull resolves the latest snapshot's restore-size and verifies scratch headroom // for a FULL restore WITHOUT starting it (the two-step size-first gate). Returns the human size on // success, or a Hungarian error to flash on refusal (size unknown / no headroom — fail-closed). func (m *Manager) OffboxRestorePrepareFull(ctx context.Context, stack string) (string, error) { if !m.OffboxConfigured() { return "", fmt.Errorf("off-box backup not configured") } // R-411 — THE SECOND ENTRY POINT, and the one the customer's UI actually reaches FIRST. // // The full restore is TWO HTTP requests: this one computes the size for the confirm screen, and a // later one does the restore. They are separate calls, so the flag taken in RestoreOffboxScratch // does not cover this, and NOTHING nests. `offboxSnapshotSize` below shells `restic stats`, which // takes a repository lock — so without this, the collision R-411 records is still reachable // through the ordinary two-step flow even after the restore itself is flagged. // // Found by re-reading the call graph while fixing the other one, not by the original report. if err := m.acquireRunning(); err != nil { return "", err } defer m.releaseRunning() if !isSafeStackName(stack) { return "", fmt.Errorf("invalid stack name") } id, _, err := m.offboxLatestSnapshot(ctx, stack) if err != nil { return "", err } size, serr := m.offboxSnapshotSize(ctx, id) if serr != nil { return "", fmt.Errorf(offsiteSizeUnknownMsg) } _, nsRoot, derr := m.offboxRestoreScratchDir(stack) if derr != nil { return "", derr } need := size + size/10 if free := m.offboxFree()(nsRoot); free < need { return "", fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free)) } return humanizeBytes(size), nil } // R-357 customer-facing refusal strings, shared by every headroom gate on the off-site restore // surface. Named constants because a test asserts them verbatim and because the destructive gate added // in v0.226.0 MUST read identically to the two non-destructive ones that predate it — a customer who // meets this refusal on one path and a differently-worded one on another has to work out whether they // are the same problem. const ( offsiteNoSpaceMsgFmt = "Nincs elég szabad hely a visszaállításhoz (%s szükséges, %s szabad)." offsiteSizeUnknownMsg = "A mentés mérete nem állapítható meg — a teljes visszaállítás biztonsági okból nem indítható." ) // OffboxFullScratchReady reports whether a COMPLETED FULL restore scratch exists for stack — the gate // for the place-to-live and reconstitute actions. // // R-358 — WHAT THIS USED TO ANSWER, AND WHY IT WAS THE WRONG QUESTION. It used to be "the directory // exists and is non-empty", and its doc comment reassured the reader that // `PlaceOffsiteRestore re-validates per-path completeness`. That sentence is what made the weak gate // look adequate, and it is not true in the way it reads: PlaceOffsiteRestore stats the top-level // PLACEMENTS, not the files inside them, so a placement directory that exists but was only half // downloaded passes it. A restic run that died part-way leaves a non-empty directory, so the product // offered „Teljes visszaállítás indítása" over a part-copy and reported success on it (demo-hp, // 2026-08-21). // // It now asks the only question that distinguishes them: did the run that wrote this scratch FINISH, // and was it the full one. Every other answer — no marker, unreadable marker, wrong schema, full=false // — is FALSE, and says at WARN which one it was. **Fail closed: an unreadable marker is not a // completion certificate.** func (m *Manager) OffboxFullScratchReady(stack string) bool { if !isSafeStackName(stack) { return false } scratch, _, err := m.offboxRestoreScratchDir(stack) if err != nil { return false } if fi, sErr := os.Stat(scratch); sErr != nil || !fi.IsDir() { return false } data, rErr := os.ReadFile(filepath.Join(scratch, scratchMarkerName)) if rErr != nil { if !os.IsNotExist(rErr) { m.logger.Printf("[WARN] [offbox] %s: scratch completion marker unreadable (%v) — treating the copy as INCOMPLETE", stack, rErr) } return false } var mk scratchMarker if uErr := json.Unmarshal(data, &mk); uErr != nil { m.logger.Printf("[WARN] [offbox] %s: scratch completion marker does not parse (%v) — treating the copy as INCOMPLETE", stack, uErr) return false } if mk.Schema != scratchMarkerSchema { m.logger.Printf("[WARN] [offbox] %s: scratch completion marker has schema %d, expected %d — treating the copy as INCOMPLETE", stack, mk.Schema, scratchMarkerSchema) return false } if !mk.Full { m.logger.Printf("[INFO] [offbox] %s: scratch holds a UNIT-ONLY restore (snapshot %s) — not a full copy, so place-to-live stays closed", stack, mk.SnapshotID) return false } return true } // placement is one source→dest pair for place-to-live: src is the reconstructed absolute path under the // scratch (SP-3.1), dst is the live location under the app's current namespace root. type placement struct { src string dst string isUnit bool } // mapOffsiteRestorePaths maps a completed full-scratch restore to live placements (pure). The anchor // oldNs is derived by trimming backups/primary/ off the unit path (the snapshot may come from a // DIFFERENT drive after churn — liveNsRoot is where it goes). Refuses the WHOLE placement (no partial // writes) on: no unit path; a path outside oldNs (escape); a `..` segment; a non-unit path in the // reserved backups/ zone. func mapOffsiteRestorePaths(snapPaths []string, stack, scratch, liveNsRoot string) ([]placement, error) { unitSuffix := "/backups/primary/" + stack oldNs := "" for _, p := range snapPaths { if strings.HasSuffix(p, unitSuffix) { oldNs = strings.TrimSuffix(p, unitSuffix) break } } if oldNs == "" { return nil, fmt.Errorf("a pillanatképben nincs mentési egység (backups/primary/%s)", stack) } out := make([]placement, 0, len(snapPaths)) for _, p := range snapPaths { // Every captured path must be a STRICT descendant of oldNs. Requiring the trailing "/" also // catches p == oldNs (the namespace root itself — F-3a-3), which would otherwise map to a junk // placement nesting the whole old namespace under the live root. if !strings.HasPrefix(p, oldNs+"/") { return nil, fmt.Errorf("a pillanatkép egy útvonala a névtéren kívülre mutat: %s", p) } rel := strings.TrimPrefix(p, oldNs+"/") for _, seg := range strings.Split(rel, "/") { if seg == ".." { return nil, fmt.Errorf("a pillanatkép egy útvonala érvénytelen (..): %s", p) } } isUnit := rel == "backups/primary/"+stack if !isUnit && (rel == "backups" || strings.HasPrefix(rel, "backups/")) { return nil, fmt.Errorf("nem-egység útvonal a fenntartott backups zónában: %s", p) } out = append(out, placement{ src: filepath.Join(scratch, p), // SP-3.1: abs source reconstructed under the target dst: filepath.Join(liveNsRoot, rel), isUnit: isUnit, }) } return out, nil } // placeCopier returns the place-to-live missing-only merge (nil seam → rsyncRestoreMissing, the // `-a --ignore-existing` additive copy). NEVER rsyncMirror (--delete). func (m *Manager) placeCopier() func(src, dst string) (int, error) { if m.offboxPlaceCopier != nil { return m.offboxPlaceCopier } return rsyncRestoreMissing } // PlaceOffsiteRestore places a COMPLETED full-scratch restore into the app's live locations via a // missing-only merge (§7.3), so the SQ3 immich case is restorable from offsite alone. The recovery // unit is placed ONLY if the live unit is ABSENT (never overwrites a local unit); every other path is // merged missing-only. Does NOT deploy/start anything — RecreateStackFromUnit / the restore flow owns // that. Single-flight. Requires a completed full scratch (deterministic path + existence check). func (m *Manager) PlaceOffsiteRestore(ctx context.Context, stack string) error { if !m.OffboxConfigured() { return fmt.Errorf("off-box backup not configured") } if !isSafeStackName(stack) { return fmt.Errorf("invalid stack name") } if err := m.acquireRunning(); err != nil { return fmt.Errorf("egy másik mentési/visszaállítási művelet már fut") } defer m.releaseRunning() scratch, _, err := m.offboxRestoreScratchDir(stack) if err != nil { return err } if _, sErr := os.Stat(scratch); sErr != nil { return fmt.Errorf("nincs előkészített teljes visszaállítás — futtass előbb egy teljes visszaállítást") } id, paths, err := m.offboxLatestSnapshot(ctx, stack) if err != nil { return err } _ = id // F-3a-1a said: the live target uses the RAW HDD path, and an empty HDD means undeployed. // R-356 split that: "undeployed" is now asked directly, and the destination is resolved by the // SAME rule the capture side wrote this snapshot with (GetAppDrivePath — drive if the app has // one, system data path otherwise). For the 13 needs_hdd apps nothing changes; for the 40 that // were never offered a drive the old test was permanently true and this merge was unreachable. if !m.isStackDeployed(stack) { return fmt.Errorf("a(z) %s nincs telepítve — előbb állítsd helyre az alkalmazást, utána az adatokat", stack) } hdd := strings.TrimSpace(m.GetAppDrivePath(stack)) if hdd == "" { // Installed, but the box cannot name its own data root. Distinct reason ⇒ distinct sentence: // telling the customer to reinstall a running app would hide the real fault. return fmt.Errorf("a(z) %s telepítve van, de a vezérlő nem tudja megállapítani, hová tartoznak az adatai "+ "(nincs beállítva rendszer-adatterület). Ellenőrizd a tárhely beállításait (Tárhely), utána indítsd újra a visszaállítást", stack) } liveNs := m.namespaceRoot(hdd) // F-3a-1b: headroom gate — a missing-only merge copies at most the scratch size; refuse before any // copy if the live drive lacks that (conservative — scratch and live often share a drive). if free, need := m.offboxFree()(liveNs), m.offboxSize()(scratch); free < need { return fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free)) } placements, err := mapOffsiteRestorePaths(paths, stack, scratch, liveNs) if err != nil { return err // whole-placement refusal (no partial writes) } // F-3a-4: stat pre-pass over EVERY placement BEFORE the first copy — an incomplete scratch (e.g. a // unit-only restore, userdata srcs absent) refuses with ZERO copies, making "no partial writes" true. for _, pl := range placements { if _, sErr := os.Stat(pl.src); sErr != nil { return fmt.Errorf("a teljes visszaállítás hiányos (%s nincs meg) — futtass előbb egy teljes visszaállítást", filepath.Base(pl.src)) } } copier := m.placeCopier() var placed int for _, pl := range placements { if pl.isUnit { if _, liveErr := os.Stat(pl.dst); liveErr == nil { m.logger.Printf("[INFO] [offbox] place %s: live recovery unit present — not overwriting", stack) continue // never overwrite a local unit } } n, cErr := copier(pl.src, pl.dst) if cErr != nil { return fmt.Errorf("a(z) %s helyreállítása sikertelen: %w", stack, cErr) // scratch KEPT for retry } placed += n } // F-3a-2: on FULL success, remove the scratch best-effort (OffboxFullScratchReady then turns false → // the place button disappears). A failed placement returned above, keeping the scratch for a retry. if rmErr := os.RemoveAll(scratch); rmErr != nil { m.logger.Printf("[WARN] [offbox] place %s: scratch cleanup failed (harmless): %v", stack, rmErr) } else { m.logger.Printf("[INFO] [offbox] place %s: scratch removed after successful placement", stack) } m.logger.Printf("[INFO] [offbox] placed %s from offsite scratch: %d file(s) merged (missing-only)", stack, placed) return nil }