Files
felhom-controller/controller/internal/backup/offbox_restore.go
T
admin 7c05b59708
gates / gates (push) Successful in 24s
v0.253.0 — errors carry the key of the sentence they are (R-557 slice 2 release B)
179 Hungarian sentences were built deep inside a package with fmt.Errorf and printed by
whoever caught them: too late to translate where they are shown, too early where they are
made. Every one now carries its key across that gap. ZERO Hungarian error literals remain.

util.MsgError does three things at once, each earned:
  - Error() is the Hungarian, byte for byte, so every un-converted printer is unchanged;
  - errors.Is answers for the kind AND for a wrapped cause (KindErrorf dropped the cause);
  - an error ARGUMENT renders recursively, so "formázás sikertelen: %w" translates whole.
A foreign error — restic, docker, ssh, the stdlib — prints verbatim. It is not ours.

76 display sites go through errText, and TestNoErrErrorInPageOutput convicts any that do
not. memoryVerdict returns an error rather than a sentence, so the deploy's 409 and the
household's language come from one value; UpdateRefusal gained a Cause to carry it.

Plurals, one rule, stated once: a key with .one/.other takes its COUNT first. Not a
per-call-site flag — the producer somebody forgot would read "3 app is not running". The
guard caught a real key collision (alert.deadapp.one) the day the rule landed.

TWO DEFECTS FOUND IN MY OWN TOOLING, recorded rather than quietly fixed. The bulk converter
silently dropped multi-line concatenations, damaging 7 producers — and the parity gate could
not see it, because every surviving fragment WAS a real base literal while the CALL had lost
text; two behaviour tests caught it. And the counting script was case-sensitive, so it said
"0 left" while five remained.

MinAgent: 0.131.0 (unchanged). No hub release needed.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-18 11:44:30 +02:00

736 lines
35 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package backup
import (
"context"
"encoding/json"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"os"
"os/exec"
"path/filepath"
"strings"
"time"
)
// Offsite restore rework (Task 3a §7). With mandatory userdata now in snapshots, restore needs three
// changes over the old dump-to-rootfs-scratch:
// 1. scratch relocated off the ~8 GB guest rootfs onto a data drive, behind a headroom gate (F-A1);
// 2. a unit-only DEFAULT restore (`--include <absolute-unit-path>`, SP-3.2) — full is a deliberate,
// size-gated second action;
// 3. place-to-live = a missing-only merge (never --delete) so the SQ3 immich case is restorable
// from offsite alone.
// ID-first everywhere (§3): `restic stats --tag` is UNPROVEN on 0.14.0, so the size lookup resolves the
// snapshot ID via `snapshots latest --tag` and calls `stats <ID>`.
const (
// offboxUnitOnlyFreeFloor — a unit-only restore needs at least this much free on the scratch drive.
// Catalog recovery units are MB–1 GB (SQ4); 2 GiB is a safe floor without a per-snapshot size probe.
offboxUnitOnlyFreeFloor = int64(2) << 30
)
// unitOnlyHeadroom is THE unit-only free-space gate, shared by the customer's scratch restore and the
// R-87 nightly proof so the two can never disagree about how much room a unit restore needs or about
// what the customer is told when there is not enough (REUSE.md's `offsiteNoSpaceMsgFmt` rule).
//
// FAIL-CLOSED ON AN UNMEASURABLE PROBE. `offboxFree` returns 0 when it cannot read the filesystem,
// and `0 < floor` is true, so an unreadable drive REFUSES rather than sailing through — the inverse
// of the R-357 shape where `free < need` with `need == 0` made a gate inert.
func unitOnlyHeadroom(free int64) error {
if free < offboxUnitOnlyFreeFloor {
return fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(offboxUnitOnlyFreeFloor), humanizeBytes(free))
}
return nil
}
// SetOffboxFreeFn overrides the restore free-space probe (tests; the Windows go-test host has no df).
func (m *Manager) SetOffboxFreeFn(fn func(path string) int64) { m.offboxFreeFn = fn }
// WriteScratchMarkerForTest exposes the marker writer to the web package's flow test. Test-only by
// name so a production caller reads as obviously wrong: only RestoreOffboxScratch may certify a
// scratch, because only it knows whether the download finished.
func (m *Manager) WriteScratchMarkerForTest(scratch, snapshotID string, full bool) error {
return m.writeScratchMarker(scratch, snapshotID, full)
}
// SetOffboxLatestSnapshotFn overrides the restic snapshot lookup (tests; no restic needed). See the
// field comment on Manager.offboxLatestSnapFn for why this seam exists rather than a code-reading
// argument that the R-357 gate sits early enough.
func (m *Manager) SetOffboxLatestSnapshotFn(fn func(ctx context.Context, stack string) (string, []string, error)) {
m.offboxLatestSnapFn = fn
}
// SetOffboxFullPlaceCopier overrides the FULL-restore overwrite copier (tests; no rsync needed).
func (m *Manager) SetOffboxFullPlaceCopier(fn func(src, dst string) (int, error)) {
m.offboxFullPlaceCopier = fn
}
// SetRollbackImportFn overrides the ROLLBACK's ImportDump (tests; no Docker needed). Separate from
// the replay's own import seam on purpose — see the field comment on Manager.rollbackImport.
func (m *Manager) SetRollbackImportFn(fn func(ctx context.Context, db DiscoveredDB, dumpPath string) error) {
m.rollbackImport = fn
}
// SetSafetyDumpFn overrides the pre-restore safety dump (tests; no Docker needed).
func (m *Manager) SetSafetyDumpFn(fn func(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult) {
m.safetyDumpFn = fn
}
// offboxFree returns the free-space probe (nil seam → the real diskFreeBytes).
func (m *Manager) offboxFree() func(string) int64 {
if m.offboxFreeFn != nil {
return m.offboxFreeFn
}
return diskFreeBytes
}
// diskFreeBytes returns available bytes on the filesystem holding path (0 on any error). Mirrors
// appexport.DiskFree; kept local so the backup package needs no cross-package dependency.
func diskFreeBytes(path string) int64 {
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
defer cancel()
out, err := exec.CommandContext(ctx, "df", "--output=avail", "-B1", path).Output()
if err != nil {
return 0
}
lines := strings.Split(strings.TrimSpace(string(out)), "\n")
if len(lines) < 2 {
return 0
}
var size int64
fmt.Sscanf(strings.TrimSpace(lines[1]), "%d", &size)
return size
}
// offboxUnitPathOf returns the snapshot path that is the recovery unit for stack (suffix
// backups/primary/<stack>), or "" if none is present.
func offboxUnitPathOf(paths []string, stack string) string {
suffix := "/backups/primary/" + stack
for _, p := range paths {
if strings.HasSuffix(p, suffix) {
return p
}
}
return ""
}
// offboxLatestSnapshot resolves the newest snapshot for stack: its short ID + captured paths, via
// `snapshots latest --tag <stack> --json`. When the tag spans more than one group (old unit-only shape
// + new enlarged shape), it returns the newest by time.
func (m *Manager) offboxLatestSnapshot(ctx context.Context, stack string) (id string, paths []string, err error) {
if m.offboxLatestSnapFn != nil {
return m.offboxLatestSnapFn(ctx, stack)
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
out, serr := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "latest", "--tag", stack, "--json")...)
if serr != nil {
return "", nil, fmt.Errorf("offbox snapshots %s: %w: %s", stack, serr, truncate(out))
}
var snaps []struct {
ShortID string `json:"short_id"`
ID string `json:"id"`
Time time.Time `json:"time"`
Paths []string `json:"paths"`
}
if json.Unmarshal(out, &snaps) != nil || len(snaps) == 0 {
return "", nil, util.MsgError("err.backup.offbox_nincs_pillanatkep_a_z_alkalmazashoz", stack)
}
best := 0
for i := 1; i < len(snaps); i++ {
if snaps[i].Time.After(snaps[best].Time) {
best = i
}
}
id = snaps[best].ShortID
if id == "" {
id = snaps[best].ID
}
return id, snaps[best].Paths, nil
}
// offboxSnapshotSize returns the restore-size (logical bytes) of ONE snapshot via `stats <ID> --json`
// (default mode — for a single snapshot ID this is exactly that snapshot's on-disk-when-restored size,
// the correct headroom meaning; SP-1). ID-first: never `stats --tag` (unproven on 0.14.0).
func (m *Manager) offboxSnapshotSize(ctx context.Context, id string) (int64, error) {
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "stats", id, "--json")...)
if err != nil {
return 0, fmt.Errorf("offbox stats %s: %w: %s", id, err, truncate(out))
}
var st struct {
TotalSize int64 `json:"total_size"`
}
if json.Unmarshal(out, &st) != nil || st.TotalSize <= 0 {
return 0, util.MsgError("err.backup.offbox_a_z_pillanatkep_merete_ismeretlen", id)
}
return st.TotalSize, nil
}
// offboxRestoreScratchDir returns the on-DATA-DRIVE scratch dir for an app's offsite restore
// (<nsRoot>/backups/offsite-restore/<app>) plus the namespace root (an existing dir, for the free-space
// probe). NEVER cfg.Paths.DataDir (the rootfs — the F-A1 filler). App's HDD drive first; else the first
// schedulable storage path; else a Hungarian refusal.
func (m *Manager) offboxRestoreScratchDir(stack string) (scratch, nsRoot string, err error) {
// offsiteRestoreRootFor is THE place `backups/offsite-restore` is spelled (offbox_verify_copies.go)
// — the listing/delete surface must resolve byte-identical paths to the ones written here.
// unitOnly=false: the CUSTOMER's scratch can hold a full restore (bulk userdata), so it must NOT
// fall back to the state-only system disk — see offboxScratchDirIn.
return m.offboxScratchDirIn(stack, m.offsiteRestoreRootFor, false)
}
// offboxProofScratchDir is the R-87 nightly proof's scratch, resolved by the SAME drive-preference
// rules and into a DIFFERENT root (`backups/offsite-proof`).
//
// THE SEPARATE ROOT IS NOT TIDINESS, IT PREVENTS A DELETE. The proof removes its scratch on every
// path, including failure. Sharing `backups/offsite-restore/<app>` would mean a nightly background
// job deleting the verification copy a CUSTOMER made and is looking at — a poorer actor destroying a
// richer one, which is R-403's shape wearing different clothes. A separate root also keeps the proof
// copy invisible to `DeleteOffsiteRestoreCopy`, the copy listing and `OffboxFullScratchReady`, so it
// can never be offered for placement into a live app.
func (m *Manager) offboxProofScratchDir(stack string) (scratch, nsRoot string, err error) {
// unitOnly=true: the proof restores ONE recovery unit (`--include <unit>`) and deletes it. For a
// driveless app that unit already lives permanently on the system data path, so a scratch there is
// at most a second copy of something already present — see offboxScratchDirIn.
return m.offboxScratchDirIn(stack, m.offsiteProofRootFor, true)
}
// offboxScratchDirIn holds the drive-preference rules once. `rootFor` chooses WHICH root under the
// namespace the scratch lands in; everything else — the network-storage refusal, the ordering, the
// R-252 wording — is shared, so the proof path can never drift from the customer path on the parts
// that must not differ.
// R-414 — THE SYSTEM-DATA FALLBACK, AND WHY IT IS SCOPED BY WHAT IS BEING RESTORED.
//
// THE GAP. `demo-felhom` has ZERO registered storage paths, so steps (1)-(3) all miss and this
// refused. The nightly proof therefore could not run AT ALL on that box — every night, with only a
// WARN — and because its error path reaches no verdict, `last_proof_result` stayed ABSENT, which is
// also what a controller too old to have the feature sends. The hub could not tell them apart.
//
// WAS IT MISSED OR DELIBERATE? Established from R-356's own commit (`08eb1a6`, 2026-08-22), whose test
// comments say the scratch resolver *"still resolves to the registered storage path … only the
// DESTINATION moves"* — i.e. it was OUT OF SCOPE for that change, which was about where restored data
// LANDS. It was never ruled out on state-only grounds: the one comment about a `systemDataPath`
// fallback belonged to `PlaceOffsiteRestore` and concerned merging bulk USERDATA onto the SSD, and
// R-356 deliberately overruled even that. This function's own documented exclusion is
// `cfg.Paths.DataDir` — the ROOTFS — which is a different filesystem entirely.
//
// SO §6.3's [DESIGN] RULE APPLIES, AND IT NOW HAS A FOURTH CONSUMER: "the restore destination is
// resolved by the same rule as the capture destination — the drive if the app declares one, the system
// data path otherwise."
//
// BUT THE TWO CALLERS ASK DIFFERENT QUESTIONS, and answering both with one predicate is the R-356
// defect itself. So the fallback is scoped:
//
// - unitOnly=true (the R-87 proof): may fall back. `07` §7 records as [FACT] that a driveless app's
// recovery unit ALREADY sits on `systemDataPath` indefinitely — "the SSD-only system-data
// fallback" — and that the same-device placement is "intended, not a defect". The scratch is
// bounded by that unit's own size and is deleted on every path.
// - unitOnly=false (the customer's scratch): must NOT. A full restore pulls the app's bulk userdata,
// and `07` §2.2 makes the internal SSD a STATE-ONLY tier. This is exactly the case the deleted
// `PlaceOffsiteRestore` comment worried about, and the R-252 refusal below stays correct for it.
//
// The headroom gate still applies on the fallback path — it is the caller's `unitOnlyHeadroom`, which
// refuses when the floor is not met, so a small system disk is protected by the same floor as a drive.
func (m *Manager) offboxScratchDirIn(stack string, rootFor func(string) string, unitOnly bool) (scratch, nsRoot string, err error) {
scratchFor := func(root string) (string, string) {
return filepath.Join(rootFor(root), stack), m.namespaceRoot(root)
}
isNet := func(path string) bool { return m.settings != nil && m.settings.IsNetworkStoragePath(path) }
// (1) the app's own drive — preferred, but ONLY if it is not NETWORK storage (F-3afix-1). restic
// restores uid/gid/setgid fully onto a LOCAL fs (SP-3.3); a squashed network scratch would feed
// PlaceOffsiteRestore wrong-owner files — the F-6C-1 silently-broken-restore class, offsite-side.
if m.stackProvider != nil {
if hdd := strings.TrimSpace(m.stackProvider.GetStackHDDPath(stack)); hdd != "" && !isNet(hdd) {
s, nr := scratchFor(hdd)
return s, nr, nil
}
}
// (2) the first NON-network schedulable path.
if m.settings != nil {
for _, sp := range m.settings.GetSchedulableStoragePaths() {
if strings.TrimSpace(sp.Path) != "" && !sp.IsNetwork() {
s, nr := scratchFor(sp.Path)
return s, nr, nil
}
}
// (3) last resort ONLY: any schedulable path, with a loud WARN — a network scratch cannot
// guarantee ownership fidelity under root_squash.
for _, sp := range m.settings.GetSchedulableStoragePaths() {
if strings.TrimSpace(sp.Path) != "" {
m.logger.Printf("[WARN] [offbox] %s: restore scratch on network storage %s — ownership fidelity not guaranteed under squash", stack, sp.Path)
s, nr := scratchFor(sp.Path)
return s, nr, nil
}
}
}
// (4) R-414: a UNIT-ONLY restore falls back to the system data path, which is where a driveless
// app's unit already lives. Deliberately AFTER the network last-resort: a registered drive,
// even a network one, is still a better scratch for ownership fidelity than the system disk.
if unitOnly {
if sysPath := strings.TrimSpace(m.cfg.Paths.SystemDataPath); sysPath != "" {
s, nr := scratchFor(sysPath)
m.logger.Printf("[INFO] [offbox] %s: no registered data drive — unit-only scratch falls back to the system data path %s (R-414; the unit already lives there)", stack, sysPath)
return s, nr, nil
}
}
// R-252: name the reason AND the way to act on it. This refusal is what a rebuilt box hits — the
// drives are physically fine and still mounted, it is their REGISTRATION that the destroyed guest
// took with it — and until v0.207.0 it said only that a drive was missing, which reads like data
// loss and offers nothing to do.
return "", "", util.MsgError("err.backup.no_registered_drive")
}
// HasRestoreDestination reports whether an offsite restore has anywhere on this box to write.
//
// R-252: the restore PAGE asks this question through the same helper the resolver answers it with,
// so the notice cannot appear on a box that would restore fine (Scenario E) nor stay hidden on one
// that would refuse. A second copy of the predicate is exactly how a page ends up promising what the
// handler then refuses — which is the neighbouring defect, R-253.
//
// It mirrors the resolver's BOX-level branches (2) and (3) — the schedulable storage paths. Branch
// (1), the app's own HDD path, is deliberately not consulted: an installed app's HDD path IS a
// registered storage path, so the two cannot disagree in practice, and where they could, erring
// toward showing the notice is erring toward telling the customer something true.
func (m *Manager) HasRestoreDestination() bool {
if m.settings == nil {
return false
}
for _, sp := range m.settings.GetSchedulableStoragePaths() {
if strings.TrimSpace(sp.Path) != "" {
return true
}
}
return false
}
// RestoreOffboxScratch restores an app's latest offsite snapshot to an on-data-drive scratch dir
// (non-destructive — never overwrites live data). full=false (the default) restores the recovery UNIT
// only (`--include <absolute-unit-path>`, SP-3.2); full=true restores the whole snapshot (unit +
// mandatory userdata) behind a size×1.1 headroom gate. Fail-closed: an unknown snapshot size refuses a
// full restore.
func (m *Manager) RestoreOffboxScratch(ctx context.Context, stack string, full bool) error {
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
// R-411/R-408 — THE SINGLE-WRITER FLAG, and it must be taken HERE, before anything touches the
// repository.
//
// WHAT IT COSTS TO OMIT IT, measured on demo-hp 2026-08-31 and not reasoned about: this function
// runs `offboxSnapshotSize` for a full restore, which shells `restic stats` — and **`stats` TAKES
// A REPOSITORY LOCK** (clean-room test: nothing else running, four invocations, the sampler reads
// `locks=1`). Without this flag the integrity check is not blocked, starts, meets that lock, and
// `resticStep` escalates to `unlock --remove-all` — the argv sampler caught `restore …` and
// `unlock --remove-all` in the SAME sample at 20:50:51 — while logging *"a stale exclusive lock
// left by a previous crash"*. There was no crash. `resticStep`'s own safety argument is that the
// in-process mutex proves no sibling is live; this is the caller that made that false.
//
// BEFORE the snapshot lookup and the size probe, deliberately: a flag taken after the probe
// protects nothing, because the probe is what takes the lock.
//
// The refusal shape matches the five siblings, so the handler's Hungarian wording is unchanged and
// `restoreOpBlocked()` still refuses a second press exactly as it does today.
if err := m.acquireRunning(); err != nil {
return err
}
defer m.releaseRunning()
if !isSafeStackName(stack) {
return fmt.Errorf("invalid stack name")
}
id, paths, err := m.offboxLatestSnapshot(ctx, stack)
if err != nil {
return err
}
unitPath := offboxUnitPathOf(paths, stack)
if unitPath == "" {
return util.MsgError("err.backup.a_z_pillanatkepeben_nincs_mentesi_egyseg", stack)
}
scratch, nsRoot, err := m.offboxRestoreScratchDir(stack)
if err != nil {
return err
}
// Headroom gate (F-A1) — probed on the namespace root (an existing dir).
free := m.offboxFree()(nsRoot)
if full {
size, serr := m.offboxSnapshotSize(ctx, id)
if serr != nil {
// SizeUnknown never renders as fits — fail closed.
return fmt.Errorf(offsiteSizeUnknownMsg)
}
need := size + size/10 // ×1.1
if free < need {
return fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free))
}
} else if herr := unitOnlyHeadroom(free); herr != nil {
return herr
}
// F-A1 hygiene: drop the legacy rootfs scratch (DataDir/offbox-restore/<app>) best-effort.
legacy := filepath.Join(m.cfg.Paths.DataDir, "offbox-restore", stack)
if _, sErr := os.Stat(legacy); sErr == nil {
if rmErr := os.RemoveAll(legacy); rmErr != nil {
m.logger.Printf("[WARN] [offbox] could not remove legacy rootfs restore scratch %s: %v", legacy, rmErr)
} else {
m.logger.Printf("[INFO] [offbox] removed legacy rootfs restore scratch %s", legacy)
}
}
if err := os.MkdirAll(scratch, 0o755); err != nil {
return fmt.Errorf("restore dir: %w", err)
}
// R-358: a marker from a PREVIOUS run must never certify this one. Cleared here, before restic
// touches anything, so the window in which a stale certificate could vouch for a part-copy does not
// exist. If this run fails, the scratch is left with files and NO marker — which is precisely the
// state OffboxFullScratchReady must read as "not ready".
m.clearScratchMarker(scratch)
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
defer cancel()
m.unlockStale(rctx, base, env) // pre-restore hygiene
args := []string{"restore", id, "--target", scratch}
if !full {
args = append(args, "--include", unitPath) // SP-3.2: absolute snapshot unit path = unit-only
}
out, rerr := m.resticStep(rctx, env, base, "restore:"+stack, args...)
if rerr != nil {
return fmt.Errorf("offbox restore %s: %w: %s", stack, rerr, truncate(out))
}
m.logger.Printf("[INFO] [offbox] restored %s (%s, full=%v) → %s", stack, id, full, scratch)
// R-358: the completion certificate, written ONLY now — after restic returned nil. Writing it
// earlier would certify a download that has not happened, which is the defect with an extra step.
// Written for full=false runs too: the `full` field inside it, not its presence, is what
// distinguishes a unit-only scratch from a complete one.
if err := m.writeScratchMarker(scratch, id, full); err != nil {
// The restore itself succeeded, so this is not an error to fail the operation on — but it is
// NOT silent, and the consequence is stated: without the marker the scratch reads as not-ready,
// which is the fail-closed direction. Better a re-run than a placement over an uncertified copy.
m.logger.Printf("[ERROR] [offbox] %s: restore succeeded but the completion marker could not be written: %v — the scratch will read as NOT ready and the download must be re-run", stack, err)
}
return nil
}
// --- R-358: the scratch completion marker ------------------------------------------------------
//
// THE DEFECT. `OffboxFullScratchReady` used to answer "the directory exists and is non-empty". A restic
// download that failed part-way leaves exactly that: a directory with files in it. So the product
// offered „Teljes visszaállítás indítása" over a part-copy, and pressing it reported success —
// observed on demo-hp 2026-08-21. A non-empty directory is evidence that something was written, never
// that everything was.
//
// The marker is the missing fact: not "are there files" but "did the run that wrote them FINISH, and
// was it the full one". Only the run itself can know that, so only the run writes it.
//
// It lives at the scratch ROOT, which is safe from placement for a reason worth stating rather than
// assuming: `mapOffsiteRestorePaths` builds placements from the SNAPSHOT's own path list, not from a
// directory walk, so a file that exists only locally is invisible to it. That is pinned by
// TestR358_MarkerIsNeverPlaced rather than left as a comment.
const scratchMarkerName = ".felhom-restore-complete.json"
// scratchMarker is the on-disk completion certificate. `Schema` is carried so a future format change
// is a refusal rather than a misreading — an unrecognised schema fails closed like every other
// unreadable marker.
type scratchMarker struct {
Schema int `json:"schema"`
SnapshotID string `json:"snapshot_id"`
Full bool `json:"full"`
FinishedAt string `json:"finished_at"`
}
const scratchMarkerSchema = 1
// clearScratchMarker removes any existing marker, best-effort. A failure to remove is logged and NOT
// returned: the caller is about to overwrite the scratch anyway, and refusing a restore because a stale
// certificate would not delete trades a real capability for a bookkeeping problem.
func (m *Manager) clearScratchMarker(scratch string) {
if err := os.Remove(filepath.Join(scratch, scratchMarkerName)); err != nil && !os.IsNotExist(err) {
m.logger.Printf("[WARN] [offbox] could not clear the stale scratch marker in %s: %v", scratch, err)
}
}
// writeScratchMarker writes the certificate atomically (tmp + fsync + rename) at mode 0600. Atomic
// because a torn marker read as valid is the one failure this whole mechanism cannot tolerate — it
// would certify a part-copy, which is the original defect wearing a new hat.
func (m *Manager) writeScratchMarker(scratch, snapshotID string, full bool) error {
data, err := json.Marshal(scratchMarker{
Schema: scratchMarkerSchema,
SnapshotID: snapshotID,
Full: full,
FinishedAt: time.Now().UTC().Format(time.RFC3339),
})
if err != nil {
return err
}
final := filepath.Join(scratch, scratchMarkerName)
tmp := final + ".tmp"
f, err := os.OpenFile(tmp, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o600)
if err != nil {
return err
}
if _, err := f.Write(data); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Sync(); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Close(); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, final)
}
// OffboxRestorePrepareFull resolves the latest snapshot's restore-size and verifies scratch headroom
// for a FULL restore WITHOUT starting it (the two-step size-first gate). Returns the human size on
// success, or a Hungarian error to flash on refusal (size unknown / no headroom — fail-closed).
func (m *Manager) OffboxRestorePrepareFull(ctx context.Context, stack string) (string, error) {
if !m.OffboxConfigured() {
return "", fmt.Errorf("off-box backup not configured")
}
// R-411 — THE SECOND ENTRY POINT, and the one the customer's UI actually reaches FIRST.
//
// The full restore is TWO HTTP requests: this one computes the size for the confirm screen, and a
// later one does the restore. They are separate calls, so the flag taken in RestoreOffboxScratch
// does not cover this, and NOTHING nests. `offboxSnapshotSize` below shells `restic stats`, which
// takes a repository lock — so without this, the collision R-411 records is still reachable
// through the ordinary two-step flow even after the restore itself is flagged.
//
// Found by re-reading the call graph while fixing the other one, not by the original report.
if err := m.acquireRunning(); err != nil {
return "", err
}
defer m.releaseRunning()
if !isSafeStackName(stack) {
return "", fmt.Errorf("invalid stack name")
}
id, _, err := m.offboxLatestSnapshot(ctx, stack)
if err != nil {
return "", err
}
size, serr := m.offboxSnapshotSize(ctx, id)
if serr != nil {
return "", fmt.Errorf(offsiteSizeUnknownMsg)
}
_, nsRoot, derr := m.offboxRestoreScratchDir(stack)
if derr != nil {
return "", derr
}
need := size + size/10
if free := m.offboxFree()(nsRoot); free < need {
return "", fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free))
}
return humanizeBytes(size), nil
}
// R-357 customer-facing refusal strings, shared by every headroom gate on the off-site restore
// surface. Named constants because a test asserts them verbatim and because the destructive gate added
// in v0.226.0 MUST read identically to the two non-destructive ones that predate it — a customer who
// meets this refusal on one path and a differently-worded one on another has to work out whether they
// are the same problem.
const (
offsiteNoSpaceMsgFmt = "Nincs elég szabad hely a visszaállításhoz (%s szükséges, %s szabad)."
offsiteSizeUnknownMsg = "A mentés mérete nem állapítható meg — a teljes visszaállítás biztonsági okból nem indítható."
)
// OffboxFullScratchReady reports whether a COMPLETED FULL restore scratch exists for stack — the gate
// for the place-to-live and reconstitute actions.
//
// R-358 — WHAT THIS USED TO ANSWER, AND WHY IT WAS THE WRONG QUESTION. It used to be "the directory
// exists and is non-empty", and its doc comment reassured the reader that
// `PlaceOffsiteRestore re-validates per-path completeness`. That sentence is what made the weak gate
// look adequate, and it is not true in the way it reads: PlaceOffsiteRestore stats the top-level
// PLACEMENTS, not the files inside them, so a placement directory that exists but was only half
// downloaded passes it. A restic run that died part-way leaves a non-empty directory, so the product
// offered „Teljes visszaállítás indítása" over a part-copy and reported success on it (demo-hp,
// 2026-08-21).
//
// It now asks the only question that distinguishes them: did the run that wrote this scratch FINISH,
// and was it the full one. Every other answer — no marker, unreadable marker, wrong schema, full=false
// — is FALSE, and says at WARN which one it was. **Fail closed: an unreadable marker is not a
// completion certificate.**
func (m *Manager) OffboxFullScratchReady(stack string) bool {
if !isSafeStackName(stack) {
return false
}
scratch, _, err := m.offboxRestoreScratchDir(stack)
if err != nil {
return false
}
if fi, sErr := os.Stat(scratch); sErr != nil || !fi.IsDir() {
return false
}
data, rErr := os.ReadFile(filepath.Join(scratch, scratchMarkerName))
if rErr != nil {
if !os.IsNotExist(rErr) {
m.logger.Printf("[WARN] [offbox] %s: scratch completion marker unreadable (%v) — treating the copy as INCOMPLETE", stack, rErr)
}
return false
}
var mk scratchMarker
if uErr := json.Unmarshal(data, &mk); uErr != nil {
m.logger.Printf("[WARN] [offbox] %s: scratch completion marker does not parse (%v) — treating the copy as INCOMPLETE", stack, uErr)
return false
}
if mk.Schema != scratchMarkerSchema {
m.logger.Printf("[WARN] [offbox] %s: scratch completion marker has schema %d, expected %d — treating the copy as INCOMPLETE", stack, mk.Schema, scratchMarkerSchema)
return false
}
if !mk.Full {
m.logger.Printf("[INFO] [offbox] %s: scratch holds a UNIT-ONLY restore (snapshot %s) — not a full copy, so place-to-live stays closed", stack, mk.SnapshotID)
return false
}
return true
}
// placement is one source→dest pair for place-to-live: src is the reconstructed absolute path under the
// scratch (SP-3.1), dst is the live location under the app's current namespace root.
type placement struct {
src string
dst string
isUnit bool
}
// mapOffsiteRestorePaths maps a completed full-scratch restore to live placements (pure). The anchor
// oldNs is derived by trimming backups/primary/<stack> off the unit path (the snapshot may come from a
// DIFFERENT drive after churn — liveNsRoot is where it goes). Refuses the WHOLE placement (no partial
// writes) on: no unit path; a path outside oldNs (escape); a `..` segment; a non-unit path in the
// reserved backups/ zone.
func mapOffsiteRestorePaths(snapPaths []string, stack, scratch, liveNsRoot string) ([]placement, error) {
unitSuffix := "/backups/primary/" + stack
oldNs := ""
for _, p := range snapPaths {
if strings.HasSuffix(p, unitSuffix) {
oldNs = strings.TrimSuffix(p, unitSuffix)
break
}
}
if oldNs == "" {
return nil, util.MsgError("err.backup.a_pillanatkepben_nincs_mentesi_egyseg_backups", stack)
}
out := make([]placement, 0, len(snapPaths))
for _, p := range snapPaths {
// Every captured path must be a STRICT descendant of oldNs. Requiring the trailing "/" also
// catches p == oldNs (the namespace root itself — F-3a-3), which would otherwise map to a junk
// placement nesting the whole old namespace under the live root.
if !strings.HasPrefix(p, oldNs+"/") {
return nil, util.MsgError("err.backup.a_pillanatkep_egy_utvonala_a_nevteren", p)
}
rel := strings.TrimPrefix(p, oldNs+"/")
for _, seg := range strings.Split(rel, "/") {
if seg == ".." {
return nil, util.MsgError("err.backup.a_pillanatkep_egy_utvonala_ervenytelen", p)
}
}
isUnit := rel == "backups/primary/"+stack
if !isUnit && (rel == "backups" || strings.HasPrefix(rel, "backups/")) {
return nil, util.MsgError("err.backup.nem_egyseg_utvonal_a_fenntartott_backups", p)
}
out = append(out, placement{
src: filepath.Join(scratch, p), // SP-3.1: abs source reconstructed under the target
dst: filepath.Join(liveNsRoot, rel),
isUnit: isUnit,
})
}
return out, nil
}
// placeCopier returns the place-to-live missing-only merge (nil seam → rsyncRestoreMissing, the
// `-a --ignore-existing` additive copy). NEVER rsyncMirror (--delete).
func (m *Manager) placeCopier() func(src, dst string) (int, error) {
if m.offboxPlaceCopier != nil {
return m.offboxPlaceCopier
}
return rsyncRestoreMissing
}
// PlaceOffsiteRestore places a COMPLETED full-scratch restore into the app's live locations via a
// missing-only merge (§7.3), so the SQ3 immich case is restorable from offsite alone. The recovery
// unit is placed ONLY if the live unit is ABSENT (never overwrites a local unit); every other path is
// merged missing-only. Does NOT deploy/start anything — RecreateStackFromUnit / the restore flow owns
// that. Single-flight. Requires a completed full scratch (deterministic path + existence check).
func (m *Manager) PlaceOffsiteRestore(ctx context.Context, stack string) error {
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
if !isSafeStackName(stack) {
return fmt.Errorf("invalid stack name")
}
if err := m.acquireRunning(); err != nil {
return util.MsgError("err.backup.egy_masik_mentesi_visszaallitasi_muvelet_mar")
}
defer m.releaseRunning()
scratch, _, err := m.offboxRestoreScratchDir(stack)
if err != nil {
return err
}
if _, sErr := os.Stat(scratch); sErr != nil {
return util.MsgError("err.backup.nincs_elokeszitett_teljes_visszaallitas_futtass_elobb")
}
id, paths, err := m.offboxLatestSnapshot(ctx, stack)
if err != nil {
return err
}
_ = id
// F-3a-1a said: the live target uses the RAW HDD path, and an empty HDD means undeployed.
// R-356 split that: "undeployed" is now asked directly, and the destination is resolved by the
// SAME rule the capture side wrote this snapshot with (GetAppDrivePath — drive if the app has
// one, system data path otherwise). For the 13 needs_hdd apps nothing changes; for the 40 that
// were never offered a drive the old test was permanently true and this merge was unreachable.
if !m.isStackDeployed(stack) {
return util.MsgError("err.backup.a_z_nincs_telepitve_elobb_allitsd", stack)
}
hdd := strings.TrimSpace(m.GetAppDrivePath(stack))
if hdd == "" {
// Installed, but the box cannot name its own data root. Distinct reason ⇒ distinct sentence:
// telling the customer to reinstall a running app would hide the real fault.
return util.MsgError("err.backup.no_data_root", stack)
}
liveNs := m.namespaceRoot(hdd)
// F-3a-1b: headroom gate — a missing-only merge copies at most the scratch size; refuse before any
// copy if the live drive lacks that (conservative — scratch and live often share a drive).
if free, need := m.offboxFree()(liveNs), m.offboxSize()(scratch); free < need {
return fmt.Errorf(offsiteNoSpaceMsgFmt, humanizeBytes(need), humanizeBytes(free))
}
placements, err := mapOffsiteRestorePaths(paths, stack, scratch, liveNs)
if err != nil {
return err // whole-placement refusal (no partial writes)
}
// F-3a-4: stat pre-pass over EVERY placement BEFORE the first copy — an incomplete scratch (e.g. a
// unit-only restore, userdata srcs absent) refuses with ZERO copies, making "no partial writes" true.
for _, pl := range placements {
if _, sErr := os.Stat(pl.src); sErr != nil {
return util.MsgError("err.backup.a_teljes_visszaallitas_hianyos_nincs_meg", filepath.Base(pl.src))
}
}
copier := m.placeCopier()
var placed int
for _, pl := range placements {
if pl.isUnit {
if _, liveErr := os.Stat(pl.dst); liveErr == nil {
m.logger.Printf("[INFO] [offbox] place %s: live recovery unit present — not overwriting", stack)
continue // never overwrite a local unit
}
}
n, cErr := copier(pl.src, pl.dst)
if cErr != nil {
return util.MsgError("err.backup.a_z_helyreallitasa_sikertelen", stack, cErr) // scratch KEPT for retry
}
placed += n
}
// F-3a-2: on FULL success, remove the scratch best-effort (OffboxFullScratchReady then turns false →
// the place button disappears). A failed placement returned above, keeping the scratch for a retry.
if rmErr := os.RemoveAll(scratch); rmErr != nil {
m.logger.Printf("[WARN] [offbox] place %s: scratch cleanup failed (harmless): %v", stack, rmErr)
} else {
m.logger.Printf("[INFO] [offbox] place %s: scratch removed after successful placement", stack)
}
m.logger.Printf("[INFO] [offbox] placed %s from offsite scratch: %d file(s) merged (missing-only)", stack, placed)
return nil
}