package backup import ( "context" "crypto/rand" "crypto/sha256" "encoding/hex" "encoding/json" "errors" "fmt" "os" "os/exec" "path/filepath" "regexp" "sort" "strings" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // Off-box (NAS) backup target — Part B. An ENCRYPTED restic repo reached over SFTP (no kernel mount; // restic talks SFTP to the NAS directly). This is the "1 off-site" leg of 3-2-1 for the app-data tier // (each off-box app's recovery unit + DB dumps + volume tars), distinct from the local cross-drive rsync // copy and from the agent's PBS whole-CT DR. The NAS sees only ciphertext. // // THE load-bearing lesson (spike Q8): a dead NAS must FAIL FAST, never hang the backup runner — every // restic invocation carries `-o sftp.args=…-oConnectTimeout=N…` so a black-holed endpoint errors in // ~N seconds instead of a multi-minute TCP retry. We also check restic's OWN exit code (never // pipe-swallow). Secrets (repo password + SSH key) live in 0600 files in the data dir — never logged, // never in a committed/non-0600 file; they ride DR via the PBS whole-CT snapshot of the rootfs. const ( // offboxConnectTimeoutSec is the SSH ConnectTimeout (spike Q8) — load-bearing fail-fast. offboxConnectTimeoutSec = 10 // offboxBackupTimeout bounds a full off-box run; offboxProbeTimeout bounds the quick repo probes. offboxBackupTimeout = 2 * time.Hour offboxProbeTimeout = 90 * time.Second ) // offboxRunner is the restic-exec seam (tests inject a fake so no real restic/ssh runs). It runs restic // with args + extra env and returns combined output + the process error (whose ExitCode the caller checks). type offboxRunner func(ctx context.Context, env []string, args ...string) ([]byte, error) func defaultOffboxRunner(ctx context.Context, env []string, args ...string) ([]byte, error) { cmd := exec.CommandContext(ctx, "restic", args...) cmd.Env = append(os.Environ(), env...) return cmd.CombinedOutput() } // SetOffboxRunner overrides the restic exec (tests). SetOffboxNotify wires the failure→operator alert. func (m *Manager) SetOffboxRunner(r offboxRunner) { m.offboxRunner = r } func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err error)) { m.offboxNotify = fn } // SetOffboxOrphanEvent wires the offsite-repo continuity event push (main.go → notifier). func (m *Manager) SetOffboxOrphanEvent(fn func(eventType, renamedTo string)) { m.offboxOrphanEvent = fn } // SetOffboxSSH overrides the raw-ssh exec used for the orphaned-repo move-aside (tests). func (m *Manager) SetOffboxSSH(fn func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error)) { m.offboxSSH = fn } // ErrOffboxOrphaned is the sentinel returned when the offsite repo exists but is keyed under a // passphrase this controller no longer has (the reinstall shape) — the run skips and the UI shows // the orphan card instead of the raw restic error. var ErrOffboxOrphaned = fmt.Errorf("offbox repo orphaned: exists but keyed under a previous, no-longer-available passphrase") // classifyResticProbe maps a `restic cat config` failure to a repo class. The signatures are the exact // restic stderr matched in the 2026-07-17 diagnosis + restic's no-repo message: // - "orphaned": repo present, wrong key ("wrong password or no key found") — the definitive signal // - "norepo": no repo at the location (init is the correct path) // - "other": network/SFTP-auth/unknown — NOT orphaned; existing error handling func classifyResticProbe(out []byte, err error) string { if err == nil { return "" // success — repo good } s := strings.ToLower(string(out)) switch { case strings.Contains(s, "wrong password or no key found"): return "orphaned" case strings.Contains(s, "unable to open config file"), strings.Contains(s, "is there a repository at the following location"), strings.Contains(s, "no such file"), strings.Contains(s, "does not exist"): return "norepo" default: return "other" } } func defaultOffboxSSH(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) { if port == 0 { port = 22 } args := []string{ "-p", fmt.Sprint(port), "-oBatchMode=yes", fmt.Sprintf("-oConnectTimeout=%d", offboxConnectTimeoutSec), "-oStrictHostKeyChecking=yes", "-oUserKnownHostsFile=" + knownHosts, "-i", keyPath, user + "@" + host, remoteCmd, } cmd := exec.CommandContext(ctx, "ssh", args...) return cmd.CombinedOutput() } func (m *Manager) sshRunner() func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) { if m.offboxSSH != nil { return m.offboxSSH } return defaultOffboxSSH } // OffboxOrphaned reports whether the offsite repo is in the ORPHANED state (persisted). func (m *Manager) OffboxOrphaned() bool { t := m.settings.GetOffboxTarget() return t != nil && t.RepoState == "orphaned" } // OffboxOrphanedRenamedTo returns the last move-aside path (for the card copy; "" if none). func (m *Manager) OffboxOrphanedRenamedTo() string { t := m.settings.GetOffboxTarget() if t == nil { return "" } return t.OrphanedRenamedTo } // markOrphaned sets the persistent ORPHANED state and, ONLY on the transition into it (not already // orphaned), pushes the offbox_repo_orphaned event — so scheduled runs never nightly-spam. func (m *Manager) markOrphaned() { already := m.OffboxOrphaned() if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoState = "orphaned" if o.OrphanedAt == "" || !already { o.OrphanedAt = time.Now().UTC().Format(time.RFC3339) } }); err != nil { m.logger.Printf("[WARN] [offbox] persist orphaned state failed: %v", err) } if !already { m.logger.Printf("[WARN] [offbox] offsite repo ORPHANED — remote holds backups written under a previous, no-longer-available key; runs will skip until reset") if m.offboxOrphanEvent != nil { m.offboxOrphanEvent("offbox_repo_orphaned", "") } } } // resetOrphanedRepo moves the orphaned repo aside (never deletes) and re-inits a fresh repo under the // CURRENT passphrase. Reversible. Used by the unclaimed auto-reset (Scenario B) and the claimed // confirmed reset (Scenario C). Caller holds the single-flight guarantee (run mutex) OR is the handler. func (m *Manager) resetOrphanedRepo(ctx context.Context, base, env []string, reason string) error { t := m.settings.GetOffboxTarget() if t == nil { return fmt.Errorf("no offsite target configured") } port := t.Port if port == 0 { port = 22 } // Choose a move-aside name that never overwrites an earlier orphaned copy (edge rule: -2, -3). date := time.Now().UTC().Format("20060102") base1 := t.RepoPath + ".orphaned-" + date newPath := base1 for i := 2; i <= 20; i++ { // `test -e
` returns non-zero (exit 1) when absent — that is the name we want. A transport
// error also lands here; we then just try the mv and let it fail loudly rather than loop.
out, err := m.sshRunner()(ctx, t.Host, t.User, port, m.offboxKeyPath(), m.offboxKnownHosts(), "test -e "+shellQuote(newPath))
if err != nil && !strings.Contains(strings.ToLower(string(out)), "denied") {
break // absent (test -e exit 1) → free name
}
newPath = fmt.Sprintf("%s-%d", base1, i)
}
m.logger.Printf("[WARN] [offbox] resetting orphaned repo (%s): move-aside %s -> %s, then re-init", reason, t.RepoPath, newPath)
if out, err := m.sshRunner()(ctx, t.Host, t.User, port, m.offboxKeyPath(), m.offboxKnownHosts(),
fmt.Sprintf("mv %s %s", shellQuote(t.RepoPath), shellQuote(newPath))); err != nil {
return fmt.Errorf("offbox move-aside failed: %w: %s", err, truncate(out))
}
// Fresh init under the current passphrase.
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer icancel()
if out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...); err != nil {
return fmt.Errorf("offbox re-init after move-aside failed: %w: %s", err, truncate(out))
}
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.RepoState = ""
o.OrphanedAt = ""
o.OrphanedRenamedTo = newPath
o.LastError = ""
}); err != nil {
m.logger.Printf("[WARN] [offbox] clear orphaned state failed: %v", err)
}
m.logger.Printf("[INFO] [offbox] orphaned repo reset complete — old history set aside at %s (move-aside, not deleted); fresh repo initialized", newPath)
if m.offboxOrphanEvent != nil {
m.offboxOrphanEvent("offbox_repo_reset", newPath)
}
return nil
}
// ResetOrphanedRepo is the handler entry point for the CLAIMED confirmed reset (Scenario C). It refuses
// unless the repo is currently orphaned. It builds the base/env and runs the move-aside + re-init.
func (m *Manager) ResetOrphanedRepo(ctx context.Context) error {
if !m.OffboxOrphaned() {
return fmt.Errorf("az offsite tároló nincs elárvult állapotban")
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
return m.resetOrphanedRepo(ctx, base, env, "operator-confirmed (claimed)")
}
// shellQuote single-quotes a path for the remote shell (our repo paths have no single quotes).
func shellQuote(s string) string { return "'" + strings.ReplaceAll(s, "'", `'\''`) + "'" }
// SetOffboxSizer overrides the mandatory-set byte estimator (tests). SetOffboxEnlargeBlockedNotifier
// wires the edge-triggered enlargement-blocked notification (main.go). SetOffboxPlaceCopier overrides
// the place-to-live missing-only merge (tests).
func (m *Manager) SetOffboxSizer(fn func(path string) int64) { m.offboxSizer = fn }
func (m *Manager) SetOffboxEnlargeBlockedNotifier(fn func(stack string, estBytes int64, usedGB, quotaGB int)) {
m.offboxEnlargeBlockedNotify = fn
}
func (m *Manager) SetOffboxPlaceCopier(fn func(src, dst string) (int, error)) { m.offboxPlaceCopier = fn }
// offboxSize returns the mandatory-set byte estimator (nil seam → the real du -sb dirSizeBytes).
func (m *Manager) offboxSize() func(string) int64 {
if m.offboxSizer != nil {
return m.offboxSizer
}
return dirSizeBytes
}
func (m *Manager) runner() offboxRunner {
if m.offboxRunner != nil {
return m.offboxRunner
}
return defaultOffboxRunner
}
func (m *Manager) offboxDir() string { return filepath.Join(m.cfg.Paths.DataDir, "offbox") }
func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir(), "ssh_key") }
func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") }
func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") }
// WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned
// known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator
// (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty
// sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys).
func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
return fmt.Errorf("offbox dir: %w", err)
}
if strings.TrimSpace(sshKey) != "" {
key := sshKey
if !strings.HasSuffix(key, "\n") {
key += "\n"
}
if err := os.WriteFile(m.offboxKeyPath(), []byte(key), 0o600); err != nil {
return fmt.Errorf("offbox ssh key: %w", err)
}
}
if strings.TrimSpace(knownHosts) != "" {
kh := knownHosts
if !strings.HasSuffix(kh, "\n") {
kh += "\n"
}
if err := os.WriteFile(m.offboxKnownHosts(), []byte(kh), 0o644); err != nil {
return fmt.Errorf("offbox known_hosts: %w", err)
}
}
// Auto-generate the repo password once (0600), never log it.
if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) {
pw, gerr := generateOffboxPassword()
if gerr != nil {
return gerr
}
if werr := os.WriteFile(m.offboxPwPath(), []byte(pw), 0o600); werr != nil {
return fmt.Errorf("offbox repo password: %w", werr)
}
}
return nil
}
// generateOffboxPassword returns a 256-bit hex repo password.
func generateOffboxPassword() (string, error) {
b := make([]byte, 32)
if _, err := rand.Read(b); err != nil {
return "", fmt.Errorf("offbox password gen: %w", err)
}
return hex.EncodeToString(b), nil
}
// Off-box target field validation — the security boundary for the values that flow into the `ssh … -s
// sftp` command restic runs. Host/user are charset-restricted AND must not start with '-' (an ssh
// OPTION-INJECTION vector: a host like "-oProxyCommand=evil" would make ssh execute an arbitrary command).
// RepoPath is an absolute, traversal-free, metacharacter-free path. This mirrors the agent's validate.go
// discipline: validate before any value reaches an exec.
var (
reOffboxHost = regexp.MustCompile(`^[A-Za-z0-9._-]+$`)
reOffboxUser = regexp.MustCompile(`^[A-Za-z0-9._-]+$`)
reOffboxPath = regexp.MustCompile(`^/[A-Za-z0-9._/-]+$`)
)
// ValidateOffboxTarget rejects values that could inject into the ssh command line (option injection via a
// leading '-', shell/space metacharacters, path traversal). Returns nil for a safe target.
func ValidateOffboxTarget(t *settings.OffboxTarget) error {
if t == nil {
return fmt.Errorf("no off-box target")
}
if t.Host == "" || len(t.Host) > 255 || !reOffboxHost.MatchString(t.Host) || strings.HasPrefix(t.Host, "-") || strings.HasPrefix(t.Host, ".") {
return fmt.Errorf("invalid NAS host (letters, digits, '.', '-', '_'; must not start with '-' or '.')")
}
if t.User == "" || len(t.User) > 64 || !reOffboxUser.MatchString(t.User) || strings.HasPrefix(t.User, "-") {
return fmt.Errorf("invalid user (letters, digits, '.', '-', '_'; must not start with '-')")
}
if len(t.RepoPath) > 512 || !reOffboxPath.MatchString(t.RepoPath) || strings.Contains(t.RepoPath, "..") {
return fmt.Errorf("invalid repo path (absolute, no spaces/metacharacters, no '..')")
}
if p := t.Port; p != 0 && (p < 1 || p > 65535) {
return fmt.Errorf("invalid port")
}
return nil
}
// OffboxConfigured reports whether the target is set, enabled, VALID, and the key + password files exist
// (so the UI/scheduler can gate a run without leaking why). A target that fails validation is treated as
// not-configured — fail-closed, so a bad/hostile persisted target can never reach the ssh exec.
func (m *Manager) OffboxConfigured() bool {
t := m.settings.GetOffboxTarget()
if t == nil || !t.Enabled || t.Host == "" || t.User == "" || t.RepoPath == "" {
return false
}
if err := ValidateOffboxTarget(t); err != nil {
return false
}
if _, err := os.Stat(m.offboxKeyPath()); err != nil {
return false
}
if _, err := os.Stat(m.offboxPwPath()); err != nil {
return false
}
return true
}
// offboxRepoPwPattern matches a valid restic repo password (generateOffboxPassword = 32 rand bytes → 64 hex).
var offboxRepoPwPattern = regexp.MustCompile(`^[0-9a-fA-F]{64}$`)
// ApplyOffsiteTarget configures the offbox target from a hub-provisioned descriptor (SLICE 2 apply-bridge):
// it writes the 0600 SSH key + pinned known_hosts, sets the target with EscrowState="pending", and pushes
// the repo password to the agent for escrow — the SAME fork-4 enable path a manual config takes. `stage` is
// the agent escrow-stage push (nil skips it, e.g. when the agent is unreachable — the run gate still holds).
func (m *Manager) ApplyOffsiteTarget(ctx context.Context, tgt *settings.OffboxTarget, sshKeyPEM, knownHosts string, stage func(ctx context.Context, pw string) error) error {
if err := m.WriteOffboxSecrets(sshKeyPEM, knownHosts); err != nil {
return fmt.Errorf("apply offsite secrets: %w", err)
}
// Re-apply (v0.109.1 live finding): the bridge rebuilds the target from the descriptor, but the
// EXISTING target's custody + runtime status must carry over — EscrowState tracks the REPO PASSWORD
// (preserved by WriteOffboxSecrets above, never rotated by this path), not the target coords; and the
// status fields belong to the runner. Without this, a quota bump demoted an escrowed demo target to
// pending and wiped its history (which would also false-trigger the hub's staleness alert).
if cur := m.settings.GetOffboxTarget(); cur != nil {
tgt.EscrowState = cur.EscrowState
tgt.LastRun, tgt.LastStatus, tgt.LastError = cur.LastRun, cur.LastStatus, cur.LastError
tgt.LastDuration, tgt.LastWarning = cur.LastDuration, cur.LastWarning
tgt.RepoSizeHuman, tgt.RepoSizeBytes, tgt.SnapshotCount = cur.RepoSizeHuman, cur.RepoSizeBytes, cur.SnapshotCount
}
if tgt.EscrowState != "escrowed" {
tgt.EscrowState = "pending"
}
if err := m.settings.SetOffboxTarget(tgt); err != nil {
return fmt.Errorf("apply offsite target: %w", err)
}
if stage != nil {
// Best-effort: the offbox is configured + pending regardless. A stage-push failure (agent momentarily
// unreachable) is logged, not fatal — the escrow can be (re-)staged later (operator ceremony / re-enable).
if err := m.PushOffboxPasswordForEscrow(ctx, stage); err != nil {
m.logger.Printf("[WARN] [offbox] apply-offsite: escrow stage push failed (agent unreachable?) — offbox configured pending, re-stage later: %v", err)
}
}
return nil
}
// HashResticPassword is the CANONICAL hasher for the offsite repo password (SLICE 3 hub-verified escrow
// auto-confirm): sha256 hex of the TRIMMED password string — the SAME convention as the agent's
// escrow.HashResticPassword (both sides TrimSpace their file reads; pinned by the SAME cross-repo test
// vector in felhom-agent). The hash of a 256-bit random secret is non-reversible and non-brute-forceable —
// safe to log/compare; the PASSWORD itself is never logged.
func HashResticPassword(pw string) string {
sum := sha256.Sum256([]byte(strings.TrimSpace(pw)))
return hex.EncodeToString(sum[:])
}
// OffboxRepoPasswordHash returns the canonical hash of the local repo password (false when no password
// file exists — nothing to match; the auto-confirm check skips).
func (m *Manager) OffboxRepoPasswordHash() (string, bool) {
pw, err := os.ReadFile(m.offboxPwPath())
if err != nil {
return "", false
}
return HashResticPassword(string(pw)), true
}
// PushOffboxPasswordForEscrow reads the 0600 repo password and hands it to `stage` (the agent push), so
// the web/handler caller never sees the value — used by the enable flow to escrow-stage the offsite key.
func (m *Manager) PushOffboxPasswordForEscrow(ctx context.Context, stage func(ctx context.Context, pw string) error) error {
pw, err := os.ReadFile(m.offboxPwPath())
if err != nil {
return fmt.Errorf("read offbox password: %w", err)
}
return stage(ctx, strings.TrimSpace(string(pw)))
}
// InjectOffboxPassword pre-places a RECOVERED repo password at offboxPwPath (fork-4 DR seam) so a
// subsequent WriteOffboxSecrets uses it instead of generating a new one. Refuses to clobber an existing
// password unless force. Written 0600 via tmp+rename. The value is NEVER logged.
func (m *Manager) InjectOffboxPassword(pw string, force bool) error {
pw = strings.TrimSpace(pw)
if !offboxRepoPwPattern.MatchString(pw) {
return fmt.Errorf("invalid repo password (expected 64 hex characters)")
}
if _, err := os.Stat(m.offboxPwPath()); err == nil && !force {
return fmt.Errorf("a repo password already exists (pass force to overwrite)")
}
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
return fmt.Errorf("offbox dir: %w", err)
}
tmp := m.offboxPwPath() + ".tmp"
if err := os.WriteFile(tmp, []byte(pw), 0o600); err != nil {
return fmt.Errorf("write injected password: %w", err)
}
if err := os.Rename(tmp, m.offboxPwPath()); err != nil {
_ = os.Remove(tmp)
return fmt.Errorf("place injected password: %w", err)
}
return nil
}
// offboxEscrowed reports whether the offsite repo password is confirmed escrowed under R (fork-4).
func (m *Manager) offboxEscrowed() bool {
t := m.settings.GetOffboxTarget()
return t != nil && t.EscrowState == "escrowed"
}
// OffboxRunnable reports whether an off-box RUN may proceed: configured AND escrowed. Config/UI still work
// when not runnable — only actual backup writes are gated (the atomicity guarantee). For the run handler.
func (m *Manager) OffboxRunnable() bool { return m.OffboxConfigured() && m.offboxEscrowed() }
// OffboxCoord returns the non-secret offsite repo coordinates for the DR recipe (fork-4). ok=false when no
// offbox target is configured. NEVER returns the repo password or the SSH key (those are escrowed/regenerable).
func (m *Manager) OffboxCoord() (host, user string, port int, repoPath string, ok bool) {
t := m.settings.GetOffboxTarget()
if t == nil || t.Host == "" || t.User == "" || t.RepoPath == "" {
return "", "", 0, "", false
}
return t.Host, t.User, t.Port, t.RepoPath, true
}
// OffboxEscrowState returns the current escrow state ("" | "pending" | "escrowed") for the UI/handlers.
func (m *Manager) OffboxEscrowState() string {
t := m.settings.GetOffboxTarget()
if t == nil {
return ""
}
return t.EscrowState
}
// offboxBaseArgs builds the restic global args (repo + sftp.args carrying the ConnectTimeout, key, pinned
// known_hosts, port) and the env (RESTIC_PASSWORD_FILE). The ConnectTimeout is MANDATORY (fail-fast).
func (m *Manager) offboxBaseArgs(t *settings.OffboxTarget) ([]string, []string) {
port := t.Port
if port == 0 {
port = 22
}
// restic's sftp backend connects via the `-o sftp.command` SSH invocation (the portable form across
// restic versions — `sftp.args` is not recognized by restic 0.14). The ConnectTimeout makes a dead NAS
// fail in ~N s (the load-bearing spike Q8 knob); StrictHostKeyChecking + a pinned known_hosts avoid
// blind TOFU; BatchMode prevents any interactive prompt from hanging the runner. The value is one -o
// token (restic takes everything after `sftp.command=`); our paths have no spaces (data dir).
sftpCmd := fmt.Sprintf("ssh %s@%s -p %d -oBatchMode=yes -oConnectTimeout=%d -oStrictHostKeyChecking=yes -oUserKnownHostsFile=%s -i %s -s sftp",
t.User, t.Host, port, offboxConnectTimeoutSec, m.offboxKnownHosts(), m.offboxKeyPath())
repo := "sftp:" + t.User + "@" + t.Host + ":" + t.RepoPath
args := []string{"-r", repo, "-o", "sftp.command=" + sftpCmd}
env := []string{"RESTIC_PASSWORD_FILE=" + m.offboxPwPath()}
return args, env
}
// offboxLockRe matches restic's "already locked" error (both the exclusive and shared forms).
var offboxLockRe = regexp.MustCompile(`repository is already locked`)
// unlockStale runs `restic unlock` (stale-only) — cheap pre-run hygiene that removes any lock restic can
// itself prove dead/old. Non-fatal (logged at debug). Called before every offbox run/restore.
func (m *Manager) unlockStale(ctx context.Context, base, env []string) {
uctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
if out, err := m.runner()(uctx, env, append(append([]string{}, base...), "unlock")...); err != nil {
m.logger.Printf("[DEBUG] [offbox] pre-run unlock (stale-only) non-fatal: %v: %s", err, truncate(out))
}
}
// resticStep runs one restic step (backup/prune/restore) under the offbox single-flight guarantee and
// self-heals the C2 crash lock. On a lock error it escalates to `unlock --remove-all` and retries ONCE,
// because THIS controller is the repo's ONLY legitimate writer — per-customer sub-account isolation gives
// one repo one writer, and the in-process single-flight mutex (held by every caller of this method) proves
// no sibling operation is live. Plain `restic unlock` is stale-ONLY and does NOT clear a crash lock: the
// recreated container has a new hostname, so restic can't verify the dead PID and won't treat the lock as
// stale for ~30 min (the overnight-campaign C2 finding — `unlock --remove-all` is required). A second lock
// failure surfaces the error (never loops). BOUNDARY: a DR-cloned SECOND controller writing the same repo
// would defeat the single-writer premise — that is operator-supervised territory (see README), out of scope.
func (m *Manager) resticStep(ctx context.Context, env, base []string, label string, args ...string) ([]byte, error) {
full := append(append([]string{}, base...), args...)
out, err := m.runner()(ctx, env, full...)
if err == nil || !offboxLockRe.Match(out) {
return out, err
}
m.logger.Printf("[WARN] [offbox] cleared a stale exclusive lock left by a previous crash (single-writer repo) before %s; retrying once", label)
uctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
if uout, uerr := m.runner()(uctx, env, append(append([]string{}, base...), "unlock", "--remove-all")...); uerr != nil {
cancel()
m.logger.Printf("[WARN] [offbox] unlock --remove-all failed: %v: %s", uerr, truncate(uout))
return out, err // surface the original lock error (never loop)
}
cancel()
return m.runner()(ctx, env, full...) // retry exactly ONCE
}
// ensureOffboxRepo makes sure the SFTP repo exists: probe `cat config`; if absent, `init` (idempotent —
// a present repo is reused, never re-init). A connect failure surfaces here (fast, via ConnectTimeout).
func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) error {
pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
pout, perr := m.runner()(pctx, env, append(append([]string{}, base...), "cat", "config")...)
switch classifyResticProbe(pout, perr) {
case "": // success — repo good (and clear any stale orphaned flag)
if m.OffboxOrphaned() {
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoState = ""; o.OrphanedAt = "" })
}
return nil
case "orphaned":
// The repo EXISTS but is keyed under a passphrase we no longer have (the reinstall shape). An
// UNCLAIMED (as-delivered) box auto-resets (Scenario B); a CLAIMED box surfaces the orphan card
// and skips until the customer confirms a reset (Scenario C). Move-aside, never delete.
if !m.settings.GetClaimed() {
m.logger.Printf("[INFO] [offbox] orphaned repo on an UNCLAIMED box — auto-resetting (move-aside + re-init)")
if m.offboxOrphanEvent != nil {
m.offboxOrphanEvent("offbox_repo_orphaned", "")
}
if rerr := m.resetOrphanedRepo(ctx, base, env, "auto (unclaimed)"); rerr != nil {
m.markOrphaned() // auto-reset failed → fall back to the orphan card so it isn't silent
return ErrOffboxOrphaned
}
return nil // repo is fresh under the current passphrase → the run proceeds
}
m.markOrphaned()
return ErrOffboxOrphaned
case "norepo":
// No repo at the location → init (the normal first-run path). A race where it already exists is
// treated as success; any other init error is real (e.g. a dead NAS — fail fast).
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer icancel()
out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...)
if err == nil {
m.logger.Printf("[INFO] [offbox] initialized restic repo")
return nil
}
if strings.Contains(string(out), "already initialized") || strings.Contains(string(out), "already exists") {
return nil
}
return fmt.Errorf("offbox repo unreachable / init failed: %w: %s", err, truncate(out))
default: // "other" — network/SFTP-auth/unknown; NOT orphaned. Surface as before (fail fast).
return fmt.Errorf("offbox repo unreachable: %w: %s", perr, truncate(pout))
}
}
// RunOffboxBackup backs up every off-box-toggled app's recovery unit (recovery unit + DB dumps + volume
// tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A
// failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error.
func (m *Manager) RunOffboxBackup(ctx context.Context) error {
return m.runOffboxBackup(ctx, false)
}
// RunOffboxBackupWithProgress is the MANUAL („Távoli mentés most") entry point: identical work, but
// with the live progress sink installed so the page can show total bytes, percent and current app
// (v0.147.0, 4c). The nightly scheduled run keeps calling RunOffboxBackup and stays silent — nobody
// is watching a progress bar at 03:00, and a sink left installed would publish stale percentages
// into a page that never asked for them.
func (m *Manager) RunOffboxBackupWithProgress(ctx context.Context) error {
return m.runOffboxBackup(ctx, true)
}
func (m *Manager) runOffboxBackup(ctx context.Context, withProgress bool) error {
if withProgress {
defer m.beginManualProgress()()
}
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
// fork-4 atomicity gate: no offsite RUN until the repo password is confirmed escrowed under R, so no
// un-recoverable offsite ciphertext can exist. Not an error (config/UI still work) — a skip.
if !m.offboxEscrowed() {
m.logger.Printf("[INFO] [offbox] skipped — pending key escrow (no offsite run until the repo password is escrowed under R)")
return nil
}
// Offsite-repo continuity (v0.142.0): once ORPHANED, scheduled runs SKIP (the event fired on the
// detection transition — no nightly spam) until a reset clears it. The remote page shows the card.
if m.OffboxOrphaned() {
m.logger.Printf("[INFO] [offbox] skipped — offsite repo orphaned (awaiting reset)")
return nil
}
if m.migrationActive() {
m.logger.Printf("[INFO] [offbox] skipped — migration in progress")
return nil
}
if err := m.acquireRunning(); err != nil {
m.logger.Printf("[INFO] [offbox] skipped — another backup is running")
return nil // single-flight: don't race; the next scheduled run retries
}
defer m.releaseRunning()
apps := m.settings.GetOffboxApps()
// Reserved-name defense in depth (R-7b): an app keyed `_shares` would collide with the shares
// leg's restic tag and blocked-set entry. Catalog names cannot realistically produce this, but a
// silent collision would corrupt both sources, so it is refused loudly instead.
for i, a := range apps {
if a == SharesPseudoStack {
m.logger.Printf("[ERROR] [offbox] app %q uses the RESERVED shares key — excluded from the run to protect the shares leg", a)
apps = append(apps[:i:i], apps[i+1:]...)
break
}
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
// Edge-trigger for the enlarge-blocked notification: capture the PRIOR blocked set so we notify only
// apps that NEWLY cross into the blocked state (a persistently-blocked app doesn't re-notify nightly).
priorBlocked := map[string]bool{}
if t != nil {
for _, s := range t.EnlargedBlocked {
priorBlocked[s] = true
}
}
start := time.Now()
m.logger.Printf("[INFO] [offbox] backup run started (%d app(s) toggled)", len(apps))
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastStatus = "running"; o.LastError = "" }); err != nil {
m.logger.Printf("[WARN] [offbox] status persist (running) failed: %v", err)
}
var backedUp int
var missing []string
var runResult offboxRunResult
var runErr error
if usedGB, quota, over := offboxQuotaState(t); over {
// SLICE 4 soft-quota gate (pre-run): NEW backups are refused at ≥100% of the shared-model quota —
// but the retention/prune step STILL RUNS (pruning is the customer's only way back under quota;
// gating it too would deadlock them over-quota) and restore paths are untouched. The CURRENT run's
// gate uses the last-known repo size; a run that crosses 100% mid-flight finishes and the NEXT
// run refuses.
m.offboxPruneOnly(ctx, base, env)
m.offboxRecordStats(ctx, base, env) // the prune may have brought the size back down — refresh
runErr = fmt.Errorf("A távoli mentés túllépte a tárhelykeretet (%d/%d GB) — törölj régi mentéseket vagy kérj nagyobb keretet.", usedGB, quota)
} else {
runResult, runErr = m.runOffboxInternal(ctx, apps, base, env, t)
backedUp = runResult.backedUp
missing = runResult.missing
}
// Sorted names of apps whose enlargement was blocked this run (replaces the persisted set; empty clears).
var blockedNames []string
for _, b := range runResult.blocked {
blockedNames = append(blockedNames, b.stack)
}
sort.Strings(blockedNames)
// No-silent-success: apps were toggled but NOTHING was captured (every unit missing) → promote to a
// hard error so the run reports "error" and the operator is alerted, instead of a misleading ok/0.
if runErr == nil && len(apps) > 0 && backedUp == 0 {
runErr = fmt.Errorf("off-box backup produced no snapshots: %d app(s) toggled but no recovery unit was found on any connected drive (missing: %s)",
len(apps), strings.Join(missing, ", "))
}
dur := time.Since(start)
snapshots := 0
if runErr == nil {
snapshots = m.offboxRecordStats(ctx, base, env)
}
if perr := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.LastRun = time.Now().UTC().Format(time.RFC3339)
o.LastDuration = dur.Round(time.Second).String()
if errors.Is(runErr, ErrOffboxOrphaned) {
// First-detection of the orphaned repo: RepoState (set by markOrphaned) drives the orphan
// card — do NOT surface the raw restic/sentinel text as the last-error banner.
o.LastStatus = "error"
o.LastError = ""
o.LastWarning = ""
} else if runErr != nil {
o.LastStatus = "error"
o.LastError = runErr.Error()
o.LastWarning = ""
} else {
o.LastStatus = "ok"
o.LastError = ""
o.SnapshotCount = snapshots
o.EnlargedBlocked = blockedNames // replace each run (sorted); empty slice clears it
var warns []string
// Zero-toggle honesty (take-two obs.): a configured target with NOTHING selected reports
// its emptiness instead of a bare success — the customer thinks offsite runs, but nothing
// is covered until at least one app is toggled.
// R-7b: the shares leg counts as coverage — a box whose only cloud content is its shares
// must not be told "nothing is selected".
if len(apps) == 0 && !runResult.sharesBackedUp {
warns = append(warns, "Sikeres — nincs mentésre jelölt alkalmazás")
}
if len(missing) > 0 {
warns = append(warns, fmt.Sprintf("Figyelmeztetés: %d alkalmazásnak nincs elérhető mentése, ezek kimaradtak: %s",
len(missing), strings.Join(missing, ", ")))
}
// 3a: capture-gap warnings (structurally-refused / on-disk-missing mandatory paths, undeployed).
warns = append(warns, runResult.warns...)
// 3a: the pre-push enlargement gate blocked some apps' userdata — config+DB still saved.
// R-7b: the shares source is not an "app" and its degraded floor is the DEFINITIONS, not a
// recovery unit — so it gets its own sentence and is excluded from the app count. The
// persisted EnlargedBlocked set keeps the RAW `_shares` key (it is a lookup key the
// templates index by); only this prose maps it through the display vocabulary.
var blockedApps []string
sharesBlocked := false
for _, n := range blockedNames {
if n == SharesPseudoStack {
sharesBlocked = true
continue
}
blockedApps = append(blockedApps, n)
}
if len(blockedApps) > 0 {
warns = append(warns, fmt.Sprintf("Figyelmeztetés: a tárhelykeret miatt %d alkalmazásnál csak konfiguráció- és adatbázis-mentés készült: %s.",
len(blockedApps), strings.Join(blockedApps, ", ")))
}
if sharesBlocked {
warns = append(warns, sharesBlockedWarning())
}
// SLICE 4: approaching the soft quota (≥80%, <100%) — warn on an otherwise-OK run.
if qw := offboxQuotaWarning(o); qw != "" {
warns = append(warns, qw)
}
o.LastWarning = strings.Join(warns, " ")
}
}); perr != nil {
m.logger.Printf("[WARN] [offbox] status persist (final) failed: %v", perr)
}
// The orphaned case has its OWN dedicated event (offbox_repo_orphaned) — do NOT also fire the
// generic backup-failed notification (no double/raw alert; the orphan card is the customer surface).
if m.offboxNotify != nil && !errors.Is(runErr, ErrOffboxOrphaned) {
m.offboxNotify(dur, snapshots, runErr)
}
// Edge-triggered enlarge-blocked notification: only apps that NEWLY crossed into the blocked state
// (vs the prior persisted set) notify — a persistently-blocked app never re-notifies nightly. Uses
// the pre-run last-known repo size (the same figure the gate used).
if runErr == nil && m.offboxEnlargeBlockedNotify != nil && t != nil && t.QuotaGB > 0 {
usedGB := int(t.RepoSizeBytes / offboxGiB)
for _, b := range runResult.blocked {
if !priorBlocked[b.stack] {
// DISPLAY BOUNDARY (R-7b): the notification is a customer-facing surface (it becomes a
// Hungarian e-mail), so the reserved `_shares` key is mapped here — and ONLY here plus
// the warning prose above. The persisted set and the restic tag stay raw.
m.offboxEnlargeBlockedNotify(DisplayStackName(b.stack), b.estBytes, usedGB, t.QuotaGB)
}
}
}
switch {
case errors.Is(runErr, ErrOffboxOrphaned):
m.logger.Printf("[WARN] [offbox] run skipped — offsite repo orphaned (card shown; awaiting reset)")
return nil // the orphaned STATE + event are the signal; not a hard run error for the scheduler
case runErr != nil:
m.logger.Printf("[ERROR] [offbox] backup failed after %s: %v", dur.Round(time.Second), runErr)
case len(missing) > 0:
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s) backed up, %d skipped (no unit), %d snapshot(s), %s",
backedUp, len(missing), snapshots, dur.Round(time.Second))
default:
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s) backed up, %d snapshot(s), %s", backedUp, snapshots, dur.Round(time.Second))
}
return runErr
}
// offboxCandidateNSRoots is the durable, deployment-state-INDEPENDENT set of felhom-data namespace roots
// to search for a recovery unit: every registered SCHEDULABLE (non-decommissioned) storage path ∪ the
// system-data fallback drive, deduped by resolved nsRoot string. This deliberately does NOT consult
// GetAppDrivePath/AppNamespaceRoot — those read the app's LIVE app.yaml HDD_PATH and silently fall back
// to systemDataPath when the app isn't currently deployed, which made offbox look on the wrong drive
// (DIAG root cause). A disconnected drive's path simply isn't present on disk → os.Stat fails → the unit
// is "not here" (correct: a disconnected drive can't be offsited). Boundary: a decommissioned or
// non-schedulable drive is not searched (not an active managed backup location).
func (m *Manager) offboxCandidateNSRoots() []string {
seen := map[string]bool{}
var nsRoots []string
add := func(nr string) {
if nr != "" && !seen[nr] {
seen[nr] = true
nsRoots = append(nsRoots, nr)
}
}
for _, sp := range m.settings.GetSchedulableStoragePaths() {
add(m.namespaceRoot(sp.Path))
}
if m.systemDataPath != "" {
add(m.namespaceRoot(m.systemDataPath))
}
return nsRoots
}
// discoverOffboxUnit locates an app's recovery unit (backups/primary/