package backup import ( "context" "crypto/rand" "encoding/hex" "encoding/json" "fmt" "os" "os/exec" "path/filepath" "strings" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // Off-box (NAS) backup target — Part B. An ENCRYPTED restic repo reached over SFTP (no kernel mount; // restic talks SFTP to the NAS directly). This is the "1 off-site" leg of 3-2-1 for the app-data tier // (each off-box app's recovery unit + DB dumps + volume tars), distinct from the local cross-drive rsync // copy and from the agent's PBS whole-CT DR. The NAS sees only ciphertext. // // THE load-bearing lesson (spike Q8): a dead NAS must FAIL FAST, never hang the backup runner — every // restic invocation carries `-o sftp.args=…-oConnectTimeout=N…` so a black-holed endpoint errors in // ~N seconds instead of a multi-minute TCP retry. We also check restic's OWN exit code (never // pipe-swallow). Secrets (repo password + SSH key) live in 0600 files in the data dir — never logged, // never in a committed/non-0600 file; they ride DR via the PBS whole-CT snapshot of the rootfs. const ( // offboxConnectTimeoutSec is the SSH ConnectTimeout (spike Q8) — load-bearing fail-fast. offboxConnectTimeoutSec = 10 // offboxBackupTimeout bounds a full off-box run; offboxProbeTimeout bounds the quick repo probes. offboxBackupTimeout = 2 * time.Hour offboxProbeTimeout = 90 * time.Second ) // offboxRunner is the restic-exec seam (tests inject a fake so no real restic/ssh runs). It runs restic // with args + extra env and returns combined output + the process error (whose ExitCode the caller checks). type offboxRunner func(ctx context.Context, env []string, args ...string) ([]byte, error) func defaultOffboxRunner(ctx context.Context, env []string, args ...string) ([]byte, error) { cmd := exec.CommandContext(ctx, "restic", args...) cmd.Env = append(os.Environ(), env...) return cmd.CombinedOutput() } // SetOffboxRunner overrides the restic exec (tests). SetOffboxNotify wires the failure→operator alert. func (m *Manager) SetOffboxRunner(r offboxRunner) { m.offboxRunner = r } func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err error)) { m.offboxNotify = fn } func (m *Manager) runner() offboxRunner { if m.offboxRunner != nil { return m.offboxRunner } return defaultOffboxRunner } func (m *Manager) offboxDir() string { return filepath.Join(m.cfg.Paths.DataDir, "offbox") } func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir(), "ssh_key") } func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") } func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") } // WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned // known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator // (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty // sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys). func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error { if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil { return fmt.Errorf("offbox dir: %w", err) } if strings.TrimSpace(sshKey) != "" { key := sshKey if !strings.HasSuffix(key, "\n") { key += "\n" } if err := os.WriteFile(m.offboxKeyPath(), []byte(key), 0o600); err != nil { return fmt.Errorf("offbox ssh key: %w", err) } } if strings.TrimSpace(knownHosts) != "" { kh := knownHosts if !strings.HasSuffix(kh, "\n") { kh += "\n" } if err := os.WriteFile(m.offboxKnownHosts(), []byte(kh), 0o644); err != nil { return fmt.Errorf("offbox known_hosts: %w", err) } } // Auto-generate the repo password once (0600), never log it. if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) { pw, gerr := generateOffboxPassword() if gerr != nil { return gerr } if werr := os.WriteFile(m.offboxPwPath(), []byte(pw), 0o600); werr != nil { return fmt.Errorf("offbox repo password: %w", werr) } } return nil } // generateOffboxPassword returns a 256-bit hex repo password. func generateOffboxPassword() (string, error) { b := make([]byte, 32) if _, err := rand.Read(b); err != nil { return "", fmt.Errorf("offbox password gen: %w", err) } return hex.EncodeToString(b), nil } // OffboxConfigured reports whether the target is set, enabled, and the key + password files exist (so the // UI/scheduler can gate a run without leaking why). func (m *Manager) OffboxConfigured() bool { t := m.settings.GetOffboxTarget() if t == nil || !t.Enabled || t.Host == "" || t.User == "" || t.RepoPath == "" { return false } if _, err := os.Stat(m.offboxKeyPath()); err != nil { return false } if _, err := os.Stat(m.offboxPwPath()); err != nil { return false } return true } // offboxBaseArgs builds the restic global args (repo + sftp.args carrying the ConnectTimeout, key, pinned // known_hosts, port) and the env (RESTIC_PASSWORD_FILE). The ConnectTimeout is MANDATORY (fail-fast). func (m *Manager) offboxBaseArgs(t *settings.OffboxTarget) ([]string, []string) { port := t.Port if port == 0 { port = 22 } // One -o sftp.args token; restic splits it on spaces. Our paths have no spaces (data dir). The // ConnectTimeout makes a dead NAS fail in ~N s; StrictHostKeyChecking + a pinned known_hosts avoid // blind TOFU; BatchMode prevents any interactive prompt from hanging the runner. sftpArgs := fmt.Sprintf("-oBatchMode=yes -oConnectTimeout=%d -oStrictHostKeyChecking=yes -oUserKnownHostsFile=%s -oPort=%d -i %s", offboxConnectTimeoutSec, m.offboxKnownHosts(), port, m.offboxKeyPath()) repo := "sftp:" + t.User + "@" + t.Host + ":" + t.RepoPath args := []string{"-r", repo, "-o", "sftp.args=" + sftpArgs} env := []string{"RESTIC_PASSWORD_FILE=" + m.offboxPwPath()} return args, env } // ensureOffboxRepo makes sure the SFTP repo exists: probe `cat config`; if absent, `init` (idempotent — // a present repo is reused, never re-init). A connect failure surfaces here (fast, via ConnectTimeout). func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) error { pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout) defer cancel() if _, err := m.runner()(pctx, env, append(append([]string{}, base...), "cat", "config")...); err == nil { return nil // repo exists } // Repo (probably) absent OR unreachable. Try init; if init succeeds the repo was absent. If init // fails because it already exists (a race), treat as success; otherwise the error is real (e.g. a // dead NAS — fail fast). ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout) defer icancel() out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...) if err == nil { m.logger.Printf("[INFO] [offbox] initialized restic repo") return nil } if strings.Contains(string(out), "already initialized") || strings.Contains(string(out), "already exists") { return nil } return fmt.Errorf("offbox repo unreachable / init failed: %w: %s", err, truncate(out)) } // RunOffboxBackup backs up every off-box-toggled app's recovery unit (recovery unit + DB dumps + volume // tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A // failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error. func (m *Manager) RunOffboxBackup(ctx context.Context) error { if !m.OffboxConfigured() { return fmt.Errorf("off-box backup not configured") } if m.migrationActive() { m.logger.Printf("[INFO] [offbox] skipped — migration in progress") return nil } if err := m.acquireRunning(); err != nil { m.logger.Printf("[INFO] [offbox] skipped — another backup is running") return nil // single-flight: don't race; the next scheduled run retries } defer m.releaseRunning() apps := m.settings.GetOffboxApps() t := m.settings.GetOffboxTarget() base, env := m.offboxBaseArgs(t) start := time.Now() _ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastStatus = "running"; o.LastError = "" }) runErr := m.runOffboxInternal(ctx, apps, base, env) dur := time.Since(start) snapshots := 0 if runErr == nil { snapshots = m.offboxRecordStats(ctx, base, env) } _ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastRun = time.Now().UTC().Format(time.RFC3339) o.LastDuration = dur.Round(time.Second).String() if runErr != nil { o.LastStatus = "error" o.LastError = runErr.Error() } else { o.LastStatus = "ok" o.LastError = "" o.SnapshotCount = snapshots } }) if m.offboxNotify != nil { m.offboxNotify(dur, snapshots, runErr) } if runErr != nil { m.logger.Printf("[ERROR] [offbox] backup failed after %s: %v", dur.Round(time.Second), runErr) } else { m.logger.Printf("[INFO] [offbox] backup OK: %d app(s), %d snapshot(s), %s", len(apps), snapshots, dur.Round(time.Second)) } return runErr } // runOffboxInternal does the repo-ensure + per-app backup + prune. Caller holds the running flag. func (m *Manager) runOffboxInternal(ctx context.Context, apps []string, base, env []string) error { if err := m.ensureOffboxRepo(ctx, base, env); err != nil { return err // fail fast (dead NAS surfaces here) } var firstErr error for _, stack := range apps { nsRoot := m.AppNamespaceRoot(stack) if nsRoot == "" { continue } src := RecoveryUnitPath(nsRoot, stack) // backups/primary/ = recovery unit + db-dumps + vol-tars if _, err := os.Stat(src); err != nil { m.logger.Printf("[INFO] [offbox] %s: no backup data yet (%s) — skipping", stack, src) continue } bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout) args := append(append([]string{}, base...), "backup", "--tag", "felhom-offbox", "--tag", stack, src) out, err := m.runner()(bctx, env, args...) cancel() if err != nil { m.logger.Printf("[ERROR] [offbox] backup %s failed: %v: %s", stack, err, truncate(out)) if firstErr == nil { firstErr = fmt.Errorf("offbox backup %s: %w", stack, err) } continue } m.logger.Printf("[INFO] [offbox] backed up %s", stack) } if firstErr != nil { return firstErr } // Retention: keep a sane window, prune the rest. Repo-wide (grouped by host+paths by default). fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout) defer cancel() args := append(append([]string{}, base...), "forget", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune") if out, err := m.runner()(fctx, env, args...); err != nil { // A prune failure is non-fatal to the backup itself (data is safe) — log, don't fail the run. m.logger.Printf("[WARN] [offbox] forget --prune failed (backups are safe): %v: %s", err, truncate(out)) } return nil } // offboxRecordStats reads the snapshot count (best-effort) for the UI; also fills repo size when stats works. func (m *Manager) offboxRecordStats(ctx context.Context, base, env []string) int { sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout) defer cancel() out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...) if err != nil { return 0 } var snaps []struct { ID string `json:"id"` } if json.Unmarshal(out, &snaps) != nil { return 0 } // Repo size (best-effort, restore-size). if so, serr := m.runner()(sctx, env, append(append([]string{}, base...), "stats", "--json")...); serr == nil { var st struct { TotalSize int64 `json:"total_size"` } if json.Unmarshal(so, &st) == nil && st.TotalSize > 0 { human := humanizeBytes(st.TotalSize) _ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoSizeHuman = human }) } } return len(snaps) } // RestoreOffbox restores an app's latest off-box snapshot to destDir (a scratch/verify location — it does // NOT overwrite live data). Returns an error on failure (checks restic's own exit code). func (m *Manager) RestoreOffbox(ctx context.Context, stackName, destDir string) error { if !m.OffboxConfigured() { return fmt.Errorf("off-box backup not configured") } if !isSafeStackName(stackName) { return fmt.Errorf("invalid stack name") } if err := os.MkdirAll(destDir, 0o755); err != nil { return fmt.Errorf("restore dir: %w", err) } t := m.settings.GetOffboxTarget() base, env := m.offboxBaseArgs(t) rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout) defer cancel() args := append(append([]string{}, base...), "restore", "latest", "--tag", stackName, "--target", destDir) out, err := m.runner()(rctx, env, args...) if err != nil { return fmt.Errorf("offbox restore %s: %w: %s", stackName, err, truncate(out)) } m.logger.Printf("[INFO] [offbox] restored %s → %s", stackName, destDir) return nil } // isSafeStackName guards a stack name used as a restic tag / path component. func isSafeStackName(s string) bool { if s == "" || len(s) > 64 { return false } for _, c := range s { if (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '-' || c == '_' { continue } return false } return true } // truncate caps subprocess output for a log line + strips a trailing newline. func truncate(b []byte) string { s := strings.TrimSpace(string(b)) if len(s) > 400 { return s[:400] + "…" } return s }