Files
felhom-controller/controller/internal/backup/offbox.go
T
admin 2a7deadc93 controller v0.93.0: NAS Part B off-box backup target (restic-over-SFTP)
Encrypted restic repo over SFTP for the app-data tier (the off-site 3-2-1 leg). A dead
NAS fails fast via -oConnectTimeout (spike Q8), never hangs the runner; secrets are 0600
files (ride DR via PBS whole-CT); init-if-absent, retention forget --prune, restore,
single-flight, per-app toggle + UI. restic re-added to the image.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HxLA1mZurFq9kt8hneFeCs
2026-06-30 15:26:38 +02:00

339 lines
13 KiB
Go

package backup
import (
"context"
"crypto/rand"
"encoding/hex"
"encoding/json"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// Off-box (NAS) backup target — Part B. An ENCRYPTED restic repo reached over SFTP (no kernel mount;
// restic talks SFTP to the NAS directly). This is the "1 off-site" leg of 3-2-1 for the app-data tier
// (each off-box app's recovery unit + DB dumps + volume tars), distinct from the local cross-drive rsync
// copy and from the agent's PBS whole-CT DR. The NAS sees only ciphertext.
//
// THE load-bearing lesson (spike Q8): a dead NAS must FAIL FAST, never hang the backup runner — every
// restic invocation carries `-o sftp.args=…-oConnectTimeout=N…` so a black-holed endpoint errors in
// ~N seconds instead of a multi-minute TCP retry. We also check restic's OWN exit code (never
// pipe-swallow). Secrets (repo password + SSH key) live in 0600 files in the data dir — never logged,
// never in a committed/non-0600 file; they ride DR via the PBS whole-CT snapshot of the rootfs.
const (
// offboxConnectTimeoutSec is the SSH ConnectTimeout (spike Q8) — load-bearing fail-fast.
offboxConnectTimeoutSec = 10
// offboxBackupTimeout bounds a full off-box run; offboxProbeTimeout bounds the quick repo probes.
offboxBackupTimeout = 2 * time.Hour
offboxProbeTimeout = 90 * time.Second
)
// offboxRunner is the restic-exec seam (tests inject a fake so no real restic/ssh runs). It runs restic
// with args + extra env and returns combined output + the process error (whose ExitCode the caller checks).
type offboxRunner func(ctx context.Context, env []string, args ...string) ([]byte, error)
func defaultOffboxRunner(ctx context.Context, env []string, args ...string) ([]byte, error) {
cmd := exec.CommandContext(ctx, "restic", args...)
cmd.Env = append(os.Environ(), env...)
return cmd.CombinedOutput()
}
// SetOffboxRunner overrides the restic exec (tests). SetOffboxNotify wires the failure→operator alert.
func (m *Manager) SetOffboxRunner(r offboxRunner) { m.offboxRunner = r }
func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err error)) { m.offboxNotify = fn }
func (m *Manager) runner() offboxRunner {
if m.offboxRunner != nil {
return m.offboxRunner
}
return defaultOffboxRunner
}
func (m *Manager) offboxDir() string { return filepath.Join(m.cfg.Paths.DataDir, "offbox") }
func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir(), "ssh_key") }
func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") }
func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") }
// WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned
// known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator
// (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty
// sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys).
func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
return fmt.Errorf("offbox dir: %w", err)
}
if strings.TrimSpace(sshKey) != "" {
key := sshKey
if !strings.HasSuffix(key, "\n") {
key += "\n"
}
if err := os.WriteFile(m.offboxKeyPath(), []byte(key), 0o600); err != nil {
return fmt.Errorf("offbox ssh key: %w", err)
}
}
if strings.TrimSpace(knownHosts) != "" {
kh := knownHosts
if !strings.HasSuffix(kh, "\n") {
kh += "\n"
}
if err := os.WriteFile(m.offboxKnownHosts(), []byte(kh), 0o644); err != nil {
return fmt.Errorf("offbox known_hosts: %w", err)
}
}
// Auto-generate the repo password once (0600), never log it.
if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) {
pw, gerr := generateOffboxPassword()
if gerr != nil {
return gerr
}
if werr := os.WriteFile(m.offboxPwPath(), []byte(pw), 0o600); werr != nil {
return fmt.Errorf("offbox repo password: %w", werr)
}
}
return nil
}
// generateOffboxPassword returns a 256-bit hex repo password.
func generateOffboxPassword() (string, error) {
b := make([]byte, 32)
if _, err := rand.Read(b); err != nil {
return "", fmt.Errorf("offbox password gen: %w", err)
}
return hex.EncodeToString(b), nil
}
// OffboxConfigured reports whether the target is set, enabled, and the key + password files exist (so the
// UI/scheduler can gate a run without leaking why).
func (m *Manager) OffboxConfigured() bool {
t := m.settings.GetOffboxTarget()
if t == nil || !t.Enabled || t.Host == "" || t.User == "" || t.RepoPath == "" {
return false
}
if _, err := os.Stat(m.offboxKeyPath()); err != nil {
return false
}
if _, err := os.Stat(m.offboxPwPath()); err != nil {
return false
}
return true
}
// offboxBaseArgs builds the restic global args (repo + sftp.args carrying the ConnectTimeout, key, pinned
// known_hosts, port) and the env (RESTIC_PASSWORD_FILE). The ConnectTimeout is MANDATORY (fail-fast).
func (m *Manager) offboxBaseArgs(t *settings.OffboxTarget) ([]string, []string) {
port := t.Port
if port == 0 {
port = 22
}
// One -o sftp.args token; restic splits it on spaces. Our paths have no spaces (data dir). The
// ConnectTimeout makes a dead NAS fail in ~N s; StrictHostKeyChecking + a pinned known_hosts avoid
// blind TOFU; BatchMode prevents any interactive prompt from hanging the runner.
sftpArgs := fmt.Sprintf("-oBatchMode=yes -oConnectTimeout=%d -oStrictHostKeyChecking=yes -oUserKnownHostsFile=%s -oPort=%d -i %s",
offboxConnectTimeoutSec, m.offboxKnownHosts(), port, m.offboxKeyPath())
repo := "sftp:" + t.User + "@" + t.Host + ":" + t.RepoPath
args := []string{"-r", repo, "-o", "sftp.args=" + sftpArgs}
env := []string{"RESTIC_PASSWORD_FILE=" + m.offboxPwPath()}
return args, env
}
// ensureOffboxRepo makes sure the SFTP repo exists: probe `cat config`; if absent, `init` (idempotent —
// a present repo is reused, never re-init). A connect failure surfaces here (fast, via ConnectTimeout).
func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) error {
pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
if _, err := m.runner()(pctx, env, append(append([]string{}, base...), "cat", "config")...); err == nil {
return nil // repo exists
}
// Repo (probably) absent OR unreachable. Try init; if init succeeds the repo was absent. If init
// fails because it already exists (a race), treat as success; otherwise the error is real (e.g. a
// dead NAS — fail fast).
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer icancel()
out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...)
if err == nil {
m.logger.Printf("[INFO] [offbox] initialized restic repo")
return nil
}
if strings.Contains(string(out), "already initialized") || strings.Contains(string(out), "already exists") {
return nil
}
return fmt.Errorf("offbox repo unreachable / init failed: %w: %s", err, truncate(out))
}
// RunOffboxBackup backs up every off-box-toggled app's recovery unit (recovery unit + DB dumps + volume
// tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A
// failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error.
func (m *Manager) RunOffboxBackup(ctx context.Context) error {
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
if m.migrationActive() {
m.logger.Printf("[INFO] [offbox] skipped — migration in progress")
return nil
}
if err := m.acquireRunning(); err != nil {
m.logger.Printf("[INFO] [offbox] skipped — another backup is running")
return nil // single-flight: don't race; the next scheduled run retries
}
defer m.releaseRunning()
apps := m.settings.GetOffboxApps()
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
start := time.Now()
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastStatus = "running"; o.LastError = "" })
runErr := m.runOffboxInternal(ctx, apps, base, env)
dur := time.Since(start)
snapshots := 0
if runErr == nil {
snapshots = m.offboxRecordStats(ctx, base, env)
}
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.LastRun = time.Now().UTC().Format(time.RFC3339)
o.LastDuration = dur.Round(time.Second).String()
if runErr != nil {
o.LastStatus = "error"
o.LastError = runErr.Error()
} else {
o.LastStatus = "ok"
o.LastError = ""
o.SnapshotCount = snapshots
}
})
if m.offboxNotify != nil {
m.offboxNotify(dur, snapshots, runErr)
}
if runErr != nil {
m.logger.Printf("[ERROR] [offbox] backup failed after %s: %v", dur.Round(time.Second), runErr)
} else {
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s), %d snapshot(s), %s", len(apps), snapshots, dur.Round(time.Second))
}
return runErr
}
// runOffboxInternal does the repo-ensure + per-app backup + prune. Caller holds the running flag.
func (m *Manager) runOffboxInternal(ctx context.Context, apps []string, base, env []string) error {
if err := m.ensureOffboxRepo(ctx, base, env); err != nil {
return err // fail fast (dead NAS surfaces here)
}
var firstErr error
for _, stack := range apps {
nsRoot := m.AppNamespaceRoot(stack)
if nsRoot == "" {
continue
}
src := RecoveryUnitPath(nsRoot, stack) // backups/primary/<stack> = recovery unit + db-dumps + vol-tars
if _, err := os.Stat(src); err != nil {
m.logger.Printf("[INFO] [offbox] %s: no backup data yet (%s) — skipping", stack, src)
continue
}
bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
args := append(append([]string{}, base...), "backup", "--tag", "felhom-offbox", "--tag", stack, src)
out, err := m.runner()(bctx, env, args...)
cancel()
if err != nil {
m.logger.Printf("[ERROR] [offbox] backup %s failed: %v: %s", stack, err, truncate(out))
if firstErr == nil {
firstErr = fmt.Errorf("offbox backup %s: %w", stack, err)
}
continue
}
m.logger.Printf("[INFO] [offbox] backed up %s", stack)
}
if firstErr != nil {
return firstErr
}
// Retention: keep a sane window, prune the rest. Repo-wide (grouped by host+paths by default).
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
defer cancel()
args := append(append([]string{}, base...), "forget", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune")
if out, err := m.runner()(fctx, env, args...); err != nil {
// A prune failure is non-fatal to the backup itself (data is safe) — log, don't fail the run.
m.logger.Printf("[WARN] [offbox] forget --prune failed (backups are safe): %v: %s", err, truncate(out))
}
return nil
}
// offboxRecordStats reads the snapshot count (best-effort) for the UI; also fills repo size when stats works.
func (m *Manager) offboxRecordStats(ctx context.Context, base, env []string) int {
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
if err != nil {
return 0
}
var snaps []struct {
ID string `json:"id"`
}
if json.Unmarshal(out, &snaps) != nil {
return 0
}
// Repo size (best-effort, restore-size).
if so, serr := m.runner()(sctx, env, append(append([]string{}, base...), "stats", "--json")...); serr == nil {
var st struct {
TotalSize int64 `json:"total_size"`
}
if json.Unmarshal(so, &st) == nil && st.TotalSize > 0 {
human := humanizeBytes(st.TotalSize)
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoSizeHuman = human })
}
}
return len(snaps)
}
// RestoreOffbox restores an app's latest off-box snapshot to destDir (a scratch/verify location — it does
// NOT overwrite live data). Returns an error on failure (checks restic's own exit code).
func (m *Manager) RestoreOffbox(ctx context.Context, stackName, destDir string) error {
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
if !isSafeStackName(stackName) {
return fmt.Errorf("invalid stack name")
}
if err := os.MkdirAll(destDir, 0o755); err != nil {
return fmt.Errorf("restore dir: %w", err)
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
defer cancel()
args := append(append([]string{}, base...), "restore", "latest", "--tag", stackName, "--target", destDir)
out, err := m.runner()(rctx, env, args...)
if err != nil {
return fmt.Errorf("offbox restore %s: %w: %s", stackName, err, truncate(out))
}
m.logger.Printf("[INFO] [offbox] restored %s → %s", stackName, destDir)
return nil
}
// isSafeStackName guards a stack name used as a restic tag / path component.
func isSafeStackName(s string) bool {
if s == "" || len(s) > 64 {
return false
}
for _, c := range s {
if (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '-' || c == '_' {
continue
}
return false
}
return true
}
// truncate caps subprocess output for a log line + strips a trailing newline.
func truncate(b []byte) string {
s := strings.TrimSpace(string(b))
if len(s) > 400 {
return s[:400] + "…"
}
return s
}