controller v0.93.0: NAS Part B off-box backup target (restic-over-SFTP)
Encrypted restic repo over SFTP for the app-data tier (the off-site 3-2-1 leg). A dead NAS fails fast via -oConnectTimeout (spike Q8), never hangs the runner; secrets are 0600 files (ride DR via PBS whole-CT); init-if-absent, retention forget --prune, restore, single-flight, per-app toggle + UI. restic re-added to the image. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01HxLA1mZurFq9kt8hneFeCs
This commit is contained in:
@@ -0,0 +1,338 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// Off-box (NAS) backup target — Part B. An ENCRYPTED restic repo reached over SFTP (no kernel mount;
|
||||
// restic talks SFTP to the NAS directly). This is the "1 off-site" leg of 3-2-1 for the app-data tier
|
||||
// (each off-box app's recovery unit + DB dumps + volume tars), distinct from the local cross-drive rsync
|
||||
// copy and from the agent's PBS whole-CT DR. The NAS sees only ciphertext.
|
||||
//
|
||||
// THE load-bearing lesson (spike Q8): a dead NAS must FAIL FAST, never hang the backup runner — every
|
||||
// restic invocation carries `-o sftp.args=…-oConnectTimeout=N…` so a black-holed endpoint errors in
|
||||
// ~N seconds instead of a multi-minute TCP retry. We also check restic's OWN exit code (never
|
||||
// pipe-swallow). Secrets (repo password + SSH key) live in 0600 files in the data dir — never logged,
|
||||
// never in a committed/non-0600 file; they ride DR via the PBS whole-CT snapshot of the rootfs.
|
||||
|
||||
const (
|
||||
// offboxConnectTimeoutSec is the SSH ConnectTimeout (spike Q8) — load-bearing fail-fast.
|
||||
offboxConnectTimeoutSec = 10
|
||||
// offboxBackupTimeout bounds a full off-box run; offboxProbeTimeout bounds the quick repo probes.
|
||||
offboxBackupTimeout = 2 * time.Hour
|
||||
offboxProbeTimeout = 90 * time.Second
|
||||
)
|
||||
|
||||
// offboxRunner is the restic-exec seam (tests inject a fake so no real restic/ssh runs). It runs restic
|
||||
// with args + extra env and returns combined output + the process error (whose ExitCode the caller checks).
|
||||
type offboxRunner func(ctx context.Context, env []string, args ...string) ([]byte, error)
|
||||
|
||||
func defaultOffboxRunner(ctx context.Context, env []string, args ...string) ([]byte, error) {
|
||||
cmd := exec.CommandContext(ctx, "restic", args...)
|
||||
cmd.Env = append(os.Environ(), env...)
|
||||
return cmd.CombinedOutput()
|
||||
}
|
||||
|
||||
// SetOffboxRunner overrides the restic exec (tests). SetOffboxNotify wires the failure→operator alert.
|
||||
func (m *Manager) SetOffboxRunner(r offboxRunner) { m.offboxRunner = r }
|
||||
func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err error)) { m.offboxNotify = fn }
|
||||
|
||||
func (m *Manager) runner() offboxRunner {
|
||||
if m.offboxRunner != nil {
|
||||
return m.offboxRunner
|
||||
}
|
||||
return defaultOffboxRunner
|
||||
}
|
||||
|
||||
func (m *Manager) offboxDir() string { return filepath.Join(m.cfg.Paths.DataDir, "offbox") }
|
||||
func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir(), "ssh_key") }
|
||||
func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") }
|
||||
func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") }
|
||||
|
||||
// WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned
|
||||
// known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator
|
||||
// (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty
|
||||
// sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys).
|
||||
func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
|
||||
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
|
||||
return fmt.Errorf("offbox dir: %w", err)
|
||||
}
|
||||
if strings.TrimSpace(sshKey) != "" {
|
||||
key := sshKey
|
||||
if !strings.HasSuffix(key, "\n") {
|
||||
key += "\n"
|
||||
}
|
||||
if err := os.WriteFile(m.offboxKeyPath(), []byte(key), 0o600); err != nil {
|
||||
return fmt.Errorf("offbox ssh key: %w", err)
|
||||
}
|
||||
}
|
||||
if strings.TrimSpace(knownHosts) != "" {
|
||||
kh := knownHosts
|
||||
if !strings.HasSuffix(kh, "\n") {
|
||||
kh += "\n"
|
||||
}
|
||||
if err := os.WriteFile(m.offboxKnownHosts(), []byte(kh), 0o644); err != nil {
|
||||
return fmt.Errorf("offbox known_hosts: %w", err)
|
||||
}
|
||||
}
|
||||
// Auto-generate the repo password once (0600), never log it.
|
||||
if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) {
|
||||
pw, gerr := generateOffboxPassword()
|
||||
if gerr != nil {
|
||||
return gerr
|
||||
}
|
||||
if werr := os.WriteFile(m.offboxPwPath(), []byte(pw), 0o600); werr != nil {
|
||||
return fmt.Errorf("offbox repo password: %w", werr)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// generateOffboxPassword returns a 256-bit hex repo password.
|
||||
func generateOffboxPassword() (string, error) {
|
||||
b := make([]byte, 32)
|
||||
if _, err := rand.Read(b); err != nil {
|
||||
return "", fmt.Errorf("offbox password gen: %w", err)
|
||||
}
|
||||
return hex.EncodeToString(b), nil
|
||||
}
|
||||
|
||||
// OffboxConfigured reports whether the target is set, enabled, and the key + password files exist (so the
|
||||
// UI/scheduler can gate a run without leaking why).
|
||||
func (m *Manager) OffboxConfigured() bool {
|
||||
t := m.settings.GetOffboxTarget()
|
||||
if t == nil || !t.Enabled || t.Host == "" || t.User == "" || t.RepoPath == "" {
|
||||
return false
|
||||
}
|
||||
if _, err := os.Stat(m.offboxKeyPath()); err != nil {
|
||||
return false
|
||||
}
|
||||
if _, err := os.Stat(m.offboxPwPath()); err != nil {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// offboxBaseArgs builds the restic global args (repo + sftp.args carrying the ConnectTimeout, key, pinned
|
||||
// known_hosts, port) and the env (RESTIC_PASSWORD_FILE). The ConnectTimeout is MANDATORY (fail-fast).
|
||||
func (m *Manager) offboxBaseArgs(t *settings.OffboxTarget) ([]string, []string) {
|
||||
port := t.Port
|
||||
if port == 0 {
|
||||
port = 22
|
||||
}
|
||||
// One -o sftp.args token; restic splits it on spaces. Our paths have no spaces (data dir). The
|
||||
// ConnectTimeout makes a dead NAS fail in ~N s; StrictHostKeyChecking + a pinned known_hosts avoid
|
||||
// blind TOFU; BatchMode prevents any interactive prompt from hanging the runner.
|
||||
sftpArgs := fmt.Sprintf("-oBatchMode=yes -oConnectTimeout=%d -oStrictHostKeyChecking=yes -oUserKnownHostsFile=%s -oPort=%d -i %s",
|
||||
offboxConnectTimeoutSec, m.offboxKnownHosts(), port, m.offboxKeyPath())
|
||||
repo := "sftp:" + t.User + "@" + t.Host + ":" + t.RepoPath
|
||||
args := []string{"-r", repo, "-o", "sftp.args=" + sftpArgs}
|
||||
env := []string{"RESTIC_PASSWORD_FILE=" + m.offboxPwPath()}
|
||||
return args, env
|
||||
}
|
||||
|
||||
// ensureOffboxRepo makes sure the SFTP repo exists: probe `cat config`; if absent, `init` (idempotent —
|
||||
// a present repo is reused, never re-init). A connect failure surfaces here (fast, via ConnectTimeout).
|
||||
func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) error {
|
||||
pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||||
defer cancel()
|
||||
if _, err := m.runner()(pctx, env, append(append([]string{}, base...), "cat", "config")...); err == nil {
|
||||
return nil // repo exists
|
||||
}
|
||||
// Repo (probably) absent OR unreachable. Try init; if init succeeds the repo was absent. If init
|
||||
// fails because it already exists (a race), treat as success; otherwise the error is real (e.g. a
|
||||
// dead NAS — fail fast).
|
||||
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||||
defer icancel()
|
||||
out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...)
|
||||
if err == nil {
|
||||
m.logger.Printf("[INFO] [offbox] initialized restic repo")
|
||||
return nil
|
||||
}
|
||||
if strings.Contains(string(out), "already initialized") || strings.Contains(string(out), "already exists") {
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("offbox repo unreachable / init failed: %w: %s", err, truncate(out))
|
||||
}
|
||||
|
||||
// RunOffboxBackup backs up every off-box-toggled app's recovery unit (recovery unit + DB dumps + volume
|
||||
// tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A
|
||||
// failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error.
|
||||
func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||||
if !m.OffboxConfigured() {
|
||||
return fmt.Errorf("off-box backup not configured")
|
||||
}
|
||||
if m.migrationActive() {
|
||||
m.logger.Printf("[INFO] [offbox] skipped — migration in progress")
|
||||
return nil
|
||||
}
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
m.logger.Printf("[INFO] [offbox] skipped — another backup is running")
|
||||
return nil // single-flight: don't race; the next scheduled run retries
|
||||
}
|
||||
defer m.releaseRunning()
|
||||
|
||||
apps := m.settings.GetOffboxApps()
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
start := time.Now()
|
||||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastStatus = "running"; o.LastError = "" })
|
||||
|
||||
runErr := m.runOffboxInternal(ctx, apps, base, env)
|
||||
|
||||
dur := time.Since(start)
|
||||
snapshots := 0
|
||||
if runErr == nil {
|
||||
snapshots = m.offboxRecordStats(ctx, base, env)
|
||||
}
|
||||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||||
o.LastRun = time.Now().UTC().Format(time.RFC3339)
|
||||
o.LastDuration = dur.Round(time.Second).String()
|
||||
if runErr != nil {
|
||||
o.LastStatus = "error"
|
||||
o.LastError = runErr.Error()
|
||||
} else {
|
||||
o.LastStatus = "ok"
|
||||
o.LastError = ""
|
||||
o.SnapshotCount = snapshots
|
||||
}
|
||||
})
|
||||
if m.offboxNotify != nil {
|
||||
m.offboxNotify(dur, snapshots, runErr)
|
||||
}
|
||||
if runErr != nil {
|
||||
m.logger.Printf("[ERROR] [offbox] backup failed after %s: %v", dur.Round(time.Second), runErr)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s), %d snapshot(s), %s", len(apps), snapshots, dur.Round(time.Second))
|
||||
}
|
||||
return runErr
|
||||
}
|
||||
|
||||
// runOffboxInternal does the repo-ensure + per-app backup + prune. Caller holds the running flag.
|
||||
func (m *Manager) runOffboxInternal(ctx context.Context, apps []string, base, env []string) error {
|
||||
if err := m.ensureOffboxRepo(ctx, base, env); err != nil {
|
||||
return err // fail fast (dead NAS surfaces here)
|
||||
}
|
||||
var firstErr error
|
||||
for _, stack := range apps {
|
||||
nsRoot := m.AppNamespaceRoot(stack)
|
||||
if nsRoot == "" {
|
||||
continue
|
||||
}
|
||||
src := RecoveryUnitPath(nsRoot, stack) // backups/primary/<stack> = recovery unit + db-dumps + vol-tars
|
||||
if _, err := os.Stat(src); err != nil {
|
||||
m.logger.Printf("[INFO] [offbox] %s: no backup data yet (%s) — skipping", stack, src)
|
||||
continue
|
||||
}
|
||||
bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||||
args := append(append([]string{}, base...), "backup", "--tag", "felhom-offbox", "--tag", stack, src)
|
||||
out, err := m.runner()(bctx, env, args...)
|
||||
cancel()
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [offbox] backup %s failed: %v: %s", stack, err, truncate(out))
|
||||
if firstErr == nil {
|
||||
firstErr = fmt.Errorf("offbox backup %s: %w", stack, err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] backed up %s", stack)
|
||||
}
|
||||
if firstErr != nil {
|
||||
return firstErr
|
||||
}
|
||||
// Retention: keep a sane window, prune the rest. Repo-wide (grouped by host+paths by default).
|
||||
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||||
defer cancel()
|
||||
args := append(append([]string{}, base...), "forget", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune")
|
||||
if out, err := m.runner()(fctx, env, args...); err != nil {
|
||||
// A prune failure is non-fatal to the backup itself (data is safe) — log, don't fail the run.
|
||||
m.logger.Printf("[WARN] [offbox] forget --prune failed (backups are safe): %v: %s", err, truncate(out))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// offboxRecordStats reads the snapshot count (best-effort) for the UI; also fills repo size when stats works.
|
||||
func (m *Manager) offboxRecordStats(ctx context.Context, base, env []string) int {
|
||||
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||||
defer cancel()
|
||||
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
var snaps []struct {
|
||||
ID string `json:"id"`
|
||||
}
|
||||
if json.Unmarshal(out, &snaps) != nil {
|
||||
return 0
|
||||
}
|
||||
// Repo size (best-effort, restore-size).
|
||||
if so, serr := m.runner()(sctx, env, append(append([]string{}, base...), "stats", "--json")...); serr == nil {
|
||||
var st struct {
|
||||
TotalSize int64 `json:"total_size"`
|
||||
}
|
||||
if json.Unmarshal(so, &st) == nil && st.TotalSize > 0 {
|
||||
human := humanizeBytes(st.TotalSize)
|
||||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoSizeHuman = human })
|
||||
}
|
||||
}
|
||||
return len(snaps)
|
||||
}
|
||||
|
||||
// RestoreOffbox restores an app's latest off-box snapshot to destDir (a scratch/verify location — it does
|
||||
// NOT overwrite live data). Returns an error on failure (checks restic's own exit code).
|
||||
func (m *Manager) RestoreOffbox(ctx context.Context, stackName, destDir string) error {
|
||||
if !m.OffboxConfigured() {
|
||||
return fmt.Errorf("off-box backup not configured")
|
||||
}
|
||||
if !isSafeStackName(stackName) {
|
||||
return fmt.Errorf("invalid stack name")
|
||||
}
|
||||
if err := os.MkdirAll(destDir, 0o755); err != nil {
|
||||
return fmt.Errorf("restore dir: %w", err)
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||||
defer cancel()
|
||||
args := append(append([]string{}, base...), "restore", "latest", "--tag", stackName, "--target", destDir)
|
||||
out, err := m.runner()(rctx, env, args...)
|
||||
if err != nil {
|
||||
return fmt.Errorf("offbox restore %s: %w: %s", stackName, err, truncate(out))
|
||||
}
|
||||
m.logger.Printf("[INFO] [offbox] restored %s → %s", stackName, destDir)
|
||||
return nil
|
||||
}
|
||||
|
||||
// isSafeStackName guards a stack name used as a restic tag / path component.
|
||||
func isSafeStackName(s string) bool {
|
||||
if s == "" || len(s) > 64 {
|
||||
return false
|
||||
}
|
||||
for _, c := range s {
|
||||
if (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '-' || c == '_' {
|
||||
continue
|
||||
}
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// truncate caps subprocess output for a log line + strips a trailing newline.
|
||||
func truncate(b []byte) string {
|
||||
s := strings.TrimSpace(string(b))
|
||||
if len(s) > 400 {
|
||||
return s[:400] + "…"
|
||||
}
|
||||
return s
|
||||
}
|
||||
Reference in New Issue
Block a user