cb9e992c7a
resticStep escalates a restic lock error to `unlock --remove-all` + one retry (safe: single-writer repo — sub-account isolation + single-flight mutex); plain `unlock` is stale-only and can't clear a crash lock across a container-hostname change. Pre-run stale unlock hygiene on run+restore. C1: NewManager flips a persisted LastStatus=running to a truthful error. Both red-proofed (A reproduces the exact campaign backup failure). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01PSK5g6qYLknKj8u3QAFEr6
749 lines
34 KiB
Go
749 lines
34 KiB
Go
package backup
|
||
|
||
import (
|
||
"context"
|
||
"crypto/rand"
|
||
"crypto/sha256"
|
||
"encoding/hex"
|
||
"encoding/json"
|
||
"fmt"
|
||
"os"
|
||
"os/exec"
|
||
"path/filepath"
|
||
"regexp"
|
||
"strings"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||
)
|
||
|
||
// Off-box (NAS) backup target — Part B. An ENCRYPTED restic repo reached over SFTP (no kernel mount;
|
||
// restic talks SFTP to the NAS directly). This is the "1 off-site" leg of 3-2-1 for the app-data tier
|
||
// (each off-box app's recovery unit + DB dumps + volume tars), distinct from the local cross-drive rsync
|
||
// copy and from the agent's PBS whole-CT DR. The NAS sees only ciphertext.
|
||
//
|
||
// THE load-bearing lesson (spike Q8): a dead NAS must FAIL FAST, never hang the backup runner — every
|
||
// restic invocation carries `-o sftp.args=…-oConnectTimeout=N…` so a black-holed endpoint errors in
|
||
// ~N seconds instead of a multi-minute TCP retry. We also check restic's OWN exit code (never
|
||
// pipe-swallow). Secrets (repo password + SSH key) live in 0600 files in the data dir — never logged,
|
||
// never in a committed/non-0600 file; they ride DR via the PBS whole-CT snapshot of the rootfs.
|
||
|
||
const (
|
||
// offboxConnectTimeoutSec is the SSH ConnectTimeout (spike Q8) — load-bearing fail-fast.
|
||
offboxConnectTimeoutSec = 10
|
||
// offboxBackupTimeout bounds a full off-box run; offboxProbeTimeout bounds the quick repo probes.
|
||
offboxBackupTimeout = 2 * time.Hour
|
||
offboxProbeTimeout = 90 * time.Second
|
||
)
|
||
|
||
// offboxRunner is the restic-exec seam (tests inject a fake so no real restic/ssh runs). It runs restic
|
||
// with args + extra env and returns combined output + the process error (whose ExitCode the caller checks).
|
||
type offboxRunner func(ctx context.Context, env []string, args ...string) ([]byte, error)
|
||
|
||
func defaultOffboxRunner(ctx context.Context, env []string, args ...string) ([]byte, error) {
|
||
cmd := exec.CommandContext(ctx, "restic", args...)
|
||
cmd.Env = append(os.Environ(), env...)
|
||
return cmd.CombinedOutput()
|
||
}
|
||
|
||
// SetOffboxRunner overrides the restic exec (tests). SetOffboxNotify wires the failure→operator alert.
|
||
func (m *Manager) SetOffboxRunner(r offboxRunner) { m.offboxRunner = r }
|
||
func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err error)) { m.offboxNotify = fn }
|
||
|
||
func (m *Manager) runner() offboxRunner {
|
||
if m.offboxRunner != nil {
|
||
return m.offboxRunner
|
||
}
|
||
return defaultOffboxRunner
|
||
}
|
||
|
||
func (m *Manager) offboxDir() string { return filepath.Join(m.cfg.Paths.DataDir, "offbox") }
|
||
func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir(), "ssh_key") }
|
||
func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") }
|
||
func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") }
|
||
|
||
// WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned
|
||
// known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator
|
||
// (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty
|
||
// sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys).
|
||
func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
|
||
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
|
||
return fmt.Errorf("offbox dir: %w", err)
|
||
}
|
||
if strings.TrimSpace(sshKey) != "" {
|
||
key := sshKey
|
||
if !strings.HasSuffix(key, "\n") {
|
||
key += "\n"
|
||
}
|
||
if err := os.WriteFile(m.offboxKeyPath(), []byte(key), 0o600); err != nil {
|
||
return fmt.Errorf("offbox ssh key: %w", err)
|
||
}
|
||
}
|
||
if strings.TrimSpace(knownHosts) != "" {
|
||
kh := knownHosts
|
||
if !strings.HasSuffix(kh, "\n") {
|
||
kh += "\n"
|
||
}
|
||
if err := os.WriteFile(m.offboxKnownHosts(), []byte(kh), 0o644); err != nil {
|
||
return fmt.Errorf("offbox known_hosts: %w", err)
|
||
}
|
||
}
|
||
// Auto-generate the repo password once (0600), never log it.
|
||
if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) {
|
||
pw, gerr := generateOffboxPassword()
|
||
if gerr != nil {
|
||
return gerr
|
||
}
|
||
if werr := os.WriteFile(m.offboxPwPath(), []byte(pw), 0o600); werr != nil {
|
||
return fmt.Errorf("offbox repo password: %w", werr)
|
||
}
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// generateOffboxPassword returns a 256-bit hex repo password.
|
||
func generateOffboxPassword() (string, error) {
|
||
b := make([]byte, 32)
|
||
if _, err := rand.Read(b); err != nil {
|
||
return "", fmt.Errorf("offbox password gen: %w", err)
|
||
}
|
||
return hex.EncodeToString(b), nil
|
||
}
|
||
|
||
// Off-box target field validation — the security boundary for the values that flow into the `ssh … -s
|
||
// sftp` command restic runs. Host/user are charset-restricted AND must not start with '-' (an ssh
|
||
// OPTION-INJECTION vector: a host like "-oProxyCommand=evil" would make ssh execute an arbitrary command).
|
||
// RepoPath is an absolute, traversal-free, metacharacter-free path. This mirrors the agent's validate.go
|
||
// discipline: validate before any value reaches an exec.
|
||
var (
|
||
reOffboxHost = regexp.MustCompile(`^[A-Za-z0-9._-]+$`)
|
||
reOffboxUser = regexp.MustCompile(`^[A-Za-z0-9._-]+$`)
|
||
reOffboxPath = regexp.MustCompile(`^/[A-Za-z0-9._/-]+$`)
|
||
)
|
||
|
||
// ValidateOffboxTarget rejects values that could inject into the ssh command line (option injection via a
|
||
// leading '-', shell/space metacharacters, path traversal). Returns nil for a safe target.
|
||
func ValidateOffboxTarget(t *settings.OffboxTarget) error {
|
||
if t == nil {
|
||
return fmt.Errorf("no off-box target")
|
||
}
|
||
if t.Host == "" || len(t.Host) > 255 || !reOffboxHost.MatchString(t.Host) || strings.HasPrefix(t.Host, "-") || strings.HasPrefix(t.Host, ".") {
|
||
return fmt.Errorf("invalid NAS host (letters, digits, '.', '-', '_'; must not start with '-' or '.')")
|
||
}
|
||
if t.User == "" || len(t.User) > 64 || !reOffboxUser.MatchString(t.User) || strings.HasPrefix(t.User, "-") {
|
||
return fmt.Errorf("invalid user (letters, digits, '.', '-', '_'; must not start with '-')")
|
||
}
|
||
if len(t.RepoPath) > 512 || !reOffboxPath.MatchString(t.RepoPath) || strings.Contains(t.RepoPath, "..") {
|
||
return fmt.Errorf("invalid repo path (absolute, no spaces/metacharacters, no '..')")
|
||
}
|
||
if p := t.Port; p != 0 && (p < 1 || p > 65535) {
|
||
return fmt.Errorf("invalid port")
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// OffboxConfigured reports whether the target is set, enabled, VALID, and the key + password files exist
|
||
// (so the UI/scheduler can gate a run without leaking why). A target that fails validation is treated as
|
||
// not-configured — fail-closed, so a bad/hostile persisted target can never reach the ssh exec.
|
||
func (m *Manager) OffboxConfigured() bool {
|
||
t := m.settings.GetOffboxTarget()
|
||
if t == nil || !t.Enabled || t.Host == "" || t.User == "" || t.RepoPath == "" {
|
||
return false
|
||
}
|
||
if err := ValidateOffboxTarget(t); err != nil {
|
||
return false
|
||
}
|
||
if _, err := os.Stat(m.offboxKeyPath()); err != nil {
|
||
return false
|
||
}
|
||
if _, err := os.Stat(m.offboxPwPath()); err != nil {
|
||
return false
|
||
}
|
||
return true
|
||
}
|
||
|
||
// offboxRepoPwPattern matches a valid restic repo password (generateOffboxPassword = 32 rand bytes → 64 hex).
|
||
var offboxRepoPwPattern = regexp.MustCompile(`^[0-9a-fA-F]{64}$`)
|
||
|
||
// ApplyOffsiteTarget configures the offbox target from a hub-provisioned descriptor (SLICE 2 apply-bridge):
|
||
// it writes the 0600 SSH key + pinned known_hosts, sets the target with EscrowState="pending", and pushes
|
||
// the repo password to the agent for escrow — the SAME fork-4 enable path a manual config takes. `stage` is
|
||
// the agent escrow-stage push (nil skips it, e.g. when the agent is unreachable — the run gate still holds).
|
||
func (m *Manager) ApplyOffsiteTarget(ctx context.Context, tgt *settings.OffboxTarget, sshKeyPEM, knownHosts string, stage func(ctx context.Context, pw string) error) error {
|
||
if err := m.WriteOffboxSecrets(sshKeyPEM, knownHosts); err != nil {
|
||
return fmt.Errorf("apply offsite secrets: %w", err)
|
||
}
|
||
// Re-apply (v0.109.1 live finding): the bridge rebuilds the target from the descriptor, but the
|
||
// EXISTING target's custody + runtime status must carry over — EscrowState tracks the REPO PASSWORD
|
||
// (preserved by WriteOffboxSecrets above, never rotated by this path), not the target coords; and the
|
||
// status fields belong to the runner. Without this, a quota bump demoted an escrowed demo target to
|
||
// pending and wiped its history (which would also false-trigger the hub's staleness alert).
|
||
if cur := m.settings.GetOffboxTarget(); cur != nil {
|
||
tgt.EscrowState = cur.EscrowState
|
||
tgt.LastRun, tgt.LastStatus, tgt.LastError = cur.LastRun, cur.LastStatus, cur.LastError
|
||
tgt.LastDuration, tgt.LastWarning = cur.LastDuration, cur.LastWarning
|
||
tgt.RepoSizeHuman, tgt.RepoSizeBytes, tgt.SnapshotCount = cur.RepoSizeHuman, cur.RepoSizeBytes, cur.SnapshotCount
|
||
}
|
||
if tgt.EscrowState != "escrowed" {
|
||
tgt.EscrowState = "pending"
|
||
}
|
||
if err := m.settings.SetOffboxTarget(tgt); err != nil {
|
||
return fmt.Errorf("apply offsite target: %w", err)
|
||
}
|
||
if stage != nil {
|
||
// Best-effort: the offbox is configured + pending regardless. A stage-push failure (agent momentarily
|
||
// unreachable) is logged, not fatal — the escrow can be (re-)staged later (operator ceremony / re-enable).
|
||
if err := m.PushOffboxPasswordForEscrow(ctx, stage); err != nil {
|
||
m.logger.Printf("[WARN] [offbox] apply-offsite: escrow stage push failed (agent unreachable?) — offbox configured pending, re-stage later: %v", err)
|
||
}
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// HashResticPassword is the CANONICAL hasher for the offsite repo password (SLICE 3 hub-verified escrow
|
||
// auto-confirm): sha256 hex of the TRIMMED password string — the SAME convention as the agent's
|
||
// escrow.HashResticPassword (both sides TrimSpace their file reads; pinned by the SAME cross-repo test
|
||
// vector in felhom-agent). The hash of a 256-bit random secret is non-reversible and non-brute-forceable —
|
||
// safe to log/compare; the PASSWORD itself is never logged.
|
||
func HashResticPassword(pw string) string {
|
||
sum := sha256.Sum256([]byte(strings.TrimSpace(pw)))
|
||
return hex.EncodeToString(sum[:])
|
||
}
|
||
|
||
// OffboxRepoPasswordHash returns the canonical hash of the local repo password (false when no password
|
||
// file exists — nothing to match; the auto-confirm check skips).
|
||
func (m *Manager) OffboxRepoPasswordHash() (string, bool) {
|
||
pw, err := os.ReadFile(m.offboxPwPath())
|
||
if err != nil {
|
||
return "", false
|
||
}
|
||
return HashResticPassword(string(pw)), true
|
||
}
|
||
|
||
// PushOffboxPasswordForEscrow reads the 0600 repo password and hands it to `stage` (the agent push), so
|
||
// the web/handler caller never sees the value — used by the enable flow to escrow-stage the offsite key.
|
||
func (m *Manager) PushOffboxPasswordForEscrow(ctx context.Context, stage func(ctx context.Context, pw string) error) error {
|
||
pw, err := os.ReadFile(m.offboxPwPath())
|
||
if err != nil {
|
||
return fmt.Errorf("read offbox password: %w", err)
|
||
}
|
||
return stage(ctx, strings.TrimSpace(string(pw)))
|
||
}
|
||
|
||
// InjectOffboxPassword pre-places a RECOVERED repo password at offboxPwPath (fork-4 DR seam) so a
|
||
// subsequent WriteOffboxSecrets uses it instead of generating a new one. Refuses to clobber an existing
|
||
// password unless force. Written 0600 via tmp+rename. The value is NEVER logged.
|
||
func (m *Manager) InjectOffboxPassword(pw string, force bool) error {
|
||
pw = strings.TrimSpace(pw)
|
||
if !offboxRepoPwPattern.MatchString(pw) {
|
||
return fmt.Errorf("invalid repo password (expected 64 hex characters)")
|
||
}
|
||
if _, err := os.Stat(m.offboxPwPath()); err == nil && !force {
|
||
return fmt.Errorf("a repo password already exists (pass force to overwrite)")
|
||
}
|
||
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
|
||
return fmt.Errorf("offbox dir: %w", err)
|
||
}
|
||
tmp := m.offboxPwPath() + ".tmp"
|
||
if err := os.WriteFile(tmp, []byte(pw), 0o600); err != nil {
|
||
return fmt.Errorf("write injected password: %w", err)
|
||
}
|
||
if err := os.Rename(tmp, m.offboxPwPath()); err != nil {
|
||
_ = os.Remove(tmp)
|
||
return fmt.Errorf("place injected password: %w", err)
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// offboxEscrowed reports whether the offsite repo password is confirmed escrowed under R (fork-4).
|
||
func (m *Manager) offboxEscrowed() bool {
|
||
t := m.settings.GetOffboxTarget()
|
||
return t != nil && t.EscrowState == "escrowed"
|
||
}
|
||
|
||
// OffboxRunnable reports whether an off-box RUN may proceed: configured AND escrowed. Config/UI still work
|
||
// when not runnable — only actual backup writes are gated (the atomicity guarantee). For the run handler.
|
||
func (m *Manager) OffboxRunnable() bool { return m.OffboxConfigured() && m.offboxEscrowed() }
|
||
|
||
// OffboxCoord returns the non-secret offsite repo coordinates for the DR recipe (fork-4). ok=false when no
|
||
// offbox target is configured. NEVER returns the repo password or the SSH key (those are escrowed/regenerable).
|
||
func (m *Manager) OffboxCoord() (host, user string, port int, repoPath string, ok bool) {
|
||
t := m.settings.GetOffboxTarget()
|
||
if t == nil || t.Host == "" || t.User == "" || t.RepoPath == "" {
|
||
return "", "", 0, "", false
|
||
}
|
||
return t.Host, t.User, t.Port, t.RepoPath, true
|
||
}
|
||
|
||
// OffboxEscrowState returns the current escrow state ("" | "pending" | "escrowed") for the UI/handlers.
|
||
func (m *Manager) OffboxEscrowState() string {
|
||
t := m.settings.GetOffboxTarget()
|
||
if t == nil {
|
||
return ""
|
||
}
|
||
return t.EscrowState
|
||
}
|
||
|
||
// offboxBaseArgs builds the restic global args (repo + sftp.args carrying the ConnectTimeout, key, pinned
|
||
// known_hosts, port) and the env (RESTIC_PASSWORD_FILE). The ConnectTimeout is MANDATORY (fail-fast).
|
||
func (m *Manager) offboxBaseArgs(t *settings.OffboxTarget) ([]string, []string) {
|
||
port := t.Port
|
||
if port == 0 {
|
||
port = 22
|
||
}
|
||
// restic's sftp backend connects via the `-o sftp.command` SSH invocation (the portable form across
|
||
// restic versions — `sftp.args` is not recognized by restic 0.14). The ConnectTimeout makes a dead NAS
|
||
// fail in ~N s (the load-bearing spike Q8 knob); StrictHostKeyChecking + a pinned known_hosts avoid
|
||
// blind TOFU; BatchMode prevents any interactive prompt from hanging the runner. The value is one -o
|
||
// token (restic takes everything after `sftp.command=`); our paths have no spaces (data dir).
|
||
sftpCmd := fmt.Sprintf("ssh %s@%s -p %d -oBatchMode=yes -oConnectTimeout=%d -oStrictHostKeyChecking=yes -oUserKnownHostsFile=%s -i %s -s sftp",
|
||
t.User, t.Host, port, offboxConnectTimeoutSec, m.offboxKnownHosts(), m.offboxKeyPath())
|
||
repo := "sftp:" + t.User + "@" + t.Host + ":" + t.RepoPath
|
||
args := []string{"-r", repo, "-o", "sftp.command=" + sftpCmd}
|
||
env := []string{"RESTIC_PASSWORD_FILE=" + m.offboxPwPath()}
|
||
return args, env
|
||
}
|
||
|
||
// offboxLockRe matches restic's "already locked" error (both the exclusive and shared forms).
|
||
var offboxLockRe = regexp.MustCompile(`repository is already locked`)
|
||
|
||
// unlockStale runs `restic unlock` (stale-only) — cheap pre-run hygiene that removes any lock restic can
|
||
// itself prove dead/old. Non-fatal (logged at debug). Called before every offbox run/restore.
|
||
func (m *Manager) unlockStale(ctx context.Context, base, env []string) {
|
||
uctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
defer cancel()
|
||
if out, err := m.runner()(uctx, env, append(append([]string{}, base...), "unlock")...); err != nil {
|
||
m.logger.Printf("[DEBUG] [offbox] pre-run unlock (stale-only) non-fatal: %v: %s", err, truncate(out))
|
||
}
|
||
}
|
||
|
||
// resticStep runs one restic step (backup/prune/restore) under the offbox single-flight guarantee and
|
||
// self-heals the C2 crash lock. On a lock error it escalates to `unlock --remove-all` and retries ONCE,
|
||
// because THIS controller is the repo's ONLY legitimate writer — per-customer sub-account isolation gives
|
||
// one repo one writer, and the in-process single-flight mutex (held by every caller of this method) proves
|
||
// no sibling operation is live. Plain `restic unlock` is stale-ONLY and does NOT clear a crash lock: the
|
||
// recreated container has a new hostname, so restic can't verify the dead PID and won't treat the lock as
|
||
// stale for ~30 min (the overnight-campaign C2 finding — `unlock --remove-all` is required). A second lock
|
||
// failure surfaces the error (never loops). BOUNDARY: a DR-cloned SECOND controller writing the same repo
|
||
// would defeat the single-writer premise — that is operator-supervised territory (see README), out of scope.
|
||
func (m *Manager) resticStep(ctx context.Context, env, base []string, label string, args ...string) ([]byte, error) {
|
||
full := append(append([]string{}, base...), args...)
|
||
out, err := m.runner()(ctx, env, full...)
|
||
if err == nil || !offboxLockRe.Match(out) {
|
||
return out, err
|
||
}
|
||
m.logger.Printf("[WARN] [offbox] cleared a stale exclusive lock left by a previous crash (single-writer repo) before %s; retrying once", label)
|
||
uctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
if uout, uerr := m.runner()(uctx, env, append(append([]string{}, base...), "unlock", "--remove-all")...); uerr != nil {
|
||
cancel()
|
||
m.logger.Printf("[WARN] [offbox] unlock --remove-all failed: %v: %s", uerr, truncate(uout))
|
||
return out, err // surface the original lock error (never loop)
|
||
}
|
||
cancel()
|
||
return m.runner()(ctx, env, full...) // retry exactly ONCE
|
||
}
|
||
|
||
// ensureOffboxRepo makes sure the SFTP repo exists: probe `cat config`; if absent, `init` (idempotent —
|
||
// a present repo is reused, never re-init). A connect failure surfaces here (fast, via ConnectTimeout).
|
||
func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) error {
|
||
pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
defer cancel()
|
||
if _, err := m.runner()(pctx, env, append(append([]string{}, base...), "cat", "config")...); err == nil {
|
||
return nil // repo exists
|
||
}
|
||
// Repo (probably) absent OR unreachable. Try init; if init succeeds the repo was absent. If init
|
||
// fails because it already exists (a race), treat as success; otherwise the error is real (e.g. a
|
||
// dead NAS — fail fast).
|
||
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
defer icancel()
|
||
out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...)
|
||
if err == nil {
|
||
m.logger.Printf("[INFO] [offbox] initialized restic repo")
|
||
return nil
|
||
}
|
||
if strings.Contains(string(out), "already initialized") || strings.Contains(string(out), "already exists") {
|
||
return nil
|
||
}
|
||
return fmt.Errorf("offbox repo unreachable / init failed: %w: %s", err, truncate(out))
|
||
}
|
||
|
||
// RunOffboxBackup backs up every off-box-toggled app's recovery unit (recovery unit + DB dumps + volume
|
||
// tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A
|
||
// failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error.
|
||
func (m *Manager) RunOffboxBackup(ctx context.Context) error {
|
||
if !m.OffboxConfigured() {
|
||
return fmt.Errorf("off-box backup not configured")
|
||
}
|
||
// fork-4 atomicity gate: no offsite RUN until the repo password is confirmed escrowed under R, so no
|
||
// un-recoverable offsite ciphertext can exist. Not an error (config/UI still work) — a skip.
|
||
if !m.offboxEscrowed() {
|
||
m.logger.Printf("[INFO] [offbox] skipped — pending key escrow (no offsite run until the repo password is escrowed under R)")
|
||
return nil
|
||
}
|
||
if m.migrationActive() {
|
||
m.logger.Printf("[INFO] [offbox] skipped — migration in progress")
|
||
return nil
|
||
}
|
||
if err := m.acquireRunning(); err != nil {
|
||
m.logger.Printf("[INFO] [offbox] skipped — another backup is running")
|
||
return nil // single-flight: don't race; the next scheduled run retries
|
||
}
|
||
defer m.releaseRunning()
|
||
|
||
apps := m.settings.GetOffboxApps()
|
||
t := m.settings.GetOffboxTarget()
|
||
base, env := m.offboxBaseArgs(t)
|
||
start := time.Now()
|
||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastStatus = "running"; o.LastError = "" })
|
||
|
||
var backedUp int
|
||
var missing []string
|
||
var runErr error
|
||
if usedGB, quota, over := offboxQuotaState(t); over {
|
||
// SLICE 4 soft-quota gate (pre-run): NEW backups are refused at ≥100% of the shared-model quota —
|
||
// but the retention/prune step STILL RUNS (pruning is the customer's only way back under quota;
|
||
// gating it too would deadlock them over-quota) and restore paths are untouched. The CURRENT run's
|
||
// gate uses the last-known repo size; a run that crosses 100% mid-flight finishes and the NEXT
|
||
// run refuses.
|
||
m.offboxPruneOnly(ctx, base, env)
|
||
m.offboxRecordStats(ctx, base, env) // the prune may have brought the size back down — refresh
|
||
runErr = fmt.Errorf("A NAS-mentés túllépte a tárhelykeretet (%d/%d GB) — törölj régi mentéseket vagy kérj nagyobb keretet.", usedGB, quota)
|
||
} else {
|
||
backedUp, missing, runErr = m.runOffboxInternal(ctx, apps, base, env)
|
||
}
|
||
|
||
// No-silent-success: apps were toggled but NOTHING was captured (every unit missing) → promote to a
|
||
// hard error so the run reports "error" and the operator is alerted, instead of a misleading ok/0.
|
||
if runErr == nil && len(apps) > 0 && backedUp == 0 {
|
||
runErr = fmt.Errorf("off-box backup produced no snapshots: %d app(s) toggled but no recovery unit was found on any connected drive (missing: %s)",
|
||
len(apps), strings.Join(missing, ", "))
|
||
}
|
||
|
||
dur := time.Since(start)
|
||
snapshots := 0
|
||
if runErr == nil {
|
||
snapshots = m.offboxRecordStats(ctx, base, env)
|
||
}
|
||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||
o.LastRun = time.Now().UTC().Format(time.RFC3339)
|
||
o.LastDuration = dur.Round(time.Second).String()
|
||
if runErr != nil {
|
||
o.LastStatus = "error"
|
||
o.LastError = runErr.Error()
|
||
o.LastWarning = ""
|
||
} else {
|
||
o.LastStatus = "ok"
|
||
o.LastError = ""
|
||
o.SnapshotCount = snapshots
|
||
var warns []string
|
||
if len(missing) > 0 {
|
||
warns = append(warns, fmt.Sprintf("Figyelmeztetés: %d alkalmazásnak nincs elérhető mentése, ezek kimaradtak: %s",
|
||
len(missing), strings.Join(missing, ", ")))
|
||
}
|
||
// SLICE 4: approaching the soft quota (≥80%, <100%) — warn on an otherwise-OK run.
|
||
if qw := offboxQuotaWarning(o); qw != "" {
|
||
warns = append(warns, qw)
|
||
}
|
||
o.LastWarning = strings.Join(warns, " ")
|
||
}
|
||
})
|
||
if m.offboxNotify != nil {
|
||
m.offboxNotify(dur, snapshots, runErr)
|
||
}
|
||
switch {
|
||
case runErr != nil:
|
||
m.logger.Printf("[ERROR] [offbox] backup failed after %s: %v", dur.Round(time.Second), runErr)
|
||
case len(missing) > 0:
|
||
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s) backed up, %d skipped (no unit), %d snapshot(s), %s",
|
||
backedUp, len(missing), snapshots, dur.Round(time.Second))
|
||
default:
|
||
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s) backed up, %d snapshot(s), %s", backedUp, snapshots, dur.Round(time.Second))
|
||
}
|
||
return runErr
|
||
}
|
||
|
||
// offboxCandidateNSRoots is the durable, deployment-state-INDEPENDENT set of felhom-data namespace roots
|
||
// to search for a recovery unit: every registered SCHEDULABLE (non-decommissioned) storage path ∪ the
|
||
// system-data fallback drive, deduped by resolved nsRoot string. This deliberately does NOT consult
|
||
// GetAppDrivePath/AppNamespaceRoot — those read the app's LIVE app.yaml HDD_PATH and silently fall back
|
||
// to systemDataPath when the app isn't currently deployed, which made offbox look on the wrong drive
|
||
// (DIAG root cause). A disconnected drive's path simply isn't present on disk → os.Stat fails → the unit
|
||
// is "not here" (correct: a disconnected drive can't be offsited). Boundary: a decommissioned or
|
||
// non-schedulable drive is not searched (not an active managed backup location).
|
||
func (m *Manager) offboxCandidateNSRoots() []string {
|
||
seen := map[string]bool{}
|
||
var nsRoots []string
|
||
add := func(nr string) {
|
||
if nr != "" && !seen[nr] {
|
||
seen[nr] = true
|
||
nsRoots = append(nsRoots, nr)
|
||
}
|
||
}
|
||
for _, sp := range m.settings.GetSchedulableStoragePaths() {
|
||
add(m.namespaceRoot(sp.Path))
|
||
}
|
||
if m.systemDataPath != "" {
|
||
add(m.namespaceRoot(m.systemDataPath))
|
||
}
|
||
return nsRoots
|
||
}
|
||
|
||
// discoverOffboxUnit locates an app's recovery unit (backups/primary/<app>) across the candidate nsRoots.
|
||
// Returns the src path + true when exactly one exists; when the SAME app's unit exists on more than one
|
||
// drive (drive churn / a stale copy left behind), it returns the NEWEST by manifest CreatedAt (falling
|
||
// back to the unit dir mtime) and WARNs about the others. Independent of the app's live deploy state.
|
||
func (m *Manager) discoverOffboxUnit(app string) (string, bool) {
|
||
var foundSrc, foundManifest []string
|
||
for _, nr := range m.offboxCandidateNSRoots() {
|
||
p := RecoveryUnitPath(nr, app)
|
||
if fi, err := os.Stat(p); err == nil && fi.IsDir() {
|
||
foundSrc = append(foundSrc, p)
|
||
foundManifest = append(foundManifest, RecoveryUnitManifestPath(nr, app))
|
||
}
|
||
}
|
||
switch len(foundSrc) {
|
||
case 0:
|
||
return "", false
|
||
case 1:
|
||
return foundSrc[0], true
|
||
default:
|
||
best := 0
|
||
bestT := offboxUnitTime(foundSrc[0], foundManifest[0])
|
||
for i := 1; i < len(foundSrc); i++ {
|
||
if t := offboxUnitTime(foundSrc[i], foundManifest[i]); t.After(bestT) {
|
||
best, bestT = i, t
|
||
}
|
||
}
|
||
var others []string
|
||
for i, p := range foundSrc {
|
||
if i != best {
|
||
others = append(others, p)
|
||
}
|
||
}
|
||
m.logger.Printf("[WARN] [offbox] %s: multiple recovery units found, using newest (%s); ignoring: %s",
|
||
app, foundSrc[best], strings.Join(others, ", "))
|
||
return foundSrc[best], true
|
||
}
|
||
}
|
||
|
||
// offboxUnitTime returns a recovery unit's timestamp for the multi-copy tiebreak: the manifest's
|
||
// CreatedAt (RFC3339) if readable, else the unit dir's mtime (zero if neither is available).
|
||
func offboxUnitTime(src, manifestPath string) time.Time {
|
||
if mf := readManifest(manifestPath); mf != nil {
|
||
if t, err := time.Parse(time.RFC3339, mf.CreatedAt); err == nil {
|
||
return t
|
||
}
|
||
}
|
||
if fi, err := os.Stat(src); err == nil {
|
||
return fi.ModTime()
|
||
}
|
||
return time.Time{}
|
||
}
|
||
|
||
// runOffboxInternal does the repo-ensure + per-app DISCOVER-then-backup + prune. Caller holds the running
|
||
// flag. Returns how many apps were actually backed up, which toggled apps had no discoverable unit
|
||
// (skipped), and the first hard error (repo-ensure or a restic backup exec failure).
|
||
func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []string) (backedUp int, missing []string, err error) {
|
||
if rerr := m.ensureOffboxRepo(ctx, base, env); rerr != nil {
|
||
return 0, nil, rerr // fail fast (dead NAS surfaces here)
|
||
}
|
||
// Pre-run hygiene: clear any lock restic can prove stale before we start (cheap; the --remove-all
|
||
// crash-lock escalation lives in resticStep for the locks restic can't self-detect).
|
||
m.unlockStale(ctx, base, env)
|
||
var firstErr error
|
||
for _, stack := range apps {
|
||
src, ok := m.discoverOffboxUnit(stack)
|
||
if !ok {
|
||
m.logger.Printf("[WARN] [offbox] %s: no recovery unit found on any connected drive — skipping", stack)
|
||
missing = append(missing, stack)
|
||
continue
|
||
}
|
||
bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||
out, berr := m.resticStep(bctx, env, base, "backup:"+stack, "backup", "--tag", "felhom-offbox", "--tag", stack, src)
|
||
cancel()
|
||
if berr != nil {
|
||
m.logger.Printf("[ERROR] [offbox] backup %s failed: %v: %s", stack, berr, truncate(out))
|
||
if firstErr == nil {
|
||
firstErr = fmt.Errorf("offbox backup %s: %w", stack, berr)
|
||
}
|
||
continue
|
||
}
|
||
backedUp++
|
||
m.logger.Printf("[INFO] [offbox] backed up %s (%s)", stack, src)
|
||
}
|
||
if firstErr != nil {
|
||
return backedUp, missing, firstErr
|
||
}
|
||
// Retention: keep a sane window, prune the rest. Repo-wide (grouped by host+paths by default).
|
||
// prune takes an EXCLUSIVE lock — the exact step whose crash left the C2 stale lock — so it goes
|
||
// through resticStep for the --remove-all self-heal too.
|
||
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||
defer cancel()
|
||
if out, ferr := m.resticStep(fctx, env, base, "prune", "forget", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune"); ferr != nil {
|
||
// A prune failure is non-fatal to the backup itself (data is safe) — log, don't fail the run.
|
||
m.logger.Printf("[WARN] [offbox] forget --prune failed (backups are safe): %v: %s", ferr, truncate(out))
|
||
}
|
||
return backedUp, missing, nil
|
||
}
|
||
|
||
// offboxGiB is the soft-quota unit: QuotaGB counts binary gigabytes (GiB) of restic restore-size.
|
||
const offboxGiB = int64(1) << 30
|
||
|
||
// OffboxReportStatus is the NON-SECRET offsite summary carried on the hub report (SLICE 4) — the input
|
||
// to the hub's OffsiteChecker (fill + staleness alerts). nil when no offbox target is configured.
|
||
type OffboxReportStatus struct {
|
||
Enabled bool `json:"enabled"`
|
||
EscrowState string `json:"escrow_state"`
|
||
LastRun string `json:"last_run,omitempty"` // RFC3339
|
||
LastStatus string `json:"last_status,omitempty"` // "ok" | "error" | "running"
|
||
SnapshotCount int `json:"snapshot_count"`
|
||
RepoSizeBytes int64 `json:"repo_size_bytes"`
|
||
QuotaGB int `json:"quota_gb"`
|
||
}
|
||
|
||
// OffboxReportStatus returns the offsite summary for the hub report (nil = not configured; the hub's
|
||
// checker treats absence as "nothing to watch" — pre-v0.109 reports look the same).
|
||
func (m *Manager) OffboxReportStatus() *OffboxReportStatus {
|
||
t := m.settings.GetOffboxTarget()
|
||
if t == nil || !t.Enabled {
|
||
return nil
|
||
}
|
||
return &OffboxReportStatus{
|
||
Enabled: true, EscrowState: t.EscrowState, LastRun: t.LastRun, LastStatus: t.LastStatus,
|
||
SnapshotCount: t.SnapshotCount, RepoSizeBytes: t.RepoSizeBytes, QuotaGB: t.QuotaGB,
|
||
}
|
||
}
|
||
|
||
// offboxQuotaState evaluates the SLICE 4 soft-quota gate against the LAST-KNOWN repo size (a failed stats
|
||
// call keeps the previous value — stale-but-safe). quota<=0 = no soft limit (dedicated/manual targets).
|
||
func offboxQuotaState(t *settings.OffboxTarget) (usedGB, quotaGB int, over bool) {
|
||
if t == nil || t.QuotaGB <= 0 {
|
||
return 0, 0, false
|
||
}
|
||
usedGB = int(t.RepoSizeBytes / offboxGiB)
|
||
return usedGB, t.QuotaGB, t.RepoSizeBytes >= int64(t.QuotaGB)*offboxGiB
|
||
}
|
||
|
||
// offboxQuotaWarning returns the Hungarian ≥80% (<100%) usage notice, or "" (quota unset / usage fine /
|
||
// already over — over-quota is the run-refusal error, not a warning).
|
||
func offboxQuotaWarning(t *settings.OffboxTarget) string {
|
||
if t == nil || t.QuotaGB <= 0 || t.RepoSizeBytes <= 0 {
|
||
return ""
|
||
}
|
||
limit := int64(t.QuotaGB) * offboxGiB
|
||
pct := t.RepoSizeBytes * 100 / limit
|
||
if pct < 80 || t.RepoSizeBytes >= limit {
|
||
return ""
|
||
}
|
||
return fmt.Sprintf("A NAS-mentés a keret %d%%-át használja (%d/%d GB).", pct, t.RepoSizeBytes/offboxGiB, t.QuotaGB)
|
||
}
|
||
|
||
// OffboxQuotaPercent returns the usage percentage for the /backups usage bar (0 when no quota/size).
|
||
func OffboxQuotaPercent(t *settings.OffboxTarget) int {
|
||
if t == nil || t.QuotaGB <= 0 || t.RepoSizeBytes <= 0 {
|
||
return 0
|
||
}
|
||
pct := int(t.RepoSizeBytes * 100 / (int64(t.QuotaGB) * offboxGiB))
|
||
if pct > 100 {
|
||
pct = 100
|
||
}
|
||
return pct
|
||
}
|
||
|
||
// offboxPruneOnly runs ONLY the retention/prune step (the over-quota path: new backups are refused but
|
||
// pruning must stay available — it is the only way back under the quota). Repo-ensure first so a fresh
|
||
// target still fails loudly; errors are non-fatal (same as the regular run's prune).
|
||
func (m *Manager) offboxPruneOnly(ctx context.Context, base, env []string) {
|
||
if rerr := m.ensureOffboxRepo(ctx, base, env); rerr != nil {
|
||
m.logger.Printf("[WARN] [offbox] over-quota prune: repo unreachable: %v", rerr)
|
||
return
|
||
}
|
||
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||
defer cancel()
|
||
fargs := append(append([]string{}, base...), "forget", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune")
|
||
if out, ferr := m.runner()(fctx, env, fargs...); ferr != nil {
|
||
m.logger.Printf("[WARN] [offbox] over-quota prune failed: %v: %s", ferr, truncate(out))
|
||
} else {
|
||
m.logger.Printf("[INFO] [offbox] over-quota: prune executed (new backups refused until under quota)")
|
||
}
|
||
}
|
||
|
||
// offboxRecordStats reads the snapshot count (best-effort) for the UI; also fills repo size when stats works.
|
||
func (m *Manager) offboxRecordStats(ctx context.Context, base, env []string) int {
|
||
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
defer cancel()
|
||
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
||
if err != nil {
|
||
return 0
|
||
}
|
||
var snaps []struct {
|
||
ID string `json:"id"`
|
||
}
|
||
if json.Unmarshal(out, &snaps) != nil {
|
||
return 0
|
||
}
|
||
// Repo size (best-effort, restore-size). Bytes feed the soft-quota gate (SLICE 4); a failed stats
|
||
// call keeps the last-known value (stale-but-safe).
|
||
if so, serr := m.runner()(sctx, env, append(append([]string{}, base...), "stats", "--json")...); serr == nil {
|
||
var st struct {
|
||
TotalSize int64 `json:"total_size"`
|
||
}
|
||
if json.Unmarshal(so, &st) == nil && st.TotalSize > 0 {
|
||
human := humanizeBytes(st.TotalSize)
|
||
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||
o.RepoSizeHuman = human
|
||
o.RepoSizeBytes = st.TotalSize
|
||
})
|
||
}
|
||
}
|
||
return len(snaps)
|
||
}
|
||
|
||
// RestoreOffbox restores an app's latest off-box snapshot to destDir (a scratch/verify location — it does
|
||
// NOT overwrite live data). Returns an error on failure (checks restic's own exit code).
|
||
func (m *Manager) RestoreOffbox(ctx context.Context, stackName, destDir string) error {
|
||
if !m.OffboxConfigured() {
|
||
return fmt.Errorf("off-box backup not configured")
|
||
}
|
||
if !isSafeStackName(stackName) {
|
||
return fmt.Errorf("invalid stack name")
|
||
}
|
||
if err := os.MkdirAll(destDir, 0o755); err != nil {
|
||
return fmt.Errorf("restore dir: %w", err)
|
||
}
|
||
t := m.settings.GetOffboxTarget()
|
||
base, env := m.offboxBaseArgs(t)
|
||
rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||
defer cancel()
|
||
m.unlockStale(rctx, base, env) // pre-restore hygiene (crash-lock self-heal is in resticStep)
|
||
out, err := m.resticStep(rctx, env, base, "restore:"+stackName, "restore", "latest", "--tag", stackName, "--target", destDir)
|
||
if err != nil {
|
||
return fmt.Errorf("offbox restore %s: %w: %s", stackName, err, truncate(out))
|
||
}
|
||
m.logger.Printf("[INFO] [offbox] restored %s → %s", stackName, destDir)
|
||
return nil
|
||
}
|
||
|
||
// isSafeStackName guards a stack name used as a restic tag / path component.
|
||
func isSafeStackName(s string) bool {
|
||
if s == "" || len(s) > 64 {
|
||
return false
|
||
}
|
||
for _, c := range s {
|
||
if (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '-' || c == '_' {
|
||
continue
|
||
}
|
||
return false
|
||
}
|
||
return true
|
||
}
|
||
|
||
// truncate caps subprocess output for a log line + strips a trailing newline.
|
||
func truncate(b []byte) string {
|
||
s := strings.TrimSpace(string(b))
|
||
if len(s) > 400 {
|
||
return s[:400] + "…"
|
||
}
|
||
return s
|
||
}
|