Files
felhom-controller/controller/internal/backup/offbox.go
T
admin b5d78d1e0f v0.147.0 — feedback slice 1: pressing a button says something
The systemic complaint, twice in one evening: you press a button and nothing
happens. No progress, no ETA, no named result. Three worst offenders, fixed on
the two patterns already here (deploy 3-step panel, storage-init status poll).
No new framework — that is a ROADMAP item; three targeted cards ship tonight.

4a — a verification restore names its result. The flash said the app had been
restored "to a verification folder on the drive"; which folder, on which drive,
was invisible, so the customer could not go and look at what they had just asked
for. Full path now. The restore page gained a listing of existing verification
copies (app, size, date, path) — nothing anywhere showed these, so they piled up
and the only way to find them was SSH — each with a double-confirmed delete.

That delete is the only one this release adds, so it names a STACK, never a
path: the Manager resolves the name inside a backups/offsite-restore root it
computed itself and refuses anything landing outside. Red-proofed — neutralise
the name guard and stack:"" resolves to the offsite-restore ROOT and takes every
copy with it. Refusals are asserted as non-effects.

4b — Megosztás enable shows what it is waiting for. Enabling ran ReconcileSamba
synchronously inside the POST handler; on a golden without felhom-samba baked
that is compose pulling ~100MB, i.e. minutes of an apparently-hung form post
followed by "Beállítás mentve." whether or not anything came up. Detached +
polled now, distinguishing "képfájl letöltése" from "indítás" — decided BEFORE
the work starts, since afterwards the image is always present. Success is
probed, not inferred (compose up -d exits 0 on a crash-loop). The password form
starts the same job: with UserSet false reconcile deploys nothing, so on a fresh
box that is where the pull actually happens.

4c — "Távoli mentés most" streams real progress. restic was already reporting
bytes and percent; the runner seam used CombinedOutput() and discarded them. The
manual run now passes --json and scans stdout line-by-line: total bytes, percent,
current app. Manual only — the nightly stays silent, pinned by a test that fails
if it ever passes --json. The poll now arms unconditionally, closing a race the
manual trigger always ran: the redirect rendered before the goroutine wrote
LastStatus=running, so the poll never armed and the page sat static during the
very run just started. Red-proofed twice.

Also closes the golden/controller infra-image drift at the source: infra.Images()
derives from the existing pins and --print-infra-images exposes it, so the golden
bake can stop carrying its own copy. That copy had already drifted — felhom-samba
was never added, so the golden baked 3 of 4, which is why enabling Megosztás
pulled at runtime in the first place.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Nn3VgQk9iwEGgyx6QJ2NvE
2026-07-19 09:30:30 +02:00

1122 lines
52 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package backup
import (
"context"
"crypto/rand"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"os"
"os/exec"
"path/filepath"
"regexp"
"sort"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// Off-box (NAS) backup target — Part B. An ENCRYPTED restic repo reached over SFTP (no kernel mount;
// restic talks SFTP to the NAS directly). This is the "1 off-site" leg of 3-2-1 for the app-data tier
// (each off-box app's recovery unit + DB dumps + volume tars), distinct from the local cross-drive rsync
// copy and from the agent's PBS whole-CT DR. The NAS sees only ciphertext.
//
// THE load-bearing lesson (spike Q8): a dead NAS must FAIL FAST, never hang the backup runner — every
// restic invocation carries `-o sftp.args=…-oConnectTimeout=N…` so a black-holed endpoint errors in
// ~N seconds instead of a multi-minute TCP retry. We also check restic's OWN exit code (never
// pipe-swallow). Secrets (repo password + SSH key) live in 0600 files in the data dir — never logged,
// never in a committed/non-0600 file; they ride DR via the PBS whole-CT snapshot of the rootfs.
const (
// offboxConnectTimeoutSec is the SSH ConnectTimeout (spike Q8) — load-bearing fail-fast.
offboxConnectTimeoutSec = 10
// offboxBackupTimeout bounds a full off-box run; offboxProbeTimeout bounds the quick repo probes.
offboxBackupTimeout = 2 * time.Hour
offboxProbeTimeout = 90 * time.Second
)
// offboxRunner is the restic-exec seam (tests inject a fake so no real restic/ssh runs). It runs restic
// with args + extra env and returns combined output + the process error (whose ExitCode the caller checks).
type offboxRunner func(ctx context.Context, env []string, args ...string) ([]byte, error)
func defaultOffboxRunner(ctx context.Context, env []string, args ...string) ([]byte, error) {
cmd := exec.CommandContext(ctx, "restic", args...)
cmd.Env = append(os.Environ(), env...)
return cmd.CombinedOutput()
}
// SetOffboxRunner overrides the restic exec (tests). SetOffboxNotify wires the failure→operator alert.
func (m *Manager) SetOffboxRunner(r offboxRunner) { m.offboxRunner = r }
func (m *Manager) SetOffboxNotify(fn func(dur time.Duration, snapshots int, err error)) {
m.offboxNotify = fn
}
// SetOffboxOrphanEvent wires the offsite-repo continuity event push (main.go → notifier).
func (m *Manager) SetOffboxOrphanEvent(fn func(eventType, renamedTo string)) { m.offboxOrphanEvent = fn }
// SetOffboxSSH overrides the raw-ssh exec used for the orphaned-repo move-aside (tests).
func (m *Manager) SetOffboxSSH(fn func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error)) {
m.offboxSSH = fn
}
// ErrOffboxOrphaned is the sentinel returned when the offsite repo exists but is keyed under a
// passphrase this controller no longer has (the reinstall shape) — the run skips and the UI shows
// the orphan card instead of the raw restic error.
var ErrOffboxOrphaned = fmt.Errorf("offbox repo orphaned: exists but keyed under a previous, no-longer-available passphrase")
// classifyResticProbe maps a `restic cat config` failure to a repo class. The signatures are the exact
// restic stderr matched in the 2026-07-17 diagnosis + restic's no-repo message:
// - "orphaned": repo present, wrong key ("wrong password or no key found") — the definitive signal
// - "norepo": no repo at the location (init is the correct path)
// - "other": network/SFTP-auth/unknown — NOT orphaned; existing error handling
func classifyResticProbe(out []byte, err error) string {
if err == nil {
return "" // success — repo good
}
s := strings.ToLower(string(out))
switch {
case strings.Contains(s, "wrong password or no key found"):
return "orphaned"
case strings.Contains(s, "unable to open config file"),
strings.Contains(s, "is there a repository at the following location"),
strings.Contains(s, "no such file"),
strings.Contains(s, "does not exist"):
return "norepo"
default:
return "other"
}
}
func defaultOffboxSSH(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
if port == 0 {
port = 22
}
args := []string{
"-p", fmt.Sprint(port), "-oBatchMode=yes", fmt.Sprintf("-oConnectTimeout=%d", offboxConnectTimeoutSec),
"-oStrictHostKeyChecking=yes", "-oUserKnownHostsFile=" + knownHosts, "-i", keyPath,
user + "@" + host, remoteCmd,
}
cmd := exec.CommandContext(ctx, "ssh", args...)
return cmd.CombinedOutput()
}
func (m *Manager) sshRunner() func(ctx context.Context, host, user string, port int, keyPath, knownHosts, remoteCmd string) ([]byte, error) {
if m.offboxSSH != nil {
return m.offboxSSH
}
return defaultOffboxSSH
}
// OffboxOrphaned reports whether the offsite repo is in the ORPHANED state (persisted).
func (m *Manager) OffboxOrphaned() bool {
t := m.settings.GetOffboxTarget()
return t != nil && t.RepoState == "orphaned"
}
// OffboxOrphanedRenamedTo returns the last move-aside path (for the card copy; "" if none).
func (m *Manager) OffboxOrphanedRenamedTo() string {
t := m.settings.GetOffboxTarget()
if t == nil {
return ""
}
return t.OrphanedRenamedTo
}
// markOrphaned sets the persistent ORPHANED state and, ONLY on the transition into it (not already
// orphaned), pushes the offbox_repo_orphaned event — so scheduled runs never nightly-spam.
func (m *Manager) markOrphaned() {
already := m.OffboxOrphaned()
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.RepoState = "orphaned"
if o.OrphanedAt == "" || !already {
o.OrphanedAt = time.Now().UTC().Format(time.RFC3339)
}
}); err != nil {
m.logger.Printf("[WARN] [offbox] persist orphaned state failed: %v", err)
}
if !already {
m.logger.Printf("[WARN] [offbox] offsite repo ORPHANED — remote holds backups written under a previous, no-longer-available key; runs will skip until reset")
if m.offboxOrphanEvent != nil {
m.offboxOrphanEvent("offbox_repo_orphaned", "")
}
}
}
// resetOrphanedRepo moves the orphaned repo aside (never deletes) and re-inits a fresh repo under the
// CURRENT passphrase. Reversible. Used by the unclaimed auto-reset (Scenario B) and the claimed
// confirmed reset (Scenario C). Caller holds the single-flight guarantee (run mutex) OR is the handler.
func (m *Manager) resetOrphanedRepo(ctx context.Context, base, env []string, reason string) error {
t := m.settings.GetOffboxTarget()
if t == nil {
return fmt.Errorf("no offsite target configured")
}
port := t.Port
if port == 0 {
port = 22
}
// Choose a move-aside name that never overwrites an earlier orphaned copy (edge rule: -2, -3).
date := time.Now().UTC().Format("20060102")
base1 := t.RepoPath + ".orphaned-" + date
newPath := base1
for i := 2; i <= 20; i++ {
// `test -e <p>` returns non-zero (exit 1) when absent — that is the name we want. A transport
// error also lands here; we then just try the mv and let it fail loudly rather than loop.
out, err := m.sshRunner()(ctx, t.Host, t.User, port, m.offboxKeyPath(), m.offboxKnownHosts(), "test -e "+shellQuote(newPath))
if err != nil && !strings.Contains(strings.ToLower(string(out)), "denied") {
break // absent (test -e exit 1) → free name
}
newPath = fmt.Sprintf("%s-%d", base1, i)
}
m.logger.Printf("[WARN] [offbox] resetting orphaned repo (%s): move-aside %s -> %s, then re-init", reason, t.RepoPath, newPath)
if out, err := m.sshRunner()(ctx, t.Host, t.User, port, m.offboxKeyPath(), m.offboxKnownHosts(),
fmt.Sprintf("mv %s %s", shellQuote(t.RepoPath), shellQuote(newPath))); err != nil {
return fmt.Errorf("offbox move-aside failed: %w: %s", err, truncate(out))
}
// Fresh init under the current passphrase.
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer icancel()
if out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...); err != nil {
return fmt.Errorf("offbox re-init after move-aside failed: %w: %s", err, truncate(out))
}
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.RepoState = ""
o.OrphanedAt = ""
o.OrphanedRenamedTo = newPath
o.LastError = ""
}); err != nil {
m.logger.Printf("[WARN] [offbox] clear orphaned state failed: %v", err)
}
m.logger.Printf("[INFO] [offbox] orphaned repo reset complete — old history set aside at %s (move-aside, not deleted); fresh repo initialized", newPath)
if m.offboxOrphanEvent != nil {
m.offboxOrphanEvent("offbox_repo_reset", newPath)
}
return nil
}
// ResetOrphanedRepo is the handler entry point for the CLAIMED confirmed reset (Scenario C). It refuses
// unless the repo is currently orphaned. It builds the base/env and runs the move-aside + re-init.
func (m *Manager) ResetOrphanedRepo(ctx context.Context) error {
if !m.OffboxOrphaned() {
return fmt.Errorf("az offsite tároló nincs elárvult állapotban")
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
return m.resetOrphanedRepo(ctx, base, env, "operator-confirmed (claimed)")
}
// shellQuote single-quotes a path for the remote shell (our repo paths have no single quotes).
func shellQuote(s string) string { return "'" + strings.ReplaceAll(s, "'", `'\''`) + "'" }
// SetOffboxSizer overrides the mandatory-set byte estimator (tests). SetOffboxEnlargeBlockedNotifier
// wires the edge-triggered enlargement-blocked notification (main.go). SetOffboxPlaceCopier overrides
// the place-to-live missing-only merge (tests).
func (m *Manager) SetOffboxSizer(fn func(path string) int64) { m.offboxSizer = fn }
func (m *Manager) SetOffboxEnlargeBlockedNotifier(fn func(stack string, estBytes int64, usedGB, quotaGB int)) {
m.offboxEnlargeBlockedNotify = fn
}
func (m *Manager) SetOffboxPlaceCopier(fn func(src, dst string) (int, error)) { m.offboxPlaceCopier = fn }
// offboxSize returns the mandatory-set byte estimator (nil seam → the real du -sb dirSizeBytes).
func (m *Manager) offboxSize() func(string) int64 {
if m.offboxSizer != nil {
return m.offboxSizer
}
return dirSizeBytes
}
func (m *Manager) runner() offboxRunner {
if m.offboxRunner != nil {
return m.offboxRunner
}
return defaultOffboxRunner
}
func (m *Manager) offboxDir() string { return filepath.Join(m.cfg.Paths.DataDir, "offbox") }
func (m *Manager) offboxKeyPath() string { return filepath.Join(m.offboxDir(), "ssh_key") }
func (m *Manager) offboxPwPath() string { return filepath.Join(m.offboxDir(), "repo_password") }
func (m *Manager) offboxKnownHosts() string { return filepath.Join(m.offboxDir(), "known_hosts") }
// WriteOffboxSecrets persists the SSH private key + (auto-generated if empty) repo password + the pinned
// known-host line as 0600/0644 files in the data dir. The key is provided out-of-band by the operator
// (UI), never logged. Returns the repo password so the caller need not read the file. Idempotent: an empty
// sshKey/knownHosts leaves the existing file untouched (a re-save of just the target shouldn't wipe keys).
func (m *Manager) WriteOffboxSecrets(sshKey, knownHosts string) error {
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
return fmt.Errorf("offbox dir: %w", err)
}
if strings.TrimSpace(sshKey) != "" {
key := sshKey
if !strings.HasSuffix(key, "\n") {
key += "\n"
}
if err := os.WriteFile(m.offboxKeyPath(), []byte(key), 0o600); err != nil {
return fmt.Errorf("offbox ssh key: %w", err)
}
}
if strings.TrimSpace(knownHosts) != "" {
kh := knownHosts
if !strings.HasSuffix(kh, "\n") {
kh += "\n"
}
if err := os.WriteFile(m.offboxKnownHosts(), []byte(kh), 0o644); err != nil {
return fmt.Errorf("offbox known_hosts: %w", err)
}
}
// Auto-generate the repo password once (0600), never log it.
if _, err := os.Stat(m.offboxPwPath()); os.IsNotExist(err) {
pw, gerr := generateOffboxPassword()
if gerr != nil {
return gerr
}
if werr := os.WriteFile(m.offboxPwPath(), []byte(pw), 0o600); werr != nil {
return fmt.Errorf("offbox repo password: %w", werr)
}
}
return nil
}
// generateOffboxPassword returns a 256-bit hex repo password.
func generateOffboxPassword() (string, error) {
b := make([]byte, 32)
if _, err := rand.Read(b); err != nil {
return "", fmt.Errorf("offbox password gen: %w", err)
}
return hex.EncodeToString(b), nil
}
// Off-box target field validation — the security boundary for the values that flow into the `ssh … -s
// sftp` command restic runs. Host/user are charset-restricted AND must not start with '-' (an ssh
// OPTION-INJECTION vector: a host like "-oProxyCommand=evil" would make ssh execute an arbitrary command).
// RepoPath is an absolute, traversal-free, metacharacter-free path. This mirrors the agent's validate.go
// discipline: validate before any value reaches an exec.
var (
reOffboxHost = regexp.MustCompile(`^[A-Za-z0-9._-]+$`)
reOffboxUser = regexp.MustCompile(`^[A-Za-z0-9._-]+$`)
reOffboxPath = regexp.MustCompile(`^/[A-Za-z0-9._/-]+$`)
)
// ValidateOffboxTarget rejects values that could inject into the ssh command line (option injection via a
// leading '-', shell/space metacharacters, path traversal). Returns nil for a safe target.
func ValidateOffboxTarget(t *settings.OffboxTarget) error {
if t == nil {
return fmt.Errorf("no off-box target")
}
if t.Host == "" || len(t.Host) > 255 || !reOffboxHost.MatchString(t.Host) || strings.HasPrefix(t.Host, "-") || strings.HasPrefix(t.Host, ".") {
return fmt.Errorf("invalid NAS host (letters, digits, '.', '-', '_'; must not start with '-' or '.')")
}
if t.User == "" || len(t.User) > 64 || !reOffboxUser.MatchString(t.User) || strings.HasPrefix(t.User, "-") {
return fmt.Errorf("invalid user (letters, digits, '.', '-', '_'; must not start with '-')")
}
if len(t.RepoPath) > 512 || !reOffboxPath.MatchString(t.RepoPath) || strings.Contains(t.RepoPath, "..") {
return fmt.Errorf("invalid repo path (absolute, no spaces/metacharacters, no '..')")
}
if p := t.Port; p != 0 && (p < 1 || p > 65535) {
return fmt.Errorf("invalid port")
}
return nil
}
// OffboxConfigured reports whether the target is set, enabled, VALID, and the key + password files exist
// (so the UI/scheduler can gate a run without leaking why). A target that fails validation is treated as
// not-configured — fail-closed, so a bad/hostile persisted target can never reach the ssh exec.
func (m *Manager) OffboxConfigured() bool {
t := m.settings.GetOffboxTarget()
if t == nil || !t.Enabled || t.Host == "" || t.User == "" || t.RepoPath == "" {
return false
}
if err := ValidateOffboxTarget(t); err != nil {
return false
}
if _, err := os.Stat(m.offboxKeyPath()); err != nil {
return false
}
if _, err := os.Stat(m.offboxPwPath()); err != nil {
return false
}
return true
}
// offboxRepoPwPattern matches a valid restic repo password (generateOffboxPassword = 32 rand bytes → 64 hex).
var offboxRepoPwPattern = regexp.MustCompile(`^[0-9a-fA-F]{64}$`)
// ApplyOffsiteTarget configures the offbox target from a hub-provisioned descriptor (SLICE 2 apply-bridge):
// it writes the 0600 SSH key + pinned known_hosts, sets the target with EscrowState="pending", and pushes
// the repo password to the agent for escrow — the SAME fork-4 enable path a manual config takes. `stage` is
// the agent escrow-stage push (nil skips it, e.g. when the agent is unreachable — the run gate still holds).
func (m *Manager) ApplyOffsiteTarget(ctx context.Context, tgt *settings.OffboxTarget, sshKeyPEM, knownHosts string, stage func(ctx context.Context, pw string) error) error {
if err := m.WriteOffboxSecrets(sshKeyPEM, knownHosts); err != nil {
return fmt.Errorf("apply offsite secrets: %w", err)
}
// Re-apply (v0.109.1 live finding): the bridge rebuilds the target from the descriptor, but the
// EXISTING target's custody + runtime status must carry over — EscrowState tracks the REPO PASSWORD
// (preserved by WriteOffboxSecrets above, never rotated by this path), not the target coords; and the
// status fields belong to the runner. Without this, a quota bump demoted an escrowed demo target to
// pending and wiped its history (which would also false-trigger the hub's staleness alert).
if cur := m.settings.GetOffboxTarget(); cur != nil {
tgt.EscrowState = cur.EscrowState
tgt.LastRun, tgt.LastStatus, tgt.LastError = cur.LastRun, cur.LastStatus, cur.LastError
tgt.LastDuration, tgt.LastWarning = cur.LastDuration, cur.LastWarning
tgt.RepoSizeHuman, tgt.RepoSizeBytes, tgt.SnapshotCount = cur.RepoSizeHuman, cur.RepoSizeBytes, cur.SnapshotCount
}
if tgt.EscrowState != "escrowed" {
tgt.EscrowState = "pending"
}
if err := m.settings.SetOffboxTarget(tgt); err != nil {
return fmt.Errorf("apply offsite target: %w", err)
}
if stage != nil {
// Best-effort: the offbox is configured + pending regardless. A stage-push failure (agent momentarily
// unreachable) is logged, not fatal — the escrow can be (re-)staged later (operator ceremony / re-enable).
if err := m.PushOffboxPasswordForEscrow(ctx, stage); err != nil {
m.logger.Printf("[WARN] [offbox] apply-offsite: escrow stage push failed (agent unreachable?) — offbox configured pending, re-stage later: %v", err)
}
}
return nil
}
// HashResticPassword is the CANONICAL hasher for the offsite repo password (SLICE 3 hub-verified escrow
// auto-confirm): sha256 hex of the TRIMMED password string — the SAME convention as the agent's
// escrow.HashResticPassword (both sides TrimSpace their file reads; pinned by the SAME cross-repo test
// vector in felhom-agent). The hash of a 256-bit random secret is non-reversible and non-brute-forceable —
// safe to log/compare; the PASSWORD itself is never logged.
func HashResticPassword(pw string) string {
sum := sha256.Sum256([]byte(strings.TrimSpace(pw)))
return hex.EncodeToString(sum[:])
}
// OffboxRepoPasswordHash returns the canonical hash of the local repo password (false when no password
// file exists — nothing to match; the auto-confirm check skips).
func (m *Manager) OffboxRepoPasswordHash() (string, bool) {
pw, err := os.ReadFile(m.offboxPwPath())
if err != nil {
return "", false
}
return HashResticPassword(string(pw)), true
}
// PushOffboxPasswordForEscrow reads the 0600 repo password and hands it to `stage` (the agent push), so
// the web/handler caller never sees the value — used by the enable flow to escrow-stage the offsite key.
func (m *Manager) PushOffboxPasswordForEscrow(ctx context.Context, stage func(ctx context.Context, pw string) error) error {
pw, err := os.ReadFile(m.offboxPwPath())
if err != nil {
return fmt.Errorf("read offbox password: %w", err)
}
return stage(ctx, strings.TrimSpace(string(pw)))
}
// InjectOffboxPassword pre-places a RECOVERED repo password at offboxPwPath (fork-4 DR seam) so a
// subsequent WriteOffboxSecrets uses it instead of generating a new one. Refuses to clobber an existing
// password unless force. Written 0600 via tmp+rename. The value is NEVER logged.
func (m *Manager) InjectOffboxPassword(pw string, force bool) error {
pw = strings.TrimSpace(pw)
if !offboxRepoPwPattern.MatchString(pw) {
return fmt.Errorf("invalid repo password (expected 64 hex characters)")
}
if _, err := os.Stat(m.offboxPwPath()); err == nil && !force {
return fmt.Errorf("a repo password already exists (pass force to overwrite)")
}
if err := os.MkdirAll(m.offboxDir(), 0o700); err != nil {
return fmt.Errorf("offbox dir: %w", err)
}
tmp := m.offboxPwPath() + ".tmp"
if err := os.WriteFile(tmp, []byte(pw), 0o600); err != nil {
return fmt.Errorf("write injected password: %w", err)
}
if err := os.Rename(tmp, m.offboxPwPath()); err != nil {
_ = os.Remove(tmp)
return fmt.Errorf("place injected password: %w", err)
}
return nil
}
// offboxEscrowed reports whether the offsite repo password is confirmed escrowed under R (fork-4).
func (m *Manager) offboxEscrowed() bool {
t := m.settings.GetOffboxTarget()
return t != nil && t.EscrowState == "escrowed"
}
// OffboxRunnable reports whether an off-box RUN may proceed: configured AND escrowed. Config/UI still work
// when not runnable — only actual backup writes are gated (the atomicity guarantee). For the run handler.
func (m *Manager) OffboxRunnable() bool { return m.OffboxConfigured() && m.offboxEscrowed() }
// OffboxCoord returns the non-secret offsite repo coordinates for the DR recipe (fork-4). ok=false when no
// offbox target is configured. NEVER returns the repo password or the SSH key (those are escrowed/regenerable).
func (m *Manager) OffboxCoord() (host, user string, port int, repoPath string, ok bool) {
t := m.settings.GetOffboxTarget()
if t == nil || t.Host == "" || t.User == "" || t.RepoPath == "" {
return "", "", 0, "", false
}
return t.Host, t.User, t.Port, t.RepoPath, true
}
// OffboxEscrowState returns the current escrow state ("" | "pending" | "escrowed") for the UI/handlers.
func (m *Manager) OffboxEscrowState() string {
t := m.settings.GetOffboxTarget()
if t == nil {
return ""
}
return t.EscrowState
}
// offboxBaseArgs builds the restic global args (repo + sftp.args carrying the ConnectTimeout, key, pinned
// known_hosts, port) and the env (RESTIC_PASSWORD_FILE). The ConnectTimeout is MANDATORY (fail-fast).
func (m *Manager) offboxBaseArgs(t *settings.OffboxTarget) ([]string, []string) {
port := t.Port
if port == 0 {
port = 22
}
// restic's sftp backend connects via the `-o sftp.command` SSH invocation (the portable form across
// restic versions — `sftp.args` is not recognized by restic 0.14). The ConnectTimeout makes a dead NAS
// fail in ~N s (the load-bearing spike Q8 knob); StrictHostKeyChecking + a pinned known_hosts avoid
// blind TOFU; BatchMode prevents any interactive prompt from hanging the runner. The value is one -o
// token (restic takes everything after `sftp.command=`); our paths have no spaces (data dir).
sftpCmd := fmt.Sprintf("ssh %s@%s -p %d -oBatchMode=yes -oConnectTimeout=%d -oStrictHostKeyChecking=yes -oUserKnownHostsFile=%s -i %s -s sftp",
t.User, t.Host, port, offboxConnectTimeoutSec, m.offboxKnownHosts(), m.offboxKeyPath())
repo := "sftp:" + t.User + "@" + t.Host + ":" + t.RepoPath
args := []string{"-r", repo, "-o", "sftp.command=" + sftpCmd}
env := []string{"RESTIC_PASSWORD_FILE=" + m.offboxPwPath()}
return args, env
}
// offboxLockRe matches restic's "already locked" error (both the exclusive and shared forms).
var offboxLockRe = regexp.MustCompile(`repository is already locked`)
// unlockStale runs `restic unlock` (stale-only) — cheap pre-run hygiene that removes any lock restic can
// itself prove dead/old. Non-fatal (logged at debug). Called before every offbox run/restore.
func (m *Manager) unlockStale(ctx context.Context, base, env []string) {
uctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
if out, err := m.runner()(uctx, env, append(append([]string{}, base...), "unlock")...); err != nil {
m.logger.Printf("[DEBUG] [offbox] pre-run unlock (stale-only) non-fatal: %v: %s", err, truncate(out))
}
}
// resticStep runs one restic step (backup/prune/restore) under the offbox single-flight guarantee and
// self-heals the C2 crash lock. On a lock error it escalates to `unlock --remove-all` and retries ONCE,
// because THIS controller is the repo's ONLY legitimate writer — per-customer sub-account isolation gives
// one repo one writer, and the in-process single-flight mutex (held by every caller of this method) proves
// no sibling operation is live. Plain `restic unlock` is stale-ONLY and does NOT clear a crash lock: the
// recreated container has a new hostname, so restic can't verify the dead PID and won't treat the lock as
// stale for ~30 min (the overnight-campaign C2 finding — `unlock --remove-all` is required). A second lock
// failure surfaces the error (never loops). BOUNDARY: a DR-cloned SECOND controller writing the same repo
// would defeat the single-writer premise — that is operator-supervised territory (see README), out of scope.
func (m *Manager) resticStep(ctx context.Context, env, base []string, label string, args ...string) ([]byte, error) {
full := append(append([]string{}, base...), args...)
out, err := m.runner()(ctx, env, full...)
if err == nil || !offboxLockRe.Match(out) {
return out, err
}
m.logger.Printf("[WARN] [offbox] cleared a stale exclusive lock left by a previous crash (single-writer repo) before %s; retrying once", label)
uctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
if uout, uerr := m.runner()(uctx, env, append(append([]string{}, base...), "unlock", "--remove-all")...); uerr != nil {
cancel()
m.logger.Printf("[WARN] [offbox] unlock --remove-all failed: %v: %s", uerr, truncate(uout))
return out, err // surface the original lock error (never loop)
}
cancel()
return m.runner()(ctx, env, full...) // retry exactly ONCE
}
// ensureOffboxRepo makes sure the SFTP repo exists: probe `cat config`; if absent, `init` (idempotent —
// a present repo is reused, never re-init). A connect failure surfaces here (fast, via ConnectTimeout).
func (m *Manager) ensureOffboxRepo(ctx context.Context, base, env []string) error {
pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
pout, perr := m.runner()(pctx, env, append(append([]string{}, base...), "cat", "config")...)
switch classifyResticProbe(pout, perr) {
case "": // success — repo good (and clear any stale orphaned flag)
if m.OffboxOrphaned() {
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.RepoState = ""; o.OrphanedAt = "" })
}
return nil
case "orphaned":
// The repo EXISTS but is keyed under a passphrase we no longer have (the reinstall shape). An
// UNCLAIMED (as-delivered) box auto-resets (Scenario B); a CLAIMED box surfaces the orphan card
// and skips until the customer confirms a reset (Scenario C). Move-aside, never delete.
if !m.settings.GetClaimed() {
m.logger.Printf("[INFO] [offbox] orphaned repo on an UNCLAIMED box — auto-resetting (move-aside + re-init)")
if m.offboxOrphanEvent != nil {
m.offboxOrphanEvent("offbox_repo_orphaned", "")
}
if rerr := m.resetOrphanedRepo(ctx, base, env, "auto (unclaimed)"); rerr != nil {
m.markOrphaned() // auto-reset failed → fall back to the orphan card so it isn't silent
return ErrOffboxOrphaned
}
return nil // repo is fresh under the current passphrase → the run proceeds
}
m.markOrphaned()
return ErrOffboxOrphaned
case "norepo":
// No repo at the location → init (the normal first-run path). A race where it already exists is
// treated as success; any other init error is real (e.g. a dead NAS — fail fast).
ictx, icancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer icancel()
out, err := m.runner()(ictx, env, append(append([]string{}, base...), "init")...)
if err == nil {
m.logger.Printf("[INFO] [offbox] initialized restic repo")
return nil
}
if strings.Contains(string(out), "already initialized") || strings.Contains(string(out), "already exists") {
return nil
}
return fmt.Errorf("offbox repo unreachable / init failed: %w: %s", err, truncate(out))
default: // "other" — network/SFTP-auth/unknown; NOT orphaned. Surface as before (fail fast).
return fmt.Errorf("offbox repo unreachable: %w: %s", perr, truncate(pout))
}
}
// RunOffboxBackup backs up every off-box-toggled app's recovery unit (recovery unit + DB dumps + volume
// tars) to the SFTP repo, then prunes per the retention policy. Single-flight + migration-guarded. A
// failure (incl. a fail-fast dead-NAS error) records status + alerts the operator. Returns the first error.
func (m *Manager) RunOffboxBackup(ctx context.Context) error {
return m.runOffboxBackup(ctx, false)
}
// RunOffboxBackupWithProgress is the MANUAL („Távoli mentés most") entry point: identical work, but
// with the live progress sink installed so the page can show total bytes, percent and current app
// (v0.147.0, 4c). The nightly scheduled run keeps calling RunOffboxBackup and stays silent — nobody
// is watching a progress bar at 03:00, and a sink left installed would publish stale percentages
// into a page that never asked for them.
func (m *Manager) RunOffboxBackupWithProgress(ctx context.Context) error {
return m.runOffboxBackup(ctx, true)
}
func (m *Manager) runOffboxBackup(ctx context.Context, withProgress bool) error {
if withProgress {
defer m.beginManualProgress()()
}
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
// fork-4 atomicity gate: no offsite RUN until the repo password is confirmed escrowed under R, so no
// un-recoverable offsite ciphertext can exist. Not an error (config/UI still work) — a skip.
if !m.offboxEscrowed() {
m.logger.Printf("[INFO] [offbox] skipped — pending key escrow (no offsite run until the repo password is escrowed under R)")
return nil
}
// Offsite-repo continuity (v0.142.0): once ORPHANED, scheduled runs SKIP (the event fired on the
// detection transition — no nightly spam) until a reset clears it. The remote page shows the card.
if m.OffboxOrphaned() {
m.logger.Printf("[INFO] [offbox] skipped — offsite repo orphaned (awaiting reset)")
return nil
}
if m.migrationActive() {
m.logger.Printf("[INFO] [offbox] skipped — migration in progress")
return nil
}
if err := m.acquireRunning(); err != nil {
m.logger.Printf("[INFO] [offbox] skipped — another backup is running")
return nil // single-flight: don't race; the next scheduled run retries
}
defer m.releaseRunning()
apps := m.settings.GetOffboxApps()
// Reserved-name defense in depth (R-7b): an app keyed `_shares` would collide with the shares
// leg's restic tag and blocked-set entry. Catalog names cannot realistically produce this, but a
// silent collision would corrupt both sources, so it is refused loudly instead.
for i, a := range apps {
if a == SharesPseudoStack {
m.logger.Printf("[ERROR] [offbox] app %q uses the RESERVED shares key — excluded from the run to protect the shares leg", a)
apps = append(apps[:i:i], apps[i+1:]...)
break
}
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
// Edge-trigger for the enlarge-blocked notification: capture the PRIOR blocked set so we notify only
// apps that NEWLY cross into the blocked state (a persistently-blocked app doesn't re-notify nightly).
priorBlocked := map[string]bool{}
if t != nil {
for _, s := range t.EnlargedBlocked {
priorBlocked[s] = true
}
}
start := time.Now()
m.logger.Printf("[INFO] [offbox] backup run started (%d app(s) toggled)", len(apps))
if err := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) { o.LastStatus = "running"; o.LastError = "" }); err != nil {
m.logger.Printf("[WARN] [offbox] status persist (running) failed: %v", err)
}
var backedUp int
var missing []string
var runResult offboxRunResult
var runErr error
if usedGB, quota, over := offboxQuotaState(t); over {
// SLICE 4 soft-quota gate (pre-run): NEW backups are refused at ≥100% of the shared-model quota —
// but the retention/prune step STILL RUNS (pruning is the customer's only way back under quota;
// gating it too would deadlock them over-quota) and restore paths are untouched. The CURRENT run's
// gate uses the last-known repo size; a run that crosses 100% mid-flight finishes and the NEXT
// run refuses.
m.offboxPruneOnly(ctx, base, env)
m.offboxRecordStats(ctx, base, env) // the prune may have brought the size back down — refresh
runErr = fmt.Errorf("A távoli mentés túllépte a tárhelykeretet (%d/%d GB) — törölj régi mentéseket vagy kérj nagyobb keretet.", usedGB, quota)
} else {
runResult, runErr = m.runOffboxInternal(ctx, apps, base, env, t)
backedUp = runResult.backedUp
missing = runResult.missing
}
// Sorted names of apps whose enlargement was blocked this run (replaces the persisted set; empty clears).
var blockedNames []string
for _, b := range runResult.blocked {
blockedNames = append(blockedNames, b.stack)
}
sort.Strings(blockedNames)
// No-silent-success: apps were toggled but NOTHING was captured (every unit missing) → promote to a
// hard error so the run reports "error" and the operator is alerted, instead of a misleading ok/0.
if runErr == nil && len(apps) > 0 && backedUp == 0 {
runErr = fmt.Errorf("off-box backup produced no snapshots: %d app(s) toggled but no recovery unit was found on any connected drive (missing: %s)",
len(apps), strings.Join(missing, ", "))
}
dur := time.Since(start)
snapshots := 0
if runErr == nil {
snapshots = m.offboxRecordStats(ctx, base, env)
}
if perr := m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.LastRun = time.Now().UTC().Format(time.RFC3339)
o.LastDuration = dur.Round(time.Second).String()
if errors.Is(runErr, ErrOffboxOrphaned) {
// First-detection of the orphaned repo: RepoState (set by markOrphaned) drives the orphan
// card — do NOT surface the raw restic/sentinel text as the last-error banner.
o.LastStatus = "error"
o.LastError = ""
o.LastWarning = ""
} else if runErr != nil {
o.LastStatus = "error"
o.LastError = runErr.Error()
o.LastWarning = ""
} else {
o.LastStatus = "ok"
o.LastError = ""
o.SnapshotCount = snapshots
o.EnlargedBlocked = blockedNames // replace each run (sorted); empty slice clears it
var warns []string
// Zero-toggle honesty (take-two obs.): a configured target with NOTHING selected reports
// its emptiness instead of a bare success — the customer thinks offsite runs, but nothing
// is covered until at least one app is toggled.
// R-7b: the shares leg counts as coverage — a box whose only cloud content is its shares
// must not be told "nothing is selected".
if len(apps) == 0 && !runResult.sharesBackedUp {
warns = append(warns, "Sikeres — nincs mentésre jelölt alkalmazás")
}
if len(missing) > 0 {
warns = append(warns, fmt.Sprintf("Figyelmeztetés: %d alkalmazásnak nincs elérhető mentése, ezek kimaradtak: %s",
len(missing), strings.Join(missing, ", ")))
}
// 3a: capture-gap warnings (structurally-refused / on-disk-missing mandatory paths, undeployed).
warns = append(warns, runResult.warns...)
// 3a: the pre-push enlargement gate blocked some apps' userdata — config+DB still saved.
// R-7b: the shares source is not an "app" and its degraded floor is the DEFINITIONS, not a
// recovery unit — so it gets its own sentence and is excluded from the app count. The
// persisted EnlargedBlocked set keeps the RAW `_shares` key (it is a lookup key the
// templates index by); only this prose maps it through the display vocabulary.
var blockedApps []string
sharesBlocked := false
for _, n := range blockedNames {
if n == SharesPseudoStack {
sharesBlocked = true
continue
}
blockedApps = append(blockedApps, n)
}
if len(blockedApps) > 0 {
warns = append(warns, fmt.Sprintf("Figyelmeztetés: a tárhelykeret miatt %d alkalmazásnál csak konfiguráció- és adatbázis-mentés készült: %s.",
len(blockedApps), strings.Join(blockedApps, ", ")))
}
if sharesBlocked {
warns = append(warns, sharesBlockedWarning())
}
// SLICE 4: approaching the soft quota (≥80%, <100%) — warn on an otherwise-OK run.
if qw := offboxQuotaWarning(o); qw != "" {
warns = append(warns, qw)
}
o.LastWarning = strings.Join(warns, " ")
}
}); perr != nil {
m.logger.Printf("[WARN] [offbox] status persist (final) failed: %v", perr)
}
// The orphaned case has its OWN dedicated event (offbox_repo_orphaned) — do NOT also fire the
// generic backup-failed notification (no double/raw alert; the orphan card is the customer surface).
if m.offboxNotify != nil && !errors.Is(runErr, ErrOffboxOrphaned) {
m.offboxNotify(dur, snapshots, runErr)
}
// Edge-triggered enlarge-blocked notification: only apps that NEWLY crossed into the blocked state
// (vs the prior persisted set) notify — a persistently-blocked app never re-notifies nightly. Uses
// the pre-run last-known repo size (the same figure the gate used).
if runErr == nil && m.offboxEnlargeBlockedNotify != nil && t != nil && t.QuotaGB > 0 {
usedGB := int(t.RepoSizeBytes / offboxGiB)
for _, b := range runResult.blocked {
if !priorBlocked[b.stack] {
// DISPLAY BOUNDARY (R-7b): the notification is a customer-facing surface (it becomes a
// Hungarian e-mail), so the reserved `_shares` key is mapped here — and ONLY here plus
// the warning prose above. The persisted set and the restic tag stay raw.
m.offboxEnlargeBlockedNotify(DisplayStackName(b.stack), b.estBytes, usedGB, t.QuotaGB)
}
}
}
switch {
case errors.Is(runErr, ErrOffboxOrphaned):
m.logger.Printf("[WARN] [offbox] run skipped — offsite repo orphaned (card shown; awaiting reset)")
return nil // the orphaned STATE + event are the signal; not a hard run error for the scheduler
case runErr != nil:
m.logger.Printf("[ERROR] [offbox] backup failed after %s: %v", dur.Round(time.Second), runErr)
case len(missing) > 0:
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s) backed up, %d skipped (no unit), %d snapshot(s), %s",
backedUp, len(missing), snapshots, dur.Round(time.Second))
default:
m.logger.Printf("[INFO] [offbox] backup OK: %d app(s) backed up, %d snapshot(s), %s", backedUp, snapshots, dur.Round(time.Second))
}
return runErr
}
// offboxCandidateNSRoots is the durable, deployment-state-INDEPENDENT set of felhom-data namespace roots
// to search for a recovery unit: every registered SCHEDULABLE (non-decommissioned) storage path the
// system-data fallback drive, deduped by resolved nsRoot string. This deliberately does NOT consult
// GetAppDrivePath/AppNamespaceRoot — those read the app's LIVE app.yaml HDD_PATH and silently fall back
// to systemDataPath when the app isn't currently deployed, which made offbox look on the wrong drive
// (DIAG root cause). A disconnected drive's path simply isn't present on disk → os.Stat fails → the unit
// is "not here" (correct: a disconnected drive can't be offsited). Boundary: a decommissioned or
// non-schedulable drive is not searched (not an active managed backup location).
func (m *Manager) offboxCandidateNSRoots() []string {
seen := map[string]bool{}
var nsRoots []string
add := func(nr string) {
if nr != "" && !seen[nr] {
seen[nr] = true
nsRoots = append(nsRoots, nr)
}
}
for _, sp := range m.settings.GetSchedulableStoragePaths() {
add(m.namespaceRoot(sp.Path))
}
if m.systemDataPath != "" {
add(m.namespaceRoot(m.systemDataPath))
}
return nsRoots
}
// discoverOffboxUnit locates an app's recovery unit (backups/primary/<app>) across the candidate nsRoots.
// Returns the src path + true when exactly one exists; when the SAME app's unit exists on more than one
// drive (drive churn / a stale copy left behind), it returns the NEWEST by manifest CreatedAt (falling
// back to the unit dir mtime) and WARNs about the others. Independent of the app's live deploy state.
func (m *Manager) discoverOffboxUnit(app string) (string, bool) {
var foundSrc, foundManifest []string
for _, nr := range m.offboxCandidateNSRoots() {
p := RecoveryUnitPath(nr, app)
if fi, err := os.Stat(p); err == nil && fi.IsDir() {
foundSrc = append(foundSrc, p)
foundManifest = append(foundManifest, RecoveryUnitManifestPath(nr, app))
}
}
switch len(foundSrc) {
case 0:
return "", false
case 1:
return foundSrc[0], true
default:
best := 0
bestT := offboxUnitTime(foundSrc[0], foundManifest[0])
for i := 1; i < len(foundSrc); i++ {
if t := offboxUnitTime(foundSrc[i], foundManifest[i]); t.After(bestT) {
best, bestT = i, t
}
}
var others []string
for i, p := range foundSrc {
if i != best {
others = append(others, p)
}
}
m.logger.Printf("[WARN] [offbox] %s: multiple recovery units found, using newest (%s); ignoring: %s",
app, foundSrc[best], strings.Join(others, ", "))
return foundSrc[best], true
}
}
// offboxUnitTime returns a recovery unit's timestamp for the multi-copy tiebreak: the manifest's
// CreatedAt (RFC3339) if readable, else the unit dir's mtime (zero if neither is available).
func offboxUnitTime(src, manifestPath string) time.Time {
if mf := readManifest(manifestPath); mf != nil {
if t, err := time.Parse(time.RFC3339, mf.CreatedAt); err == nil {
return t
}
}
if fi, err := os.Stat(src); err == nil {
return fi.ModTime()
}
return time.Time{}
}
// offboxRunResult carries the outcome of a per-app offbox run: how many apps were backed up, which
// had no discoverable unit (skipped), which had their enlargement quota-blocked (unit-only), and the
// aggregated Hungarian customer warnings (capture gaps + undeployed).
type offboxRunResult struct {
backedUp int
missing []string
blocked []offboxBlocked
warns []string
// sharesBackedUp (R-7b) records that the sibling shares leg produced a snapshot this run. It keeps
// the zero-toggle honesty notice honest: a box with no app toggled but shares in the cloud is NOT
// "nothing is covered".
sharesBackedUp bool
}
// runOffboxInternal does the repo-ensure + per-app DISCOVER → capture-set → gate → multi-path backup +
// prune. Caller holds the running flag. Each app's snapshot is ONE multi-path restic snapshot
// (recovery unit + the app's MANDATORY offsite capture set, §6). The pre-push enlargement gate (§9,
// decision #1) blocks only the ENLARGEMENT — the unit-only push always continues. Returns the result +
// the first hard error (repo-ensure or a restic backup exec failure).
func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []string, t *settings.OffboxTarget) (res offboxRunResult, err error) {
if rerr := m.ensureOffboxRepo(ctx, base, env); rerr != nil {
return res, rerr // fail fast (dead NAS surfaces here)
}
// Pre-run hygiene: clear any lock restic can prove stale before we start (cheap; the --remove-all
// crash-lock escalation lives in resticStep for the locks restic can't self-detect).
m.unlockStale(ctx, base, env)
var firstErr error
for _, stack := range apps {
src, ok := m.discoverOffboxUnit(stack)
if !ok {
m.logger.Printf("[WARN] [offbox] %s: no recovery unit found on any connected drive — skipping", stack)
res.missing = append(res.missing, stack)
continue
}
// Task 3-core TierOffsite capture set: mandatory userdata paths added to the unit snapshot,
// plus loud warnings for structurally-refused / on-disk-missing mandatory paths (SP-3.4).
extra, capWarns := m.offboxCaptureSet(stack)
res.warns = append(res.warns, capWarns...)
// Pre-push enlargement gate (§9): if last-known repo raw-data bytes + the mandatory-set estimate
// would cross the soft quota, push UNIT-ONLY (protection never regresses) and record the block.
if len(extra) > 0 && t != nil && t.QuotaGB > 0 {
var est int64
for _, p := range extra {
est += m.offboxSize()(p)
}
if t.RepoSizeBytes+est >= int64(t.QuotaGB)*offboxGiB {
m.logger.Printf("[INFO] [offbox] %s: enlargement blocked by quota (est %s + repo %s ≥ %d GB) — unit-only push continues",
stack, humanizeBytes(est), humanizeBytes(t.RepoSizeBytes), t.QuotaGB)
res.blocked = append(res.blocked, offboxBlocked{stack: stack, estBytes: est})
extra = nil
}
}
args := append([]string{"backup", "--tag", "felhom-offbox", "--tag", stack, src}, extra...)
bctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
// Streaming twin (4c): identical to resticStep unless a MANUAL run installed a progress sink,
// in which case it adds --json and feeds the parsed status lines to the page's poll.
out, berr := m.resticBackupStep(bctx, env, base, "backup:"+stack, stack, args...)
cancel()
if berr != nil {
m.logger.Printf("[ERROR] [offbox] backup %s failed: %v: %s", stack, berr, truncate(out))
if firstErr == nil {
firstErr = fmt.Errorf("offbox backup %s: %w", stack, berr)
}
continue
}
res.backedUp++
m.logger.Printf("[INFO] [offbox] backed up %s (%s, %d mandatory path(s))", stack, src, len(extra))
}
// R-7b: the SHARES leg runs AFTER the per-app loop and BEFORE retention, so `forget --group-by
// host,tags` covers the `_shares` group for free. It is placed BEFORE the firstErr return on
// purpose: share protection must not be dropped because some unrelated app failed to push.
sharesRes, sharesErr := m.runOffboxSharesLeg(ctx, base, env, t)
m.recordSharesOffsiteStatus(sharesRes)
res.warns = append(res.warns, sharesRes.warns...)
if sharesRes.blocked {
res.blocked = append(res.blocked, offboxBlocked{stack: SharesPseudoStack, estBytes: sharesRes.estBytes})
}
if sharesRes.ran {
res.sharesBackedUp = true
}
if sharesErr != nil && firstErr == nil {
firstErr = sharesErr
}
if firstErr != nil {
return res, firstErr
}
// Retention: keep a sane window, prune the rest. SP-2: `--group-by host,tags` so an app's OLD
// unit-only-shape snapshots share a group with its NEW enlarged shape (same <stack> tag) and age
// out naturally — the default host,paths grouping would strand old-shape snapshots in their own
// permanently-retained group. prune takes an EXCLUSIVE lock (the C2 stale-lock step) → resticStep.
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
defer cancel()
if out, ferr := m.resticStep(fctx, env, base, "prune", "forget", "--group-by", "host,tags", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune"); ferr != nil {
// A prune failure is non-fatal to the backup itself (data is safe) — log, don't fail the run.
m.logger.Printf("[WARN] [offbox] forget --prune failed (backups are safe): %v: %s", ferr, truncate(out))
}
return res, nil
}
// offboxGiB is the soft-quota unit: QuotaGB counts binary gigabytes (GiB) of restic REPO SIZE. Since
// v0.134.0 the repo size is measured with `stats --mode raw-data` (actual deduplicated+compressed
// bytes — what the customer's Storage Box really fills), NOT the old modeless restore-size which
// multiplied by the retained-snapshot count (SP-1). The displayed size drops one-time after deploy.
const offboxGiB = int64(1) << 30
// OffboxReportStatus is the NON-SECRET offsite summary carried on the hub report (SLICE 4) — the input
// to the hub's OffsiteChecker (fill + staleness alerts). nil when no offbox target is configured.
type OffboxReportStatus struct {
Enabled bool `json:"enabled"`
EscrowState string `json:"escrow_state"`
LastRun string `json:"last_run,omitempty"` // RFC3339
LastStatus string `json:"last_status,omitempty"` // "ok" | "error" | "running"
SnapshotCount int `json:"snapshot_count"`
RepoSizeBytes int64 `json:"repo_size_bytes"`
QuotaGB int `json:"quota_gb"`
}
// OffboxReportStatus returns the offsite summary for the hub report (nil = not configured; the hub's
// checker treats absence as "nothing to watch" — pre-v0.109 reports look the same).
func (m *Manager) OffboxReportStatus() *OffboxReportStatus {
t := m.settings.GetOffboxTarget()
if t == nil || !t.Enabled {
return nil
}
return &OffboxReportStatus{
Enabled: true, EscrowState: t.EscrowState, LastRun: t.LastRun, LastStatus: t.LastStatus,
SnapshotCount: t.SnapshotCount, RepoSizeBytes: t.RepoSizeBytes, QuotaGB: t.QuotaGB,
}
}
// offboxQuotaState evaluates the SLICE 4 soft-quota gate against the LAST-KNOWN repo size (a failed stats
// call keeps the previous value — stale-but-safe). quota<=0 = no soft limit (dedicated/manual targets).
func offboxQuotaState(t *settings.OffboxTarget) (usedGB, quotaGB int, over bool) {
if t == nil || t.QuotaGB <= 0 {
return 0, 0, false
}
usedGB = int(t.RepoSizeBytes / offboxGiB)
return usedGB, t.QuotaGB, t.RepoSizeBytes >= int64(t.QuotaGB)*offboxGiB
}
// offboxQuotaWarning returns the Hungarian ≥80% (<100%) usage notice, or "" (quota unset / usage fine /
// already over — over-quota is the run-refusal error, not a warning).
func offboxQuotaWarning(t *settings.OffboxTarget) string {
if t == nil || t.QuotaGB <= 0 || t.RepoSizeBytes <= 0 {
return ""
}
limit := int64(t.QuotaGB) * offboxGiB
pct := t.RepoSizeBytes * 100 / limit
if pct < 80 || t.RepoSizeBytes >= limit {
return ""
}
return fmt.Sprintf("A távoli mentés a keret %d%%-át használja (%d/%d GB).", pct, t.RepoSizeBytes/offboxGiB, t.QuotaGB)
}
// OffboxQuotaPercent returns the usage percentage for the /backups usage bar (0 when no quota/size).
func OffboxQuotaPercent(t *settings.OffboxTarget) int {
if t == nil || t.QuotaGB <= 0 || t.RepoSizeBytes <= 0 {
return 0
}
pct := int(t.RepoSizeBytes * 100 / (int64(t.QuotaGB) * offboxGiB))
if pct > 100 {
pct = 100
}
return pct
}
// offboxPruneOnly runs ONLY the retention/prune step (the over-quota path: new backups are refused but
// pruning must stay available — it is the only way back under the quota). Repo-ensure first so a fresh
// target still fails loudly; errors are non-fatal (same as the regular run's prune).
func (m *Manager) offboxPruneOnly(ctx context.Context, base, env []string) {
if rerr := m.ensureOffboxRepo(ctx, base, env); rerr != nil {
m.logger.Printf("[WARN] [offbox] over-quota prune: repo unreachable: %v", rerr)
return
}
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
defer cancel()
// SP-2: `--group-by host,tags` (mirrors runOffboxInternal's forget) so old unit-only-shape snapshots
// age out with the enlarged shape instead of stranding in a permanently-retained host,paths group.
fargs := append(append([]string{}, base...), "forget", "--group-by", "host,tags", "--keep-daily", "7", "--keep-weekly", "4", "--keep-monthly", "6", "--prune")
if out, ferr := m.runner()(fctx, env, fargs...); ferr != nil {
m.logger.Printf("[WARN] [offbox] over-quota prune failed: %v: %s", ferr, truncate(out))
} else {
m.logger.Printf("[INFO] [offbox] over-quota: prune executed (new backups refused until under quota)")
}
}
// offboxRecordStats reads the snapshot count (best-effort) for the UI; also fills repo size when stats works.
func (m *Manager) offboxRecordStats(ctx context.Context, base, env []string) int {
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
defer cancel()
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
if err != nil {
return 0
}
var snaps []struct {
ID string `json:"id"`
}
if json.Unmarshal(out, &snaps) != nil {
return 0
}
// Repo size (best-effort, RAW-DATA mode). SP-1: `--mode raw-data` reports the actual
// deduplicated+compressed repo bytes (what the Storage Box really fills), not the modeless
// restore-size that multiplies by the retained-snapshot count. Bytes feed the soft-quota gate
// (SLICE 4); a failed stats call keeps the last-known value (stale-but-safe). RAW-DATA TRAP:
// total_file_count is 0 in this mode — read total_size only.
if so, serr := m.runner()(sctx, env, append(append([]string{}, base...), "stats", "--mode", "raw-data", "--json")...); serr == nil {
var st struct {
TotalSize int64 `json:"total_size"`
}
if json.Unmarshal(so, &st) == nil && st.TotalSize > 0 {
human := humanizeBytes(st.TotalSize)
_ = m.settings.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
o.RepoSizeHuman = human
o.RepoSizeBytes = st.TotalSize
})
}
}
return len(snaps)
}
// RestoreOffbox restores an app's latest off-box snapshot to destDir (a scratch/verify location — it does
// NOT overwrite live data). Returns an error on failure (checks restic's own exit code).
func (m *Manager) RestoreOffbox(ctx context.Context, stackName, destDir string) error {
if !m.OffboxConfigured() {
return fmt.Errorf("off-box backup not configured")
}
if !isSafeStackName(stackName) {
return fmt.Errorf("invalid stack name")
}
if err := os.MkdirAll(destDir, 0o755); err != nil {
return fmt.Errorf("restore dir: %w", err)
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
rctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
defer cancel()
m.unlockStale(rctx, base, env) // pre-restore hygiene (crash-lock self-heal is in resticStep)
out, err := m.resticStep(rctx, env, base, "restore:"+stackName, "restore", "latest", "--tag", stackName, "--target", destDir)
if err != nil {
return fmt.Errorf("offbox restore %s: %w: %s", stackName, err, truncate(out))
}
m.logger.Printf("[INFO] [offbox] restored %s → %s", stackName, destDir)
return nil
}
// isSafeStackName guards a stack name used as a restic tag / path component.
func isSafeStackName(s string) bool {
if s == "" || len(s) > 64 {
return false
}
for _, c := range s {
if (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '-' || c == '_' {
continue
}
return false
}
return true
}
// truncate caps subprocess output for a log line + strips a trailing newline.
func truncate(b []byte) string {
s := strings.TrimSpace(string(b))
if len(s) > 400 {
return s[:400] + "…"
}
return s
}