Files
felhom-controller/controller/internal/backup/recovery_unit.go
T
admin 0d402f711d
gates / gates (push) Successful in 13s
v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.

The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.

Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.

Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.

Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.

Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-09-13 11:41:31 +02:00

579 lines
26 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package backup
import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
"gopkg.in/yaml.v3"
)
// RecoveryManifest describes an app's self-contained recovery unit.
//
// The unit on a drive is `<nsRoot>/backups/primary/<app>/` and contains:
//
// compose/ docker-compose.yml + .felhom.yml + app.yaml (0600; carries the PORTABLE secrets)
// db-dumps/ app-consistent DB dump(s) (written by the dump flow)
// volume-dumps/ named-volume tars (written by the dump flow)
// manifest.json this file
//
// D5 (schema 2) changed what the unit holds. Before it held NO secret at all, which made
// "restore from the drive alone" false: the fast, local, customer-doable Tier-1/2 restore secretly
// depended on the slow, operator-driven whole-guest restore, because a data-encrypting key or a DB
// password absent from the guest cannot be regenerated without rendering the restored data
// unreachable. The unit now carries the PORTABLE secret class (stacks.PortableSecretEnvVars) in its
// 0600 app.yaml, and Tier-1/2 needs the DRIVE AND NOTHING ELSE.
//
// It still holds NO `type: password` admin login (those are internet-reachable, so their blast radius
// is not bounded by the drive — they stay in the guest and are regenerated on restore) and NOT the
// Docker image, only the pinned tag(s), re-pulled on restore. SecretSource records the split.
//
// A schema-1 unit carries no secrets: the restore degrades to the pre-D5 guest-only behaviour rather
// than failing, and the next capture rewrites it (the app.yaml checksum changes).
type RecoveryManifest struct {
SchemaVersion int `json:"schema_version"`
AppName string `json:"app_name"`
DisplayName string `json:"display_name"`
ControllerVer string `json:"controller_version"`
CreatedAt string `json:"created_at"`
Drive string `json:"drive"` // HDD_PATH (in-guest mount)
NamespaceRoot string `json:"namespace_root"` // resolved felhom-data namespace root
ImagePins []string `json:"image_pins"` // image NOT stored — re-pulled on restore
SecretEnvVars []string `json:"secret_env_vars"` // NAMES of every secret/password field
DataKeyEnvVars []string `json:"data_key_env_vars"` // fail-closed gate on restore
SecretSource string `json:"secret_source"` // human note: where secrets come from
ConfigFiles []string `json:"config_files"` // captured into compose/
DBDumps []string `json:"db_dumps"`
VolumeDumps []string `json:"volume_dumps"`
Checksums map[string]string `json:"checksums"` // sha256 of captured compose/ files
// PortableSecretEnvVars (D5) are the NAMES of the secrets this unit's app.yaml CARRIES. Names only
// — the manifest is 0644 and never holds a value. The restore reads it to know which app.yaml env
// entries are secrets rather than plain config; absent (schema 1) ⇒ the unit carries none.
PortableSecretEnvVars []string `json:"portable_secret_env_vars,omitempty"`
// R-43/R-44 (v0.148.0): the coherence stamp. An offsite run refreshes the dumps FIRST and then
// captures the unit, so a manifest carrying an OffsiteRunID asserts "the db-dumps/ in this unit
// were taken by that run" — i.e. the snapshot is an internally coherent {DB@T, files@T} pair.
// A manifest WITHOUT these fields is a pre-v0.148 unit whose dump age is unknown and may skew
// arbitrarily from the files beside it (the DIAG-immich-restore-2026-07-19 failure); the restore
// confirm surfaces that honestly rather than blocking. Empty on the periodic refresh, which must
// never claim a coherence it did not establish — it carries the prior stamp forward instead.
OffsiteRunID string `json:"offsite_run_id,omitempty"`
DumpsAt string `json:"dumps_at,omitempty"` // RFC3339 UTC — when this run's dump leg finished
}
// SetVersion records the controller version stamped into recovery-unit manifests.
func (m *Manager) SetVersion(v string) {
m.mu.Lock()
m.version = v
m.mu.Unlock()
}
// SetTier2Notifier wires the notification callback invoked after each Tier 2 copy.
func (m *Manager) SetTier2Notifier(fn func(stackName, destLabel string, dur time.Duration, err error)) {
m.tier2Notify = fn
}
// CaptureRecoveryUnit writes/refreshes an app's recovery unit: it captures the compose + metadata +
// an app.yaml carrying the PORTABLE secret class (D5) into compose/, enumerates the DB/volume dumps
// already present, and writes manifest.json. It never writes the Docker image (only the pinned tag),
// and never writes a WITHHELD secret — the split is decided in buildUnitAppYaml, pinned by
// TestCaptureRecoveryUnitCarriesPortableSecretsOnly.
//
// Idempotent: it builds the captured content in memory first and SKIPS all writes when the unit is
// already current (same config checksums, same dump set, same controller version) — so it can run on
// the periodic status refresh without thrashing a spinning USB drive.
func (m *Manager) CaptureRecoveryUnit(stackName string) error {
if m.stackProvider == nil {
return fmt.Errorf("no stack provider")
}
info, ok := m.stackProvider.GetStackRecoveryInfo(stackName)
if !ok {
return fmt.Errorf("stack %q not found", stackName)
}
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" || !filepath.IsAbs(drivePath) {
return fmt.Errorf("cannot determine absolute drive path for %s", stackName)
}
nsRoot := m.namespaceRoot(drivePath)
// Build the captured config CONTENT in memory (no writes yet) so we can checksum-compare.
type capFile struct {
name string
data []byte
perm os.FileMode
}
var files []capFile
checksums := make(map[string]string)
var configFiles []string
for _, fname := range []string{"docker-compose.yml", ".felhom.yml"} {
data, err := os.ReadFile(filepath.Join(info.StackDir, fname))
if err != nil {
continue // optional — capture whichever exist
}
files = append(files, capFile{fname, data, 0644})
checksums[fname] = sha256Hex(data)
configFiles = append(configFiles, fname)
}
appYaml := buildUnitAppYaml(info)
files = append(files, capFile{"app.yaml", appYaml, 0600})
checksums["app.yaml"] = sha256Hex(appYaml)
configFiles = append(configFiles, "app.yaml")
// R-361: `db_dumps` lists the app's OWN dumps and NOT the `pre-restore-*` undo copies.
//
// THE DECISION, and the reasoning, because a decision recorded only in a commit message is a
// decision nobody finds: the undo copies are LOCAL material for a restore that went wrong, not
// part of the app's recovery set. Nothing reads them from the manifest — the reconstitution skips
// `pre-restore-*` at three sites, and a grep of every consumer of `Manifest.DBDumps` found only
// this file (the declaration, this enumeration, and the change-detection compare below). Listing
// them meant three copies per app were enumerated into the manifest and pushed off-site
// permanently, for no recovery value.
//
// It also makes the compare below STABLE: the undo copies come and go with every restore and
// every prune, so including them forced a manifest rewrite each time for a change that says
// nothing about the app's backup.
//
// The files themselves are NOT deleted and NOT hidden — their visibility is a recorded design
// (see preRestoreDumpPrefix). This is about the manifest only.
dbDumps := filterOutUndoCopies(listFileNames(AppDBDumpPath(nsRoot, stackName), ".sql"))
volDumps := listFileNames(AppVolumeDumpPath(nsRoot, stackName), ".tar")
version := m.versionLocked()
manifestPath := RecoveryUnitManifestPath(nsRoot, stackName)
cur := readManifest(manifestPath)
// R-43/R-44: the coherence stamp of the offsite run currently in flight ("" on the periodic
// refresh and on the local dump run). When empty we CARRY THE PRIOR STAMP FORWARD rather than
// blanking it — a periodic refresh must neither claim a coherence it did not establish nor
// destroy the record of one that a real run did.
runID, dumpsAt := m.offsiteRunStamp()
if runID == "" && cur != nil {
runID, dumpsAt = cur.OffsiteRunID, cur.DumpsAt
}
// R-379/R-361: bound the undo copies BEFORE the already-current check, not after it.
//
// It used to sit at the end of this function, and that was fine only while `db_dumps` listed the
// `pre-restore-*` files: a new undo copy changed the list, so the check below never short-circuited
// and the prune always ran. R-361 excluded those files from `db_dumps` — correctly — and that made
// the list STABLE across restores, so the early return began firing and the prune became
// unreachable in exactly the case it exists for. Measured on demo-hp 2026-08-22: four undo copies
// on disk against a cap of three, immediately after the change.
//
// It is housekeeping on the dump directory and has nothing to do with whether the manifest needs
// rewriting, so it belongs above the check. Pruning cannot disturb `dbDumps`, which no longer
// contains those names at all.
m.pruneUndoCopies(AppDBDumpPath(nsRoot, stackName), stackName)
// Skip if the unit is already current — avoids needless drive writes on the periodic refresh.
// The run-id is part of "current": an offsite run must re-stamp the manifest even when nothing
// else changed, because the stamp is exactly the claim the restore path reads.
if cur != nil &&
cur.ControllerVer == version &&
stringMapEqual(cur.Checksums, checksums) &&
stringSliceEqual(cur.DBDumps, dbDumps) &&
stringSliceEqual(cur.VolumeDumps, volDumps) &&
cur.OffsiteRunID == runID {
return nil
}
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
if err := os.MkdirAll(composeDir, 0755); err != nil {
return fmt.Errorf("creating recovery-unit compose dir: %w", err)
}
for _, f := range files {
if err := atomicWrite(filepath.Join(composeDir, f.name), f.data, f.perm); err != nil {
return fmt.Errorf("capturing %s: %w", f.name, err)
}
}
manifest := &RecoveryManifest{
SchemaVersion: 2, // D5: compose/app.yaml carries the portable secret class
AppName: stackName,
DisplayName: info.DisplayName,
ControllerVer: version,
CreatedAt: time.Now().UTC().Format(time.RFC3339),
Drive: drivePath,
NamespaceRoot: nsRoot,
ImagePins: info.ImagePins,
SecretEnvVars: info.SecretEnvVars,
DataKeyEnvVars: info.DataKeyEnvVars,
PortableSecretEnvVars: info.PortableSecretEnvVars,
SecretSource: "portable secrets (data keys, DB passwords, internal signing secrets) are IN this unit's compose/app.yaml (0600); internet-reachable admin logins are NOT, and come from the guest's app.yaml or are regenerated on restore",
ConfigFiles: configFiles,
DBDumps: dbDumps,
VolumeDumps: volDumps,
Checksums: checksums,
OffsiteRunID: runID,
DumpsAt: dumpsAt,
}
if err := writeManifest(manifestPath, manifest); err != nil {
return fmt.Errorf("writing manifest: %w", err)
}
// Counts and NAMES only — never a value (D5 puts more secrets through this path than before).
m.logger.Printf("[INFO] [backup] Recovery unit captured for %s → %s (images=%d, secrets-referenced=%d, data_keys=%d, portable-carried=%d/%d, withheld=%d)",
stackName, RecoveryUnitPath(nsRoot, stackName), len(info.ImagePins), len(info.SecretEnvVars),
len(info.DataKeyEnvVars), len(info.PortableSecrets), len(info.PortableSecretEnvVars),
len(withheldSecretNames(info)))
// R-379: bound the undo copies. AFTER a successful capture and never before one — the capture is
// the point at which nothing is depending on those files, whereas the restore path is precisely
// where deleting one would remove the undo at the moment it is needed.
return nil
}
// UnitSpace is the target filesystem's occupancy at the moment a capture failed — the numbers that
// answer "why" without an operator logging in. Nil when the filesystem could not be read at all
// (system.GetDiskUsage returns nil on error), which is reported as unknown rather than as full.
type UnitSpace struct {
Path string
UsedGB float64
AvailGB float64
TotalGB float64
UsedPercent float64
}
// String renders the space figures for an operator, or says plainly that they are unknown. An absent
// reading must never render as zeros — "0 GB free" and "we could not look" are opposite diagnoses.
func (u *UnitSpace) String() string {
if u == nil {
return "target filesystem usage unavailable"
}
return fmt.Sprintf("%s: %.1f/%.1f GB used (%.0f%%), %.1f GB free",
u.Path, u.UsedGB, u.TotalGB, u.UsedPercent, u.AvailGB)
}
// SetUnitNotify wires the per-app recovery-unit capture failure alert (R-158 / R-167). INIT-ONLY —
// call once at startup, in main.go, alongside SetOffboxNotify. Nil-safe: an unwired seam is silently
// the pre-v0.191.0 behaviour, which is a `[WARN]` line and nothing else.
func (m *Manager) SetUnitNotify(fn func(stackName string, err error, usage *UnitSpace)) {
m.unitNotify = fn
}
// unitTargetSpace reads the occupancy of the filesystem a unit for `stackName` would be written to.
// Nil on an unreadable path — never a fabricated zero (§8.4: an unreadable filesystem is not a full
// one, and the drive gate already owns the absent-drive case).
func (m *Manager) unitTargetSpace(stackName string) *UnitSpace {
path := m.GetAppDrivePath(stackName)
if path == "" {
return nil
}
di := system.GetDiskUsage(path)
if di == nil {
return nil
}
return &UnitSpace{
Path: path, UsedGB: di.UsedGB, AvailGB: di.AvailGB,
TotalGB: di.TotalGB, UsedPercent: di.UsedPercent,
}
}
// ── The capture floor (R-165 / decision B2) ──────────────────────────────────────────────────────
//
// WHAT IT REPLACES. Until the `mp1`→`mp0` merge, the 20 G backup partition was a BULKHEAD as well as
// a ceiling: an app whose unit outgrew it was refused per app, its last good unit preserved
// byte-identical, and the overflow **could not reach `/var/lib/docker`** because that was a different
// filesystem. After the merge it can, and a full Docker data-root is a stopped box, not a slow one.
// This floor is that bulkhead, done deliberately instead of by accident.
//
// IT IS ABOUT THE FILESYSTEM'S HEADROOM, NEVER THE UNIT'S SIZE. A per-unit size cap would be R-163
// rebuilt inside one volume — the wall moved rather than removed — so a large unit on a filesystem
// with ample room is captured, whatever its size.
//
// IT REFUSES; IT NEVER DELETES. Nothing on this filesystem is generational: a unit is ONE fixed path
// per app (`backups/primary/<app>`) refreshed in place, and a DB dump is `<stack>-<dbtype>.sql`, also
// fixed. So "prune the oldest" could only mean deleting a DIFFERENT app's only local recovery unit to
// make room for this one, and that is not a trade this system makes. `pruneStalePrimaryDirs` is NOT a
// retention policy — it removes ORPHANED directories left when an app moves drives, and has no notion
// of age — so it must never be repurposed here.
const (
// FloorUsedPercent / FloorFreeGiB — the reserve. Two terms, whichever binds first, the same shape
// as `internal/fillwatch` (proven live on 2026-08-02: the critical alert fired on the free-byte
// term at 91% used, where a percent-only rule stayed silent).
//
// THEY SIT DELIBERATELY BEYOND fillwatch's CRITICAL BAND (95% / 2 GiB), so the customer is ALWAYS
// warned before a refusal can happen. A floor that fires before its own warning is a silent
// failure wearing a threshold; `TestFloorSitsBelowTheCriticalWarningBand` pins the ordering.
//
// 1 GiB is the reserve, not a working budget: §7.5 measures a DB-backed app's unit at up to ~2× its
// data, so no fixed number can guarantee a capture fits. What this guarantees is different and is
// the bulkhead's actual job — that a capture cannot consume the last of the space the container
// runtime needs to keep running.
FloorUsedPercent = 97.0
FloorFreeGiB = 1.0
)
// ErrCaptureFloor marks an app's backup refused for headroom. It is a REFUSAL, not a failure of the
// backup machinery — the distinction matters to a reader of the alert, which is why the message names
// the reserve rather than reporting an I/O error.
//
// R-181 widened what it covers: it now refuses the app's DB dump, volume dump and capture together
// (see admission.go), not the capture alone. The sentinel keeps its name because callers match it and
// "capture" still reads correctly for "capturing this app's backup"; the MESSAGE is what changed, and
// the message is what an operator sees.
var ErrCaptureFloor = errors.New("refused: backing up this app would leave the filesystem below the reserve")
// floorVerdict is the PURE predicate: given a reading and this app's estimated write, does the floor
// refuse, and on which term? Separated so the thresholds are unit-testable without a filesystem, a
// stack provider or a clock.
//
// TWO QUESTIONS, NOT ONE (R-181). "Is the filesystem already below the reserve?" is the headroom term
// and was all B2 asked. "Would THIS app's write take it below?" is the size term, and its absence is
// how an app was admitted at 96% used and then allowed to write 2 GB. Both terms are evaluated
// against BOTH thresholds — a large write can cross the percentage bound on a small volume and the
// free-byte bound on a large one, which is the same reason the reserve has two terms at all.
//
// §8.4 — A NIL READING NEITHER REFUSES NOR WARNS. An unreadable filesystem is the drive gate's
// business and has its own alert; refusing on it would block every backup on a box whose drive merely
// blipped, and warning on it would be a false alarm with a misleading cause.
//
// estGiB == 0 (no previous dump to estimate from) degrades to the headroom term alone, deliberately:
// refusing an app that has never been backed up would make the FIRST backup the one that can never
// happen (Scenario E).
func (m *Manager) floorVerdict(u *UnitSpace, estGiB float64) (*UnitSpace, floorReason) {
if u == nil {
return nil, floorAdmit
}
if u.UsedPercent >= FloorUsedPercent || u.AvailGB < FloorFreeGiB {
return u, floorHeadroom
}
if estGiB > 0 {
availAfter := u.AvailGB - estGiB
usedAfter := u.UsedPercent
if u.TotalGB > 0 {
usedAfter = (u.UsedGB + estGiB) / u.TotalGB * 100
}
if availAfter < FloorFreeGiB || usedAfter >= FloorUsedPercent {
return u, floorSize
}
}
return u, floorAdmit
}
// readUnitSpace goes through the seam when one is injected, so a test can state the filesystem's
// occupancy as an input instead of manufacturing it on a real disk. Nil seam → the real statfs.
func (m *Manager) readUnitSpace(stackName string) *UnitSpace {
if m.unitSpaceFn != nil {
return m.unitSpaceFn(stackName)
}
return m.unitTargetSpace(stackName)
}
// captureAllRecoveryUnits refreshes the recovery unit for every deployed stack. Best-effort:
// a per-app failure is logged, NOTIFIED (R-158), and does not abort the others.
//
// R-181: the reserve is consulted through `admitApp`, which is the SAME verdict the DB-dump and
// volume-dump legs of this run already consulted for this app. When a run is in flight the answer
// here is a memo lookup — an app refused before its first write is refused here too, silently,
// because it was already alerted once. Outside a run (the periodic status refresh) it decides fresh.
func (m *Manager) captureAllRecoveryUnits() {
if m.stackProvider == nil {
return
}
for _, stack := range m.stackProvider.ListDeployedStacks() {
drivePath := m.GetAppDrivePath(stack.Name)
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
continue // drive not writable — skip, the existing unit stays as-is
}
// Slice 4: a HELD app's unit is its RESTORE POINT, and the hold text names that copy's date.
// Re-capturing it would write the definition the app is held ON (after a failed update: the new
// version that would not start) into the unit, and the next Tier-2 run would mirror it over the
// copy the customer was told to restore from. The unit stays exactly as it was until the hold is
// lifted.
if m.isHeld(stack.Name) {
continue
}
m.noteAttempted(stack.Name)
// The reserve, checked BEFORE anything is written. Per app, and the loop continues.
if !m.admitApp(stack.Name) {
continue
}
if err := m.CaptureRecoveryUnit(stack.Name); err != nil {
m.noteFailure(stack.Name, "recovery-unit capture", err.Error())
m.logger.Printf("[WARN] [backup] Recovery unit capture failed for %s: %v", stack.Name, err)
// R-158: per app, and the loop CONTINUES — one app's failure must not silence the
// others, and it must not abort their captures either. The space figures are read at
// the moment of failure, because the point is to answer "why" (usually: no room).
if m.unitNotify != nil {
m.unitNotify(stack.Name, err, m.unitTargetSpace(stack.Name))
}
}
}
}
func (m *Manager) versionLocked() string {
m.mu.Lock()
defer m.mu.Unlock()
return m.version
}
// strippedAppYaml is the on-disk shape of the app.yaml captured into the unit. The name is historical:
// since D5 the `env` map carries the PORTABLE secrets alongside the plain config (see buildUnitAppYaml).
type strippedAppYaml struct {
Deployed bool `yaml:"deployed"`
Env map[string]string `yaml:"env"`
}
// buildUnitAppYaml renders the unit's app.yaml as bytes: the non-secret env PLUS the portable secret
// values (D5). Deterministic: yaml.v3 sorts map keys and the name lists come in stable metadata order,
// so identical input yields identical bytes (needed for the checksum-skip guard).
//
// This is the ONE place the capture side decides what does and does not reach the drive — there is no
// second path that writes a unit app.yaml. The caller writes the result 0600.
func buildUnitAppYaml(info RecoveryInfo) []byte {
env := make(map[string]string, len(info.NonSecretEnv)+len(info.PortableSecrets))
for k, v := range info.NonSecretEnv {
env[k] = v
}
// Portable secrets last: NonSecretEnv is disjoint from the secret set by construction
// (GetStackRecoveryInfo), so this cannot shadow a plain config value.
for k, v := range info.PortableSecrets {
env[k] = v
}
body, err := yaml.Marshal(strippedAppYaml{Deployed: true, Env: env})
if err != nil {
body = []byte("deployed: true\nenv: {}\n")
}
header := "# Captured by felhom-controller recovery unit.\n" +
"# This file CARRIES SECRETS (D5) so a Tier-1/2 restore needs the drive and nothing else:\n" +
"# data-encrypting keys, database passwords and internal signing secrets. Mode 0600.\n"
if len(info.PortableSecretEnvVars) > 0 {
header += "# Carried: " + strings.Join(info.PortableSecretEnvVars, ", ") + "\n"
}
// The withheld class is named, not valued — an operator reading the unit must be able to see WHY a
// credential is missing rather than suspecting a capture bug.
if withheld := withheldSecretNames(info); len(withheld) > 0 {
header += "# WITHHELD (internet-reachable logins — stay in the guest, regenerated on restore): " +
strings.Join(withheld, ", ") + "\n"
}
return []byte(header + string(body))
}
// withheldSecretNames returns the secret names deliberately NOT carried by the unit, in stable order.
func withheldSecretNames(info RecoveryInfo) []string {
portable := make(map[string]bool, len(info.PortableSecretEnvVars))
for _, n := range info.PortableSecretEnvVars {
portable[n] = true
}
var out []string
for _, n := range info.SecretEnvVars {
if !portable[n] {
out = append(out, n)
}
}
return out
}
// writeManifest writes the manifest JSON atomically.
func writeManifest(dst string, manifest *RecoveryManifest) error {
data, err := json.MarshalIndent(manifest, "", " ")
if err != nil {
return err
}
return atomicWrite(dst, append(data, '\n'), 0644)
}
// readManifest reads an existing recovery-unit manifest (nil if absent or unparseable).
func readManifest(path string) *RecoveryManifest {
data, err := os.ReadFile(path)
if err != nil {
return nil
}
var m RecoveryManifest
if json.Unmarshal(data, &m) != nil {
return nil
}
return &m
}
func sha256Hex(data []byte) string {
sum := sha256.Sum256(data)
return hex.EncodeToString(sum[:])
}
func stringMapEqual(a, b map[string]string) bool {
if len(a) != len(b) {
return false
}
for k, v := range a {
if b[k] != v {
return false
}
}
return true
}
func stringSliceEqual(a, b []string) bool {
if len(a) != len(b) {
return false
}
for i := range a {
if a[i] != b[i] {
return false
}
}
return true
}
// listFileNames returns the names of files with the given suffix in dir (sorted, none if absent).
func listFileNames(dir, suffix string) []string {
entries, err := os.ReadDir(dir)
if err != nil {
return nil
}
var names []string
for _, e := range entries {
if !e.IsDir() && strings.HasSuffix(e.Name(), suffix) {
names = append(names, e.Name())
}
}
sort.Strings(names)
return names
}
// atomicWrite writes data to path via a .tmp file + rename.
func atomicWrite(path string, data []byte, perm os.FileMode) error {
tmp := path + ".tmp"
f, err := os.OpenFile(tmp, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, perm)
if err != nil {
return err
}
if _, err := io.Copy(f, strings.NewReader(string(data))); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Close(); err != nil {
os.Remove(tmp)
return err
}
if err := os.Rename(tmp, path); err != nil {
os.Remove(tmp)
return err
}
return nil
}
// filterOutUndoCopies drops `pre-restore-*` names from a dump listing. R-361: the undo copies are not
// part of the app's recovery set — see the reasoning at the call site.
func filterOutUndoCopies(names []string) []string {
out := make([]string, 0, len(names))
for _, n := range names {
if strings.HasPrefix(n, preRestoreDumpPrefix) {
continue
}
out = append(out, n)
}
return out
}