Files
felhom-controller/controller/internal/backup/restore.go
T
admin 9c945688c0 R-638 option A: a replay meets the copy's own schema on the two side paths (09 §3 decision 154)
Slice 1 — the no-manifest fallback RestoreApp no longer starts the WHOLE stack
at the current definition before the replay (a newer app could migrate the
restored data underneath it). New order: resolve DB services from the live
compose -> stop -> volumes -> DB-only start (StartStackServices, the same
helper the unit restore uses, R-47) -> replay -> full start -> health wait.
A dump with no identifiable DB service is refused before any mutation (same
gate and message as the unit and off-site paths). A failed volume leg skips
the replay. restoreDockerVolumes now goes through the existing
volumeReplayFrom seam (nil in production) so the order is testable without
Docker.

Slice 2 — a unit restore whose volume leg failed no longer calls the
importer. Everything else on that failure path is unchanged: dataErr is
returned as "completed with data errors", the unit's definition is written
and the app is fully started.

No loader change, no new delete step. Red-proofs in felhom.eu
documentation/audits/design-build-2026-10-06/B/ (red-slice1-fallback-order.txt,
red-slice2-no-replay-after-volume-failure.txt).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-10-06 15:15:33 +02:00

330 lines
14 KiB
Go

package backup
import (
"context"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"os"
"strings"
"time"
)
// RestoreApp restores an app's data from its on-disk app-data backup.
//
// Disk-tier (restic snapshot) restore has moved to the host agent. This keep-side
// restore re-imports the Docker-volume tar dumps that the app-data backup produced
// (AppVolumeDumpPath) and relies on the DB dumps already present on the app's drive.
// The stack is stopped before the volume import and restarted after.
//
// R-638: the order is stop → volumes → DB-only start → replay → full start, the same as the unit
// restore (R-47). The replay is skipped when the volume leg failed, and a dump with no identifiable
// database service is refused before anything is touched.
//
// snapshotID is retained for API/UI signature compatibility; with restic removed it
// is only used for logging (the source of truth is now the on-disk volume tars).
func (m *Manager) RestoreApp(stackName, snapshotID string) error {
if m.stackProvider == nil {
return fmt.Errorf("stack provider not configured")
}
if m.isDebug() {
m.logger.Printf("[DEBUG] RestoreApp: stack=%s, snapshotID=%s", stackName, snapshotID)
}
// Prevent concurrent operations
m.mu.Lock()
if m.running {
m.mu.Unlock()
return fmt.Errorf("backup or restore already in progress")
}
m.running = true
m.mu.Unlock()
defer func() {
m.mu.Lock()
m.running = false
m.mu.Unlock()
}()
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" {
return fmt.Errorf("cannot determine drive path for %s", stackName)
}
m.logger.Printf("[INFO] [backup] Starting app-data restore for %s (drive=%s)", stackName, drivePath)
// R-638 (option A, 09 §3 decision 154): which compose service holds the database, and is there
// anything to replay? Resolved BEFORE the first mutation, from the LIVE compose — this fallback has no
// unit, so the live definition is the one that will run — exactly as the off-site path does for an app
// that keeps its definition (offbox_reconstitute.go, "WHICH SERVICE HOLDS THE DATABASE").
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), stackName)
hasDumps := hasReplayableDump(dumpDir)
var dbServices []string
if hasDumps {
if composePath, ok := m.stackProvider.GetStackComposePath(stackName); ok && composePath != "" {
svcs, dsErr := DBServiceNames(composePath)
if dsErr != nil {
// "cannot tell" is not "no database" — leave it empty and let the gate below refuse.
m.logger.Printf("[WARN] [backup] %s: could not read the live compose services: %v", stackName, dsErr)
}
dbServices = svcs
}
// Fail-closed, the same gate as the unit and off-site paths: the only alternative to refusing is
// to start the WHOLE stack and replay into it — the R-638 shape, where a newer app migrates the
// restored data before the old copy is poured over it. Refused with the live app untouched.
// Pinned by TestR638_FallbackRefusesWhenNoDBServiceIdentifiable.
if len(dbServices) == 0 {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the live compose", stackName)
return util.MsgError("err.backup.az_adatbazis_szolgaltatas_nem_azonosithato_a", stackName)
}
}
// Stop the app before restore
if m.isDebug() {
m.logger.Printf("[DEBUG] RestoreApp: step 1/4 — stopping app %s", stackName)
}
if err := m.stackProvider.StopStack(stackName); err != nil {
m.logger.Printf("[WARN] RESTORE could not stop %s: %v (proceeding anyway)", stackName, err)
}
// F17: surface a data-restore failure instead of swallowing it. We still bring the app back up so it
// isn't left dead, but the error is returned at the end so a failed restore can't read as success.
var dataErr error
// Populate Docker volumes from restored tars
if m.isDebug() {
m.logger.Printf("[DEBUG] RestoreApp: step 2/4 — restoring Docker volumes for %s", stackName)
}
_, volErr := m.restoreDockerVolumes(stackName, drivePath)
if volErr != nil {
m.logger.Printf("[ERROR] RESTORE volume restore failed for %s: %v", stackName, volErr)
dataErr = volErr
}
// F17: replay the captured .sql dump (the legacy path never did this, so DB-resident data did not
// come back). Runs after the volume restore so the dump WINS over any tar copy.
//
// R-638: with ONLY the database service up — the same DB-only start the unit restore uses (R-47).
// This used to run after a FULL StartStack at the CURRENT definition, which gave a newer app the
// window to migrate the just-restored data; the loader then overlays the copy and removes only what
// the copy knows about, so the newer tables stayed behind (measured 2026-09-23, romm 5.0 → 5.3).
// Pinned by TestR638_FallbackReplaysBeforeTheAppStarts.
//
// And NOT when the volume leg failed: the database's volume is then not known to hold the copy's
// own files (it may still be the live, newer one), and the overlay would leave the difference
// behind. The failure is already in dataErr and is returned below. Pinned by
// TestR638_FallbackVolumeFailureSkipsTheReplay.
if hasDumps {
if volErr != nil {
m.logger.Printf("[ERROR] [backup] RESTORE %s: NOT replaying the database copy — the volume restore failed, so the database is not known to be the copy's own", stackName)
} else {
if m.isDebug() {
m.logger.Printf("[DEBUG] RestoreApp: step 3/4 — starting only %v and replaying the DB copy for %s", dbServices, stackName)
}
if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil {
m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err)
if dataErr == nil {
dataErr = err
}
} else if _, err := m.reimportDBDumpsCtx(stackName, m.namespaceRoot(drivePath)); err != nil {
m.logger.Printf("[ERROR] RESTORE DB re-import failed for %s: %v", stackName, err)
if dataErr == nil {
dataErr = err
}
}
}
}
// Restart the app — every exit from the DB-only window ends in a full start.
if m.isDebug() {
m.logger.Printf("[DEBUG] RestoreApp: step 4/4 — starting app %s after restore", stackName)
}
if err := m.stackProvider.StartStack(stackName); err != nil {
m.logger.Printf("[WARN] RESTORE could not restart %s after restore: %v", stackName, err)
}
// Verify app started successfully
if err := m.waitForHealthy(stackName, healthTimeout); err != nil {
m.logger.Printf("[WARN] [backup] Restore completed but app health check failed: %v", err)
}
if dataErr != nil {
return fmt.Errorf("restore of %s completed with data errors: %w", stackName, dataErr)
}
m.logger.Printf("[INFO] RESTORE completed: stack=%s", stackName)
return nil
}
// restoreDockerVolumes populates Docker volumes from the tars in the app's LIVE recovery unit, and
// returns HOW MANY it replayed.
//
// R-353: the count used to be discarded here. `restoreDockerVolumesFrom` has always returned it, so
// the fact existed one call deep and was thrown away one line later — which left the unit-restore
// path structurally unable to tell a customer whether any data came back. On 2026-08-21 an opengist
// restore reported completion over a unit holding manifest.json and compose/ and nothing else, and no
// screen could have said otherwise. Discarding a fact the caller needs is cheaper to fix than to
// re-derive: the caller cannot count volumes afterwards without re-reading the directory the restore
// has already consumed.
func (m *Manager) restoreDockerVolumes(stackName, drivePath string) (int, error) {
// R-638: through the same volume-replay seam the unit and off-site paths use (R-354), so the
// fallback's ordering can be tested without a Docker daemon. Nil in production.
replay := m.volumeReplayFrom
if replay == nil {
replay = m.restoreDockerVolumesFrom
}
return replay(stackName, AppVolumeDumpPath(m.namespaceRoot(drivePath), stackName))
}
// restoreDockerVolumesFrom is restoreDockerVolumes with an EXPLICIT dump directory, and it returns how
// many volumes it replayed.
//
// R-354. The off-site reconstitution needs exactly this, for the same reason reimportDBDumpsFrom
// exists beside reimportDBDumps: the snapshot's archives live under the restored SCRATCH unit, because
// the live unit is deliberately never overwritten by a placement. Until now no such variant existed,
// so the off-site path had no way to replay a volume and simply did not — the tar sat in the unit, in
// the snapshot and in the verification folder, and the restore reported success without it. For an app
// whose data is entirely in a named volume — 40 of the 53 in the catalogue — that is everything the
// customer owns.
//
// ONE implementation, two callers. A second copy of this loop is what produced the divergence in the
// first place: the local path replayed volumes and the off-site path did not, and nothing compared the
// two.
//
// It only ever READS dumpDir; the recovery unit is never written to here, on either path.
func (m *Manager) restoreDockerVolumesFrom(stackName, dumpDir string) (int, error) {
entries, err := os.ReadDir(dumpDir)
if err != nil {
if os.IsNotExist(err) {
return 0, nil // No volume dumps to restore
}
return 0, fmt.Errorf("reading volume dump dir: %w", err)
}
var restored int
var failed []string
composeVer := ""
for _, entry := range entries {
if entry.IsDir() || !strings.HasSuffix(entry.Name(), ".tar") {
continue
}
volName := strings.TrimSuffix(entry.Name(), ".tar")
if composeVer == "" {
composeVer = composeVersionShort()
}
m.logger.Printf("[INFO] [backup] Restoring Docker volume %s for %s", volName, stackName)
// Remove existing volume (ignore errors — may not exist)
dockerexec.Command("docker", "volume", "rm", "-f", volName).Run()
// Create fresh volume — WITH the labels compose gives its own volumes (R-658, v0.268.0). A bare
// `volume create` made a volume the update's undo (until v0.267.0) could not find, and one
// compose warns it "was not created by Docker Compose". The config-hash label is deliberately
// NOT set: compose only compares it when present, and a guessed hash would make it offer to
// recreate (empty) the volume.
args := append([]string{"volume", "create"}, composeVolumeLabelArgs(stackName, volName, composeVer)...)
if len(args) == 2 {
m.logger.Printf("[WARN] [backup] volume %s does not carry the %s_ prefix — created WITHOUT compose labels (the undo finds it by the app's definition anyway)", volName, stackName)
}
args = append(args, volName)
if out, err := dockerexec.Command("docker", args...).CombinedOutput(); err != nil {
m.logger.Printf("[ERROR] [backup] Failed to create volume %s: %s — %v", volName, strings.TrimSpace(string(out)), err)
failed = append(failed, volName)
continue
}
// Populate from tar
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
cmd := dockerexec.CommandContext(ctx, "docker", "run", "--rm",
"-v", volName+":/vol",
"-v", dumpDir+":/in:ro",
"alpine", "tar", "xf", "/in/"+entry.Name(), "-C", "/vol")
out, err := cmd.CombinedOutput()
cancel()
if err != nil {
m.logger.Printf("[ERROR] [backup] Failed to populate volume %s: %s — %v", volName, strings.TrimSpace(string(out)), err)
failed = append(failed, volName)
continue
}
restored++
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] Volume %s restored successfully", volName)
}
}
if restored > 0 {
m.logger.Printf("[INFO] [backup] Restored %d Docker volume(s) for %s", restored, stackName)
}
// F17: a per-volume failure used to be a swallowed WARN; surface it so the restore is reported as
// failed rather than silently partial. The count is returned ALONGSIDE the error, not instead of
// it: a caller that replayed three of four volumes needs both numbers to say what happened.
if len(failed) > 0 {
return restored, fmt.Errorf("failed to restore %d volume(s): %v", len(failed), failed)
}
return restored, nil
}
// waitForHealthy's clock (R-488). These are the PRODUCTION values; they are variables only so the
// package's tests can shorten them in ONE place (main_test.go TestMain) instead of every restore
// test sleeping 3 s of settle and every not-running fixture waiting out the full 90 s — that was
// ~280 s of the package's ~440 s. TestR488_HealthWaitDefaultsAreProduction pins the values.
var (
healthSettle = 3 * time.Second // initial settling time before the first poll
healthInterval = 5 * time.Second // between polls
healthTimeout = 90 * time.Second // how long a restored stack gets to reach running
)
// waitForHealthy waits for a stack to reach running state after restore.
// Forces a docker ps refresh on each poll to avoid stale state.
func (m *Manager) waitForHealthy(stackName string, timeout time.Duration) error {
deadline := time.Now().Add(timeout)
interval := healthInterval
time.Sleep(healthSettle) // initial settling time
for time.Now().Before(deadline) {
if m.stackProvider == nil {
return fmt.Errorf("no stack provider")
}
if m.stackProvider.RefreshAndIsRunning(stackName) {
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] Post-restore health check: %s is running", stackName)
}
return nil
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] Post-restore health check: %s not yet running, waiting...", stackName)
}
time.Sleep(interval)
}
return fmt.Errorf("stack %s did not reach running state within %s after restore", stackName, timeout)
}
// composeVolumeLabelArgs returns the `--label` arguments that make a restored volume look like one
// compose created for this project: project, volume key and compose version. The key is the part of
// the Docker name after `<project>_`, which is how compose names a volume without its own `name:` —
// every catalog template today (R-658, measured 2026-09-24). A name without that prefix gets no labels
// rather than a guessed key. The version label is left out when the version could not be read.
func composeVolumeLabelArgs(project, volName, composeVersion string) []string {
key := strings.TrimPrefix(volName, project+"_")
if key == volName || key == "" {
return nil
}
args := []string{"--label", "com.docker.compose.project=" + project, "--label", "com.docker.compose.volume=" + key}
if composeVersion != "" {
args = append(args, "--label", "com.docker.compose.version="+composeVersion)
}
return args
}
// composeVersionShort is `docker compose version --short`, "" when it cannot be read.
func composeVersionShort() string {
out, err := dockerexec.Command("docker", "compose", "version", "--short").Output()
if err != nil {
return ""
}
return strings.TrimPrefix(strings.TrimSpace(string(out)), "v")
}