80e6ad8c47
gates / gates (push) Successful in 26s
R-650: internal/dockerexec — every docker exec routed through it; under go test a real docker is refused (opt-in FELHOM_TEST_REAL_DOCKER=1; a stub under the temp dir is allowed). api/stacks/web tests run under a silent stub (TestMain). TestR650_NoBareDockerExec pins it repo-wide. R-640: a dump without its engine's completion marker is refused before the first mutation (unit + off-site restore) and again before any load. R-499: the Tier-2 page's system-disk sentence has four true branches. R-518: the backup button states the measured ~8 min stop. R-626: measured on 9202, not reproduced. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
218 lines
8.4 KiB
Go
218 lines
8.4 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
|
"os"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// RestoreApp restores an app's data from its on-disk app-data backup.
|
|
//
|
|
// Disk-tier (restic snapshot) restore has moved to the host agent. This keep-side
|
|
// restore re-imports the Docker-volume tar dumps that the app-data backup produced
|
|
// (AppVolumeDumpPath) and relies on the DB dumps already present on the app's drive.
|
|
// The stack is stopped before the volume import and restarted after.
|
|
//
|
|
// snapshotID is retained for API/UI signature compatibility; with restic removed it
|
|
// is only used for logging (the source of truth is now the on-disk volume tars).
|
|
func (m *Manager) RestoreApp(stackName, snapshotID string) error {
|
|
if m.stackProvider == nil {
|
|
return fmt.Errorf("stack provider not configured")
|
|
}
|
|
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] RestoreApp: stack=%s, snapshotID=%s", stackName, snapshotID)
|
|
}
|
|
|
|
// Prevent concurrent operations
|
|
m.mu.Lock()
|
|
if m.running {
|
|
m.mu.Unlock()
|
|
return fmt.Errorf("backup or restore already in progress")
|
|
}
|
|
m.running = true
|
|
m.mu.Unlock()
|
|
defer func() {
|
|
m.mu.Lock()
|
|
m.running = false
|
|
m.mu.Unlock()
|
|
}()
|
|
|
|
drivePath := m.GetAppDrivePath(stackName)
|
|
if drivePath == "" {
|
|
return fmt.Errorf("cannot determine drive path for %s", stackName)
|
|
}
|
|
|
|
m.logger.Printf("[INFO] [backup] Starting app-data restore for %s (drive=%s)", stackName, drivePath)
|
|
|
|
// Stop the app before restore
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] RestoreApp: step 1/3 — stopping app %s", stackName)
|
|
}
|
|
if err := m.stackProvider.StopStack(stackName); err != nil {
|
|
m.logger.Printf("[WARN] RESTORE could not stop %s: %v (proceeding anyway)", stackName, err)
|
|
}
|
|
|
|
// F17: surface a data-restore failure instead of swallowing it. We still bring the app back up so it
|
|
// isn't left dead, but the error is returned at the end so a failed restore can't read as success.
|
|
var dataErr error
|
|
|
|
// Populate Docker volumes from restored tars
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] RestoreApp: step 2/3 — restoring Docker volumes for %s", stackName)
|
|
}
|
|
if _, err := m.restoreDockerVolumes(stackName, drivePath); err != nil {
|
|
m.logger.Printf("[ERROR] RESTORE volume restore failed for %s: %v", stackName, err)
|
|
dataErr = err
|
|
}
|
|
|
|
// Restart the app
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] RestoreApp: step 3/3 — restarting app %s after restore", stackName)
|
|
}
|
|
if err := m.stackProvider.StartStack(stackName); err != nil {
|
|
m.logger.Printf("[WARN] RESTORE could not restart %s after restore: %v", stackName, err)
|
|
}
|
|
|
|
// F17: replay the captured .sql dump into the now-running DB (the legacy path never did this, so
|
|
// DB-resident data did not come back). Runs after volume restore so the dump WINS over any tar copy.
|
|
if _, err := m.reimportDBDumpsCtx(stackName, m.namespaceRoot(drivePath)); err != nil {
|
|
m.logger.Printf("[ERROR] RESTORE DB re-import failed for %s: %v", stackName, err)
|
|
if dataErr == nil {
|
|
dataErr = err
|
|
}
|
|
}
|
|
|
|
// Verify app started successfully
|
|
if err := m.waitForHealthy(stackName, 90*time.Second); err != nil {
|
|
m.logger.Printf("[WARN] [backup] Restore completed but app health check failed: %v", err)
|
|
}
|
|
|
|
if dataErr != nil {
|
|
return fmt.Errorf("restore of %s completed with data errors: %w", stackName, dataErr)
|
|
}
|
|
m.logger.Printf("[INFO] RESTORE completed: stack=%s", stackName)
|
|
return nil
|
|
}
|
|
|
|
// restoreDockerVolumes populates Docker volumes from the tars in the app's LIVE recovery unit, and
|
|
// returns HOW MANY it replayed.
|
|
//
|
|
// R-353: the count used to be discarded here. `restoreDockerVolumesFrom` has always returned it, so
|
|
// the fact existed one call deep and was thrown away one line later — which left the unit-restore
|
|
// path structurally unable to tell a customer whether any data came back. On 2026-08-21 an opengist
|
|
// restore reported completion over a unit holding manifest.json and compose/ and nothing else, and no
|
|
// screen could have said otherwise. Discarding a fact the caller needs is cheaper to fix than to
|
|
// re-derive: the caller cannot count volumes afterwards without re-reading the directory the restore
|
|
// has already consumed.
|
|
func (m *Manager) restoreDockerVolumes(stackName, drivePath string) (int, error) {
|
|
return m.restoreDockerVolumesFrom(stackName, AppVolumeDumpPath(m.namespaceRoot(drivePath), stackName))
|
|
}
|
|
|
|
// restoreDockerVolumesFrom is restoreDockerVolumes with an EXPLICIT dump directory, and it returns how
|
|
// many volumes it replayed.
|
|
//
|
|
// R-354. The off-site reconstitution needs exactly this, for the same reason reimportDBDumpsFrom
|
|
// exists beside reimportDBDumps: the snapshot's archives live under the restored SCRATCH unit, because
|
|
// the live unit is deliberately never overwritten by a placement. Until now no such variant existed,
|
|
// so the off-site path had no way to replay a volume and simply did not — the tar sat in the unit, in
|
|
// the snapshot and in the verification folder, and the restore reported success without it. For an app
|
|
// whose data is entirely in a named volume — 40 of the 53 in the catalogue — that is everything the
|
|
// customer owns.
|
|
//
|
|
// ONE implementation, two callers. A second copy of this loop is what produced the divergence in the
|
|
// first place: the local path replayed volumes and the off-site path did not, and nothing compared the
|
|
// two.
|
|
//
|
|
// It only ever READS dumpDir; the recovery unit is never written to here, on either path.
|
|
func (m *Manager) restoreDockerVolumesFrom(stackName, dumpDir string) (int, error) {
|
|
entries, err := os.ReadDir(dumpDir)
|
|
if err != nil {
|
|
if os.IsNotExist(err) {
|
|
return 0, nil // No volume dumps to restore
|
|
}
|
|
return 0, fmt.Errorf("reading volume dump dir: %w", err)
|
|
}
|
|
|
|
var restored int
|
|
var failed []string
|
|
for _, entry := range entries {
|
|
if entry.IsDir() || !strings.HasSuffix(entry.Name(), ".tar") {
|
|
continue
|
|
}
|
|
volName := strings.TrimSuffix(entry.Name(), ".tar")
|
|
|
|
m.logger.Printf("[INFO] [backup] Restoring Docker volume %s for %s", volName, stackName)
|
|
|
|
// Remove existing volume (ignore errors — may not exist)
|
|
dockerexec.Command("docker", "volume", "rm", "-f", volName).Run()
|
|
|
|
// Create fresh volume
|
|
if out, err := dockerexec.Command("docker", "volume", "create", volName).CombinedOutput(); err != nil {
|
|
m.logger.Printf("[ERROR] [backup] Failed to create volume %s: %s — %v", volName, strings.TrimSpace(string(out)), err)
|
|
failed = append(failed, volName)
|
|
continue
|
|
}
|
|
|
|
// Populate from tar
|
|
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute)
|
|
cmd := dockerexec.CommandContext(ctx, "docker", "run", "--rm",
|
|
"-v", volName+":/vol",
|
|
"-v", dumpDir+":/in:ro",
|
|
"alpine", "tar", "xf", "/in/"+entry.Name(), "-C", "/vol")
|
|
out, err := cmd.CombinedOutput()
|
|
cancel()
|
|
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [backup] Failed to populate volume %s: %s — %v", volName, strings.TrimSpace(string(out)), err)
|
|
failed = append(failed, volName)
|
|
continue
|
|
}
|
|
|
|
restored++
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [backup] Volume %s restored successfully", volName)
|
|
}
|
|
}
|
|
|
|
if restored > 0 {
|
|
m.logger.Printf("[INFO] [backup] Restored %d Docker volume(s) for %s", restored, stackName)
|
|
}
|
|
// F17: a per-volume failure used to be a swallowed WARN; surface it so the restore is reported as
|
|
// failed rather than silently partial. The count is returned ALONGSIDE the error, not instead of
|
|
// it: a caller that replayed three of four volumes needs both numbers to say what happened.
|
|
if len(failed) > 0 {
|
|
return restored, fmt.Errorf("failed to restore %d volume(s): %v", len(failed), failed)
|
|
}
|
|
return restored, nil
|
|
}
|
|
|
|
// waitForHealthy waits for a stack to reach running state after restore.
|
|
// Forces a docker ps refresh on each poll to avoid stale state.
|
|
func (m *Manager) waitForHealthy(stackName string, timeout time.Duration) error {
|
|
deadline := time.Now().Add(timeout)
|
|
interval := 5 * time.Second
|
|
|
|
time.Sleep(3 * time.Second) // initial settling time
|
|
|
|
for time.Now().Before(deadline) {
|
|
if m.stackProvider == nil {
|
|
return fmt.Errorf("no stack provider")
|
|
}
|
|
if m.stackProvider.RefreshAndIsRunning(stackName) {
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [backup] Post-restore health check: %s is running", stackName)
|
|
}
|
|
return nil
|
|
}
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [backup] Post-restore health check: %s not yet running, waiting...", stackName)
|
|
}
|
|
time.Sleep(interval)
|
|
}
|
|
return fmt.Errorf("stack %s did not reach running state within %s after restore", stackName, timeout)
|
|
}
|