R-638 option A: a replay meets the copy's own schema on the two side paths (09 §3 decision 154)
Slice 1 — the no-manifest fallback RestoreApp no longer starts the WHOLE stack at the current definition before the replay (a newer app could migrate the restored data underneath it). New order: resolve DB services from the live compose -> stop -> volumes -> DB-only start (StartStackServices, the same helper the unit restore uses, R-47) -> replay -> full start -> health wait. A dump with no identifiable DB service is refused before any mutation (same gate and message as the unit and off-site paths). A failed volume leg skips the replay. restoreDockerVolumes now goes through the existing volumeReplayFrom seam (nil in production) so the order is testable without Docker. Slice 2 — a unit restore whose volume leg failed no longer calls the importer. Everything else on that failure path is unchanged: dataErr is returned as "completed with data errors", the unit's definition is written and the app is fully started. No loader change, no new delete step. Red-proofs in felhom.eu documentation/audits/design-build-2026-10-06/B/ (red-slice1-fallback-order.txt, red-slice2-no-replay-after-volume-failure.txt). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -16,6 +17,10 @@ import (
|
||||
// (AppVolumeDumpPath) and relies on the DB dumps already present on the app's drive.
|
||||
// The stack is stopped before the volume import and restarted after.
|
||||
//
|
||||
// R-638: the order is stop → volumes → DB-only start → replay → full start, the same as the unit
|
||||
// restore (R-47). The replay is skipped when the volume leg failed, and a dump with no identifiable
|
||||
// database service is refused before anything is touched.
|
||||
//
|
||||
// snapshotID is retained for API/UI signature compatibility; with restic removed it
|
||||
// is only used for logging (the source of truth is now the on-disk volume tars).
|
||||
func (m *Manager) RestoreApp(stackName, snapshotID string) error {
|
||||
@@ -48,9 +53,35 @@ func (m *Manager) RestoreApp(stackName, snapshotID string) error {
|
||||
|
||||
m.logger.Printf("[INFO] [backup] Starting app-data restore for %s (drive=%s)", stackName, drivePath)
|
||||
|
||||
// R-638 (option A, 09 §3 decision 154): which compose service holds the database, and is there
|
||||
// anything to replay? Resolved BEFORE the first mutation, from the LIVE compose — this fallback has no
|
||||
// unit, so the live definition is the one that will run — exactly as the off-site path does for an app
|
||||
// that keeps its definition (offbox_reconstitute.go, "WHICH SERVICE HOLDS THE DATABASE").
|
||||
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), stackName)
|
||||
hasDumps := hasReplayableDump(dumpDir)
|
||||
var dbServices []string
|
||||
if hasDumps {
|
||||
if composePath, ok := m.stackProvider.GetStackComposePath(stackName); ok && composePath != "" {
|
||||
svcs, dsErr := DBServiceNames(composePath)
|
||||
if dsErr != nil {
|
||||
// "cannot tell" is not "no database" — leave it empty and let the gate below refuse.
|
||||
m.logger.Printf("[WARN] [backup] %s: could not read the live compose services: %v", stackName, dsErr)
|
||||
}
|
||||
dbServices = svcs
|
||||
}
|
||||
// Fail-closed, the same gate as the unit and off-site paths: the only alternative to refusing is
|
||||
// to start the WHOLE stack and replay into it — the R-638 shape, where a newer app migrates the
|
||||
// restored data before the old copy is poured over it. Refused with the live app untouched.
|
||||
// Pinned by TestR638_FallbackRefusesWhenNoDBServiceIdentifiable.
|
||||
if len(dbServices) == 0 {
|
||||
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: a .sql dump exists but no database service is identifiable in the live compose", stackName)
|
||||
return util.MsgError("err.backup.az_adatbazis_szolgaltatas_nem_azonosithato_a", stackName)
|
||||
}
|
||||
}
|
||||
|
||||
// Stop the app before restore
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 1/3 — stopping app %s", stackName)
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 1/4 — stopping app %s", stackName)
|
||||
}
|
||||
if err := m.stackProvider.StopStack(stackName); err != nil {
|
||||
m.logger.Printf("[WARN] RESTORE could not stop %s: %v (proceeding anyway)", stackName, err)
|
||||
@@ -62,30 +93,56 @@ func (m *Manager) RestoreApp(stackName, snapshotID string) error {
|
||||
|
||||
// Populate Docker volumes from restored tars
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 2/3 — restoring Docker volumes for %s", stackName)
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 2/4 — restoring Docker volumes for %s", stackName)
|
||||
}
|
||||
if _, err := m.restoreDockerVolumes(stackName, drivePath); err != nil {
|
||||
m.logger.Printf("[ERROR] RESTORE volume restore failed for %s: %v", stackName, err)
|
||||
dataErr = err
|
||||
_, volErr := m.restoreDockerVolumes(stackName, drivePath)
|
||||
if volErr != nil {
|
||||
m.logger.Printf("[ERROR] RESTORE volume restore failed for %s: %v", stackName, volErr)
|
||||
dataErr = volErr
|
||||
}
|
||||
|
||||
// Restart the app
|
||||
// F17: replay the captured .sql dump (the legacy path never did this, so DB-resident data did not
|
||||
// come back). Runs after the volume restore so the dump WINS over any tar copy.
|
||||
//
|
||||
// R-638: with ONLY the database service up — the same DB-only start the unit restore uses (R-47).
|
||||
// This used to run after a FULL StartStack at the CURRENT definition, which gave a newer app the
|
||||
// window to migrate the just-restored data; the loader then overlays the copy and removes only what
|
||||
// the copy knows about, so the newer tables stayed behind (measured 2026-09-23, romm 5.0 → 5.3).
|
||||
// Pinned by TestR638_FallbackReplaysBeforeTheAppStarts.
|
||||
//
|
||||
// And NOT when the volume leg failed: the database's volume is then not known to hold the copy's
|
||||
// own files (it may still be the live, newer one), and the overlay would leave the difference
|
||||
// behind. The failure is already in dataErr and is returned below. Pinned by
|
||||
// TestR638_FallbackVolumeFailureSkipsTheReplay.
|
||||
if hasDumps {
|
||||
if volErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] RESTORE %s: NOT replaying the database copy — the volume restore failed, so the database is not known to be the copy's own", stackName)
|
||||
} else {
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 3/4 — starting only %v and replaying the DB copy for %s", dbServices, stackName)
|
||||
}
|
||||
if err := m.stackProvider.StartStackServices(stackName, dbServices); err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] DB-only start for %s: %v", stackName, err)
|
||||
if dataErr == nil {
|
||||
dataErr = err
|
||||
}
|
||||
} else if _, err := m.reimportDBDumpsCtx(stackName, m.namespaceRoot(drivePath)); err != nil {
|
||||
m.logger.Printf("[ERROR] RESTORE DB re-import failed for %s: %v", stackName, err)
|
||||
if dataErr == nil {
|
||||
dataErr = err
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Restart the app — every exit from the DB-only window ends in a full start.
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 3/3 — restarting app %s after restore", stackName)
|
||||
m.logger.Printf("[DEBUG] RestoreApp: step 4/4 — starting app %s after restore", stackName)
|
||||
}
|
||||
if err := m.stackProvider.StartStack(stackName); err != nil {
|
||||
m.logger.Printf("[WARN] RESTORE could not restart %s after restore: %v", stackName, err)
|
||||
}
|
||||
|
||||
// F17: replay the captured .sql dump into the now-running DB (the legacy path never did this, so
|
||||
// DB-resident data did not come back). Runs after volume restore so the dump WINS over any tar copy.
|
||||
if _, err := m.reimportDBDumpsCtx(stackName, m.namespaceRoot(drivePath)); err != nil {
|
||||
m.logger.Printf("[ERROR] RESTORE DB re-import failed for %s: %v", stackName, err)
|
||||
if dataErr == nil {
|
||||
dataErr = err
|
||||
}
|
||||
}
|
||||
|
||||
// Verify app started successfully
|
||||
if err := m.waitForHealthy(stackName, healthTimeout); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] Restore completed but app health check failed: %v", err)
|
||||
@@ -109,7 +166,13 @@ func (m *Manager) RestoreApp(stackName, snapshotID string) error {
|
||||
// re-derive: the caller cannot count volumes afterwards without re-reading the directory the restore
|
||||
// has already consumed.
|
||||
func (m *Manager) restoreDockerVolumes(stackName, drivePath string) (int, error) {
|
||||
return m.restoreDockerVolumesFrom(stackName, AppVolumeDumpPath(m.namespaceRoot(drivePath), stackName))
|
||||
// R-638: through the same volume-replay seam the unit and off-site paths use (R-354), so the
|
||||
// fallback's ordering can be tested without a Docker daemon. Nil in production.
|
||||
replay := m.volumeReplayFrom
|
||||
if replay == nil {
|
||||
replay = m.restoreDockerVolumesFrom
|
||||
}
|
||||
return replay(stackName, AppVolumeDumpPath(m.namespaceRoot(drivePath), stackName))
|
||||
}
|
||||
|
||||
// restoreDockerVolumesFrom is restoreDockerVolumes with an EXPLICIT dump directory, and it returns how
|
||||
|
||||
Reference in New Issue
Block a user