v0.275.0: a backup's data and its version travel together (R-696, 07 §6.6, D4 option A); R-695, R-691, R-694
gates / gates (push) Successful in 23s
gates / gates (push) Successful in 23s
The unit's data files are stamped with the versions that wrote them; the capture keeps the definition the data belongs to; a restore never starts data under another version's definition (unit restores refuse a mismatch; the off-site restore writes the snapshot's definition); every tier's time is its data's; the conversion-copy release needs a dump on the new engine. File-browser sync single-flight + no empty kept folder (R-695); the kept view joins the folder's owning group, language switch resyncs (R-691); a restore-generated login is not shown as the password (R-694). Red-proofs in felhom.eu/documentation/audits/version-travel-2026-09-26/. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,576 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// Part A of the version-travel brief (controller v0.275.0, R-696, `07` §6.6): a backup's data and its
|
||||
// version travel together. Measured before the fix on 9202 (`audits/version-travel-2026-09-26/A1/`): the
|
||||
// periodic refresh re-captured the unit's DEFINITION two minutes after an update, over the previous
|
||||
// version's DATA; Tier 1's time moved to the refresh; a restore in that window started a PostgreSQL 16
|
||||
// datadir under the 18 definition and left the app down.
|
||||
//
|
||||
// Every test asserts a CONSEQUENCE a household or the update would see — which definition the restore
|
||||
// starts, which time the precondition reads, which copy the release trusts — not the mechanism.
|
||||
|
||||
// vtStack is one app's stack dir + a provider whose ImagePins are what production derives them from:
|
||||
// ParseComposeImages of the stack's compose (the adapter's GetStackRecoveryInfo does exactly that).
|
||||
type vtStack struct {
|
||||
t *testing.T
|
||||
tmp string
|
||||
drive string
|
||||
stackDir string
|
||||
fake *fakeRecoveryProvider
|
||||
m *Manager
|
||||
}
|
||||
|
||||
func vtCompose(pgMajor string) string {
|
||||
return "services:\n app:\n image: example/app:0.96.0\n app-db:\n image: postgres:" + pgMajor + "-alpine\n"
|
||||
}
|
||||
|
||||
func newVTStack(t *testing.T, pgMajor string) *vtStack {
|
||||
t.Helper()
|
||||
tmp := t.TempDir()
|
||||
v := &vtStack{t: t, tmp: tmp, drive: filepath.Join(tmp, "drive"), stackDir: filepath.Join(tmp, "stack")}
|
||||
if err := os.MkdirAll(v.stackDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
v.fake = &fakeRecoveryProvider{hdd: v.drive, running: true}
|
||||
v.m = &Manager{
|
||||
logger: log.New(io.Discard, "", 0),
|
||||
systemDataPath: filepath.Join(tmp, "system"),
|
||||
stackProvider: v.fake,
|
||||
version: "vtest",
|
||||
}
|
||||
v.setVersion(pgMajor)
|
||||
return v
|
||||
}
|
||||
|
||||
// setVersion is what an update does to the stack dir: the definition moves, and what runs moves with it.
|
||||
func (v *vtStack) setVersion(pgMajor string) {
|
||||
v.t.Helper()
|
||||
mustWrite(v.t, filepath.Join(v.stackDir, "docker-compose.yml"), vtCompose(pgMajor))
|
||||
mustWrite(v.t, filepath.Join(v.stackDir, ".felhom.yml"), "display_name: App "+pgMajor+"\n")
|
||||
mustWrite(v.t, filepath.Join(v.stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: vt\n")
|
||||
v.fake.info = RecoveryInfo{
|
||||
StackDir: v.stackDir,
|
||||
DisplayName: "App",
|
||||
ImagePins: ParseComposeImages(filepath.Join(v.stackDir, "docker-compose.yml")),
|
||||
NonSecretEnv: map[string]string{"SUBDOMAIN": "vt"},
|
||||
InstalledImages: map[string]string{
|
||||
"app": "example/app:0.96.0@sha256:aaa",
|
||||
"app-db": "postgres:" + pgMajor + "-alpine@sha256:" + pgMajor + pgMajor,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func (v *vtStack) unitDir() string { return RecoveryUnitPath(v.drive, "app") }
|
||||
|
||||
// backupLegs is what a data run writes, through the SAME stamp call the legs make: a database dump and
|
||||
// a volume tar, both dated `at` (so "the refresh ran later" is a fact of the clock, not of the test).
|
||||
func (v *vtStack) backupLegs(at time.Time, marker string) {
|
||||
v.t.Helper()
|
||||
sql := filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql")
|
||||
tar := filepath.Join(UnitVolumeDumpDir(v.unitDir()), "app_db.tar")
|
||||
mustWrite(v.t, sql, pgDump(1)+"-- "+marker+"\n")
|
||||
mustWrite(v.t, tar, "tar:"+marker)
|
||||
for _, p := range []string{sql, tar} {
|
||||
if err := os.Chtimes(p, at, at); err != nil {
|
||||
v.t.Fatal(err)
|
||||
}
|
||||
}
|
||||
v.m.stampDataFile("app", v.unitDir(), "db-dumps/app-postgres.sql")
|
||||
v.m.stampDataFile("app", v.unitDir(), "volume-dumps/app_db.tar")
|
||||
}
|
||||
|
||||
func (v *vtStack) manifest() *RecoveryManifest {
|
||||
v.t.Helper()
|
||||
man := readManifest(UnitManifestFile(v.unitDir()))
|
||||
if man == nil {
|
||||
v.t.Fatal("no readable manifest")
|
||||
}
|
||||
return man
|
||||
}
|
||||
|
||||
func (v *vtStack) unitComposeImages() []string {
|
||||
return ParseComposeImages(filepath.Join(UnitComposeDir(v.unitDir()), "docker-compose.yml"))
|
||||
}
|
||||
|
||||
// theWindow builds the measured A1 state: a data run at PostgreSQL 16 two hours ago, then the update to
|
||||
// 18, then the periodic refresh NOW.
|
||||
func theWindow(t *testing.T) (*vtStack, time.Time) {
|
||||
t.Helper()
|
||||
v := newVTStack(t, "16")
|
||||
dataAt := time.Now().Add(-2 * time.Hour).UTC().Truncate(time.Second)
|
||||
v.backupLegs(dataAt, "written-by-16")
|
||||
if err := v.m.CaptureRecoveryUnit("app"); err != nil { // the data run's capture
|
||||
t.Fatal(err)
|
||||
}
|
||||
v.setVersion("18") // the guarded update moved the pin
|
||||
if err := v.m.captureRecoveryUnit("app", false); err != nil { // the 5-minute refresh
|
||||
t.Fatal(err)
|
||||
}
|
||||
return v, dataAt
|
||||
}
|
||||
|
||||
// A5 red-proof 1 — the manifest refresh after a pin change no longer moves Tier 1's time.
|
||||
// Pre-fix (v0.274.0): ListRestorePoints = newest of the manifest's and the dumps' mtimes → "now".
|
||||
func TestA5_RefreshAfterAPinChangeDoesNotMoveTier1sTime(t *testing.T) {
|
||||
v, dataAt := theWindow(t)
|
||||
pts, found := v.m.ListRestorePoints("app")
|
||||
if !found || len(pts) != 1 {
|
||||
t.Fatalf("restore points = %v found=%v, want one", pts, found)
|
||||
}
|
||||
got, err := time.Parse(time.RFC3339, pts[0].Time)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !got.Equal(dataAt) {
|
||||
t.Fatalf("Tier 1's time = %s, want the DATA's time %s — the refresh %s made a two-hour-old dump read as new",
|
||||
got.Format(time.RFC3339), dataAt.Format(time.RFC3339), time.Since(got).Round(time.Second))
|
||||
}
|
||||
}
|
||||
|
||||
// The definition stays with its data: after the refresh, compose/ still names the 16 definition, the
|
||||
// manifest says both (image_pins = what runs, data.image_pins = what wrote the data).
|
||||
// Pre-fix: the refresh rewrote compose/ to 18 (A1 S3b).
|
||||
func TestA2_TheUnitKeepsTheDefinitionItsDataBelongsTo(t *testing.T) {
|
||||
v, _ := theWindow(t)
|
||||
if got := v.unitComposeImages(); !samePins(got, ParseComposeImages(writeTmpCompose(t, vtCompose("16")))) {
|
||||
t.Fatalf("the unit's compose/ names %v — the refresh paired the 18 definition with the 16 data", got)
|
||||
}
|
||||
man := v.manifest()
|
||||
if man.Data == nil || !strings.Contains(strings.Join(man.Data.ImagePins, " "), "postgres:16-alpine") {
|
||||
t.Fatalf("manifest data = %+v, want the 16 pins", man.Data)
|
||||
}
|
||||
if !strings.Contains(strings.Join(man.ImagePins, " "), "postgres:18-alpine") {
|
||||
t.Fatalf("manifest image_pins = %v, want the app's CURRENT (18) pins", man.ImagePins)
|
||||
}
|
||||
if got := man.Data.Files["db-dumps/app-postgres.sql"].Images["app-db"]; !strings.HasPrefix(got, "postgres:16-alpine@sha256:") {
|
||||
t.Fatalf("the dump's recorded engine = %q, want postgres:16 with its digest", got)
|
||||
}
|
||||
// And a SECOND refresh writes nothing: the frozen checksums describe what compose/ holds.
|
||||
before, _ := os.Stat(UnitManifestFile(v.unitDir()))
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
if err := v.m.captureRecoveryUnit("app", false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
after, _ := os.Stat(UnitManifestFile(v.unitDir()))
|
||||
if !after.ModTime().Equal(before.ModTime()) {
|
||||
t.Fatal("a second refresh rewrote the manifest — the frozen unit thrashes the drive every five minutes")
|
||||
}
|
||||
}
|
||||
|
||||
func writeTmpCompose(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "docker-compose.yml")
|
||||
mustWrite(t, p, body)
|
||||
return p
|
||||
}
|
||||
|
||||
// The next data run replaces the data AND the definition: both are 18 afterwards.
|
||||
func TestA2_TheNextDataRunMovesDataAndDefinitionTogether(t *testing.T) {
|
||||
v, _ := theWindow(t)
|
||||
now := time.Now().UTC().Truncate(time.Second)
|
||||
v.backupLegs(now, "written-by-18")
|
||||
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := strings.Join(v.unitComposeImages(), " "); !strings.Contains(got, "postgres:18-alpine") {
|
||||
t.Fatalf("after a data run on 18 the unit's compose/ names %s, want 18", got)
|
||||
}
|
||||
if man := v.manifest(); man.Data == nil || man.Data.At != now.Format(time.RFC3339) || man.Data.Mixed {
|
||||
t.Fatalf("data = %+v, want at %s, not mixed", man.Data, now.Format(time.RFC3339))
|
||||
}
|
||||
}
|
||||
|
||||
// vtRestorer records the definition the restore STARTS the data with.
|
||||
type vtRestorer struct {
|
||||
*fakeRecoveryProvider
|
||||
startedWith []string
|
||||
}
|
||||
|
||||
func (f *vtRestorer) RecreateStackDefinitionFromUnit(name, composeDir string, env map[string]string) error {
|
||||
f.startedWith = ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
|
||||
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, env)
|
||||
}
|
||||
|
||||
func vtRestoreSeams(m *Manager) (vols *[]string) {
|
||||
var vd []string
|
||||
m.volumeReplayFrom = func(_, dir string) (int, error) { vd = append(vd, dir); return 1, nil }
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
||||
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
||||
}
|
||||
m.importDBDump = func(context.Context, DiscoveredDB, string) error { return nil }
|
||||
return &vd
|
||||
}
|
||||
|
||||
// A5 red-proof 3 — a restore in the window starts the OLD version with the OLD data, and says so.
|
||||
// Pre-fix: the unit's definition was 18 (the refresh), so the 16 datadir was started under 18 (A1 S4).
|
||||
func TestA5_ARestoreInTheWindowStartsTheOldVersionWithTheOldData(t *testing.T) {
|
||||
v, dataAt := theWindow(t)
|
||||
rec := &vtRestorer{fakeRecoveryProvider: v.fake}
|
||||
v.m.stackProvider = rec
|
||||
vtRestoreSeams(v.m)
|
||||
|
||||
res, err := v.m.RestoreFromRecoveryUnit("app")
|
||||
if err != nil {
|
||||
t.Fatalf("restore: %v", err)
|
||||
}
|
||||
if got := strings.Join(rec.startedWith, " "); !strings.Contains(got, "postgres:16-alpine") || strings.Contains(got, "postgres:18") {
|
||||
t.Fatalf("the restore started the 16 data with the definition %s — a restore must never mix versions", got)
|
||||
}
|
||||
if !res.VersionChanged || !samePins(res.DataPins, ParseComposeImages(writeTmpCompose(t, vtCompose("16")))) || !res.DataAt.Equal(dataAt) {
|
||||
t.Fatalf("result = changed %v pins %v at %s, want changed, the 16 pins, %s", res.VersionChanged, res.DataPins, res.DataAt, dataAt)
|
||||
}
|
||||
}
|
||||
|
||||
// A restore never starts data with a definition it does not belong to: a unit whose compose/ names other
|
||||
// pins than its data, and a unit whose files were written by different versions, are refused BEFORE
|
||||
// anything is touched (no stop, no volume, no recreate).
|
||||
func TestA3_AMismatchedOrMixedUnitIsRefusedBeforeAnythingMoves(t *testing.T) {
|
||||
t.Run("definition-not-the-data's", func(t *testing.T) {
|
||||
v, _ := theWindow(t)
|
||||
// What v0.274.0's refresh left behind: the NEW definition in compose/ over the old data.
|
||||
mustWrite(t, filepath.Join(UnitComposeDir(v.unitDir()), "docker-compose.yml"), vtCompose("18"))
|
||||
vols := vtRestoreSeams(v.m)
|
||||
_, err := v.m.RestoreFromRecoveryUnit("app")
|
||||
if !errors.Is(err, ErrUnitVersionMismatch) {
|
||||
t.Fatalf("err = %v, want ErrUnitVersionMismatch", err)
|
||||
}
|
||||
if len(v.fake.calls) != 0 || len(*vols) != 0 {
|
||||
t.Fatalf("the refusal touched the app: calls=%v volumes=%v", v.fake.calls, *vols)
|
||||
}
|
||||
})
|
||||
t.Run("mixed", func(t *testing.T) {
|
||||
v, _ := theWindow(t)
|
||||
// The DB leg ran under 18, the volume leg failed and kept its 16 tar.
|
||||
sql := filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql")
|
||||
mustWrite(t, sql, pgDump(1))
|
||||
v.m.stampDataFile("app", v.unitDir(), "db-dumps/app-postgres.sql")
|
||||
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if man := v.manifest(); man.Data == nil || !man.Data.Mixed {
|
||||
t.Fatalf("data = %+v, want Mixed", man.Data)
|
||||
}
|
||||
vols := vtRestoreSeams(v.m)
|
||||
_, err := v.m.RestoreFromRecoveryUnit("app")
|
||||
if !errors.Is(err, ErrUnitVersionMismatch) {
|
||||
t.Fatalf("err = %v, want ErrUnitVersionMismatch", err)
|
||||
}
|
||||
if len(v.fake.calls) != 0 || len(*vols) != 0 {
|
||||
t.Fatalf("the refusal touched the app: calls=%v volumes=%v", v.fake.calls, *vols)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// An older unit (no stamps) restores as before: its definition, VersionsUnknown, no refusal.
|
||||
func TestA3_AnUnstampedUnitRestoresAsBefore(t *testing.T) {
|
||||
v := newVTStack(t, "16")
|
||||
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), pgDump(1))
|
||||
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if man := v.manifest(); man.Data != nil {
|
||||
t.Fatalf("an unstamped dump produced data %+v — unknown must stay unknown", man.Data)
|
||||
}
|
||||
rec := &vtRestorer{fakeRecoveryProvider: v.fake}
|
||||
v.m.stackProvider = rec
|
||||
vtRestoreSeams(v.m)
|
||||
res, err := v.m.RestoreFromRecoveryUnit("app")
|
||||
if err != nil {
|
||||
t.Fatalf("restore: %v", err)
|
||||
}
|
||||
if !res.VersionsUnknown || res.VersionChanged || len(rec.startedWith) == 0 {
|
||||
t.Fatalf("result = %+v started=%v, want VersionsUnknown and the unit's definition started", res, rec.startedWith)
|
||||
}
|
||||
}
|
||||
|
||||
// A5 red-proof 4 — the update's precondition refuses a stale dump that a refresh made look new.
|
||||
// Pre-fix: Tier 1 = the refresh's manifest time → "0m old" → accepted, and the update leaned on a copy
|
||||
// of the previous version's data from two hours before.
|
||||
func TestA5_ThePreconditionRefusesAStaleDumpARefreshMadeLookNew(t *testing.T) {
|
||||
v, dataAt := theWindow(t)
|
||||
now := time.Now()
|
||||
fresh := func(p UpdateTierPoint) bool { return now.Sub(p.At) <= time.Hour }
|
||||
p, ok, seen := v.m.UpdateRestorePoints(context.Background(), "app", fresh)
|
||||
if ok {
|
||||
t.Fatalf("the precondition ACCEPTED tier %d at %s (%s old) — the data is from %s",
|
||||
p.Tier, p.At.Format(time.RFC3339), now.Sub(p.At).Round(time.Second), dataAt.Format(time.RFC3339))
|
||||
}
|
||||
if len(seen) != 1 || seen[0].Tier != UpdateTierLocal || !seen[0].At.Equal(dataAt) {
|
||||
t.Fatalf("seen = %+v, want Tier 1 at the data's time %s", seen, dataAt.Format(time.RFC3339))
|
||||
}
|
||||
}
|
||||
|
||||
// A unit with no data file at all is dated by the data run that confirmed it, and a refresh keeps it.
|
||||
func TestA4_AUnitWithoutDataFilesIsDatedByItsDataRun(t *testing.T) {
|
||||
v := newVTStack(t, "16")
|
||||
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
first := v.manifest().Data
|
||||
if first == nil || first.At == "" {
|
||||
t.Fatalf("data = %+v, want the data run's time", first)
|
||||
}
|
||||
v.setVersion("18")
|
||||
if err := v.m.captureRecoveryUnit("app", false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := v.manifest().Data; got == nil || got.At != first.At {
|
||||
t.Fatalf("a refresh moved the data time: %+v → %+v", first, got)
|
||||
}
|
||||
}
|
||||
|
||||
// The undo copies are not data: an update's safety dump written after the backup must not date Tier 1.
|
||||
func TestA4_AnUndoCopyNeverDatesTheUnit(t *testing.T) {
|
||||
v := newVTStack(t, "16")
|
||||
old := time.Now().Add(-3 * time.Hour).UTC().Truncate(time.Second)
|
||||
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), pgDump(1))
|
||||
_ = os.Chtimes(filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), old, old)
|
||||
if err := v.m.captureRecoveryUnit("app", false); err != nil { // unstamped: the legacy rule
|
||||
t.Fatal(err)
|
||||
}
|
||||
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), preRestoreDumpPrefix+"20260926T000000Z-app-postgres.sql"), pgDump(1))
|
||||
pts, _ := v.m.ListRestorePoints("app")
|
||||
if len(pts) != 1 || pts[0].Time != old.Format(time.RFC3339) {
|
||||
t.Fatalf("Tier 1 = %+v, want the dump's %s (not the undo copy's, not the manifest's)", pts, old.Format(time.RFC3339))
|
||||
}
|
||||
}
|
||||
|
||||
// The stamps are written by the PRODUCTION leg (seam discipline): RunAppBackupNow → the dump seam → the
|
||||
// stamp → the capture's `data`. A stamp helper nobody calls is the "seam built but never wired" shape.
|
||||
func TestA2_TheUpdatesOwnBackupStampsItsDataThroughTheRealLeg(t *testing.T) {
|
||||
v := newVTStack(t, "16")
|
||||
v.m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
||||
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
|
||||
}
|
||||
v.m.dumpOne = func(_ context.Context, db DiscoveredDB, dir string, _ *log.Logger, _ bool) DumpResult {
|
||||
p := filepath.Join(dir, "app-postgres.sql")
|
||||
mustWrite(t, p, pgDump(1))
|
||||
return DumpResult{DB: db, FilePath: p}
|
||||
}
|
||||
v.m.perAppTier2 = func(string) error { return nil }
|
||||
if err := v.m.RunAppBackupNow(context.Background(), "app"); err != nil {
|
||||
t.Fatalf("RunAppBackupNow: %v", err)
|
||||
}
|
||||
man := v.manifest()
|
||||
if man.Data == nil || man.Data.Mixed || len(man.Data.Files) != 1 {
|
||||
t.Fatalf("data = %+v, want one stamped dump", man.Data)
|
||||
}
|
||||
st := man.Data.Files["db-dumps/app-postgres.sql"]
|
||||
if st.Images["app-db"] != "postgres:16-alpine@sha256:1616" || !samePins(st.Pins, v.fake.info.ImagePins) {
|
||||
t.Fatalf("stamp = %+v, want the running images and the definition's pins", st)
|
||||
}
|
||||
var raw map[string]interface{}
|
||||
b, _ := os.ReadFile(UnitManifestFile(v.unitDir()))
|
||||
_ = json.Unmarshal(b, &raw)
|
||||
if _, ok := raw["data"]; !ok {
|
||||
t.Fatal("manifest.json carries no `data` key")
|
||||
}
|
||||
}
|
||||
|
||||
// Tier 2: a mirror taken after the update of a unit whose data is older is as old as that data.
|
||||
func TestA4_ATier2CopyIsAsOldAsItsData(t *testing.T) {
|
||||
copyAt := time.Now().UTC().Truncate(time.Second)
|
||||
dataAt := copyAt.Add(-2 * time.Hour)
|
||||
p := Tier2RestorePoint{Restorable: true, CopyDateProven: true, CopyLastSuccess: copyAt.Format(time.RFC3339), DataDate: dataAt.Format(time.RFC3339)}
|
||||
got, ok := p.ProvenCopyTime()
|
||||
if !ok || !got.Equal(dataAt) {
|
||||
t.Fatalf("Tier 2 proven at %s ok=%v, want the data's %s", got, ok, dataAt)
|
||||
}
|
||||
p.DataDate = ""
|
||||
if got, _ := p.ProvenCopyTime(); !got.Equal(copyAt) {
|
||||
t.Fatalf("with no data date, Tier 2 = %s, want the copy's %s (as before)", got, copyAt)
|
||||
}
|
||||
}
|
||||
|
||||
// The household's version label names every image — a PostgreSQL step is visible in it.
|
||||
func TestA3_PinsVersionNamesEveryImage(t *testing.T) {
|
||||
got := PinsVersion([]string{"docmost/docmost:0.96.0@sha256:b5", "postgres:16-alpine@sha256:72", "gitea.dooplex.hu/x/redis:7-alpine"})
|
||||
if got != "docmost:0.96.0, postgres:16-alpine, redis:7-alpine" {
|
||||
t.Fatalf("PinsVersion = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// vtReconProvider is the reconstitution fixture's provider with a LIVE version (18) and a recorder for
|
||||
// the definition the restore writes.
|
||||
type vtReconProvider struct {
|
||||
*recordingProvider
|
||||
livePins []string
|
||||
wroteDef []string
|
||||
wroteAtCall int
|
||||
}
|
||||
|
||||
func (p *vtReconProvider) GetStackRecoveryInfo(string) (RecoveryInfo, bool) {
|
||||
return RecoveryInfo{DisplayName: "Immich", ImagePins: p.livePins}, true
|
||||
}
|
||||
func (p *vtReconProvider) RecreateStackDefinitionFromUnit(_, composeDir string, _ map[string]string) error {
|
||||
p.wroteDef = ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
|
||||
p.wroteAtCall = len(p.calls)
|
||||
p.calls = append(p.calls, "recreate")
|
||||
return nil
|
||||
}
|
||||
|
||||
// vtSnapshotUnit gives the fixture's scratch unit a definition and a `data` block (a snapshot taken by
|
||||
// v0.275.0), and returns the unit dir.
|
||||
func vtSnapshotUnit(t *testing.T, m *Manager, pgMajor string, withData bool) string {
|
||||
t.Helper()
|
||||
scratch, _, err := m.offboxRestoreScratchDir("immich")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var unit string
|
||||
_ = filepath.Walk(scratch, func(p string, fi os.FileInfo, _ error) error {
|
||||
if fi != nil && !fi.IsDir() && fi.Name() == "manifest.json" {
|
||||
unit = filepath.Dir(p)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if unit == "" {
|
||||
t.Fatal("no scratch unit")
|
||||
}
|
||||
mustWrite(t, filepath.Join(UnitComposeDir(unit), "docker-compose.yml"), vtCompose(pgMajor))
|
||||
mustWrite(t, filepath.Join(UnitComposeDir(unit), "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: vt\n")
|
||||
man := readManifest(UnitManifestFile(unit))
|
||||
if withData {
|
||||
man.Data = &UnitData{At: "2026-09-26T02:15:01Z", ImagePins: ParseComposeImages(filepath.Join(UnitComposeDir(unit), "docker-compose.yml"))}
|
||||
}
|
||||
if err := writeManifest(UnitManifestFile(unit), man); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return unit
|
||||
}
|
||||
|
||||
// Off-site (Tier 3): v0.274.0 never wrote the definition — last night's snapshot data went under the
|
||||
// app's NEW definition (A1, read from source). Now the snapshot's own definition is written BEFORE any
|
||||
// file, volume or database is touched, and the database service is resolved from it.
|
||||
func TestA3_TheOffsiteRestoreBringsTheSnapshotsVersionBack(t *testing.T) {
|
||||
m, prov, imported := reconFixture(t, "20260926T021500Z", "2026-09-26T02:15:01Z", pgDump(1))
|
||||
vp := &vtReconProvider{recordingProvider: prov, livePins: ParseComposeImages(writeTmpCompose(t, vtCompose("18")))}
|
||||
m.SetStackProvider(vp)
|
||||
vtSnapshotUnit(t, m, "16", true)
|
||||
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
|
||||
if err != nil {
|
||||
t.Fatalf("reconstitute: %v", err)
|
||||
}
|
||||
if got := strings.Join(vp.wroteDef, " "); !strings.Contains(got, "postgres:16-alpine") {
|
||||
t.Fatalf("the restore wrote the definition %q — want the snapshot's 16", got)
|
||||
}
|
||||
if vp.wroteAtCall != 1 || vp.calls[0] != "stop" {
|
||||
t.Fatalf("calls = %v — the definition must be written right after the stop, before any data", vp.calls)
|
||||
}
|
||||
if !res.VersionChanged || res.DataAt.Format(time.RFC3339) != "2026-09-26T02:15:01Z" || len(*imported) != 1 {
|
||||
t.Fatalf("result changed=%v at=%s imported=%v", res.VersionChanged, res.DataAt, *imported)
|
||||
}
|
||||
if got := strings.Join(vp.gotServices, ","); got != "app-db" {
|
||||
t.Fatalf("the DB-only start was %q — the database service must come from the definition that RUNS (the snapshot's)", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Same version, or a snapshot from before v0.275.0: nothing is written — the path of every earlier release.
|
||||
func TestA3_TheOffsiteRestoreLeavesTheDefinitionWhenNothingDiffers(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
live string
|
||||
withData bool
|
||||
unknown bool
|
||||
}{{"same-version", "16", true, false}, {"pre-v0.275.0-snapshot", "18", false, true}} {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
m, prov, _ := reconFixture(t, "20260926T021500Z", "2026-09-26T02:15:01Z", pgDump(1))
|
||||
vp := &vtReconProvider{recordingProvider: prov, livePins: ParseComposeImages(writeTmpCompose(t, vtCompose(c.live)))}
|
||||
m.SetStackProvider(vp)
|
||||
vtSnapshotUnit(t, m, "16", c.withData)
|
||||
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
|
||||
if err != nil {
|
||||
t.Fatalf("reconstitute: %v", err)
|
||||
}
|
||||
if vp.wroteDef != nil || res.VersionChanged || res.VersionsUnknown != c.unknown {
|
||||
t.Fatalf("wrote %v changed=%v unknown=%v", vp.wroteDef, res.VersionChanged, res.VersionsUnknown)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Tier 3: a snapshot pushed after a failed dump leg carries older data than its own time — the recorded
|
||||
// push caps it; a snapshot newer than the last recorded push (another box) keeps its own time.
|
||||
func TestA4_AnOffsiteCopyIsAsOldAsTheDataItWasPushedWith(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
snap := time.Date(2026, 9, 26, 2, 15, 0, 0, time.UTC)
|
||||
if got := m.offsiteDataTime("app", snap); !got.Equal(snap) {
|
||||
t.Fatalf("no record: %s, want the snapshot's own time", got)
|
||||
}
|
||||
data := snap.Add(-24 * time.Hour)
|
||||
if err := sett.SetOffsiteDataAt("app", settings.OffsiteDataRecord{PushedAt: snap.Add(time.Minute).Format(time.RFC3339), DataAt: data.Format(time.RFC3339)}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := m.offsiteDataTime("app", snap); !got.Equal(data) {
|
||||
t.Fatalf("recorded push: %s, want the data's %s", got, data)
|
||||
}
|
||||
later := snap.Add(2 * time.Hour)
|
||||
if got := m.offsiteDataTime("app", later); !got.Equal(later) {
|
||||
t.Fatalf("a snapshot newer than the recorded push: %s, want its own %s", got, later)
|
||||
}
|
||||
}
|
||||
|
||||
// The push records the pushed unit's DATA time (the production call in runOffboxInternal).
|
||||
func TestA4_ThePushRecordsTheUnitsDataTime(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
v := newVTStack(t, "16")
|
||||
dataAt := time.Now().Add(-26 * time.Hour).UTC().Truncate(time.Second)
|
||||
v.backupLegs(dataAt, "x")
|
||||
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.recordOffsiteDataAt("app", v.unitDir())
|
||||
rec, ok := sett.GetOffsiteDataAt("app")
|
||||
if !ok || rec.DataAt != dataAt.Format(time.RFC3339) || rec.PushedAt == "" {
|
||||
t.Fatalf("record = %+v ok=%v, want data at %s", rec, ok, dataAt.Format(time.RFC3339))
|
||||
}
|
||||
}
|
||||
|
||||
// The production push path writes the record (seam discipline: recordOffsiteDataAt is CALLED by
|
||||
// runOffboxInternal after a successful snapshot, not only testable on its own).
|
||||
func TestA4_TheOffsiteRunRecordsThePushedDataTime(t *testing.T) {
|
||||
drive := t.TempDir()
|
||||
m, sett, prov := classifiedOffboxManager(t, drive)
|
||||
u := mkUnit(t, drive, "immich")
|
||||
if err := writeManifest(UnitManifestFile(u), &RecoveryManifest{AppName: "immich", Data: &UnitData{At: "2026-09-25T02:15:01Z"}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prov.hdd["immich"] = drive
|
||||
prov.has["immich"] = true
|
||||
_ = sett.SetAppOffbox("immich", true)
|
||||
m.SetOffsitePreDumpFn(func(context.Context) error { return nil })
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
switch {
|
||||
case contains(args, "cat") && contains(args, "config"):
|
||||
return []byte(`{"version":2}`), nil
|
||||
case contains(args, "snapshots"):
|
||||
return []byte(`[]`), nil
|
||||
case contains(args, "stats"):
|
||||
return []byte(`{"total_size":123}`), nil
|
||||
}
|
||||
return nil, nil
|
||||
})
|
||||
if err := m.RunOffboxBackup(context.Background()); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
rec, ok := sett.GetOffsiteDataAt("immich")
|
||||
if !ok || rec.DataAt != "2026-09-25T02:15:01Z" || rec.PushedAt == "" {
|
||||
t.Fatalf("after a successful push the data-time record is %+v ok=%v, want the unit's data time", rec, ok)
|
||||
}
|
||||
}
|
||||
@@ -219,7 +219,7 @@ func (h *admissionHarness) runOneBackupRun() {
|
||||
done := h.m.beginAdmissionRun()
|
||||
defer done()
|
||||
h.m.runVolumeDumps()
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
}
|
||||
|
||||
// ── The instrument: a checksum of the whole backup tree ──────────────────────────────────────────
|
||||
@@ -664,7 +664,7 @@ func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
|
||||
|
||||
// 2. The DB leg consults it BEFORE the dump. Order is the whole point: a gate after the write is
|
||||
// the defect, relocated.
|
||||
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "DumpOne",
|
||||
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "dumpOneOrDefault", // v0.275.0: the dump goes through the seam wrapper
|
||||
"the DATABASE leg dumps before consulting the reserve")
|
||||
|
||||
// 3. The volume leg consults it BEFORE the dump seam — which stops the stack as its first act.
|
||||
@@ -691,7 +691,7 @@ func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
|
||||
}
|
||||
return true
|
||||
})
|
||||
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "CaptureRecoveryUnit",
|
||||
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "captureRecoveryUnit", // v0.275.0: the data-run flag is passed through
|
||||
"the CAPTURE leg captures before consulting the reserve")
|
||||
}
|
||||
|
||||
|
||||
@@ -172,6 +172,14 @@ type Manager struct {
|
||||
// disconnected) can be unit-tested without Docker. Nil → the real DumpAppVolumesSafe.
|
||||
dumpVolumesSafe func(stackName string) error
|
||||
|
||||
// dumpOne (v0.275.0) — the per-database dump seam, nil → the real DumpOne. It lets a test drive the
|
||||
// REAL legs (and so the stamps they write) without a database container.
|
||||
dumpOne func(ctx context.Context, db DiscoveredDB, dumpDir string, logger *log.Logger, debug bool) DumpResult
|
||||
|
||||
// stampMu serialises writes of a unit's data-stamps.json (v0.275.0, data_versions.go). The legs run
|
||||
// under the running flag already; this keeps a stray concurrent caller from losing a stamp.
|
||||
stampMu sync.Mutex
|
||||
|
||||
// updatingCheck (slice 4) — nil-safe; see isHeld / SetUpdatingCheck.
|
||||
updatingCheck func(stackName string) bool
|
||||
// undoCopyRemover (R-671, v0.272.0) deletes an app's leftover undo copies — stacks.Manager.RemoveUndoCopies,
|
||||
@@ -539,7 +547,13 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
defer m.beginRunSummary(kind, newRunID())()
|
||||
defer m.emitRunSummary()
|
||||
|
||||
dbs, err := DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||||
discover := m.discoverDBs
|
||||
if discover == nil {
|
||||
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
||||
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||||
}
|
||||
}
|
||||
dbs, err := discover(ctx)
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Database discovery failed: %v", err)
|
||||
return err
|
||||
@@ -586,7 +600,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
|
||||
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), db.StackName)
|
||||
|
||||
result := DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
|
||||
result := m.dumpOneOrDefault(ctx, db, dumpDir)
|
||||
results = append(results, result)
|
||||
|
||||
if result.Error != nil {
|
||||
@@ -597,6 +611,8 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
} else {
|
||||
totalSize += result.Size
|
||||
summary = append(summary, fmt.Sprintf("OK %s (%s)", result.DB.ContainerName, humanizeBytes(result.Size)))
|
||||
// v0.275.0 (R-696): the dump records the versions that wrote it, at the moment it is written.
|
||||
m.stampDataFile(db.StackName, RecoveryUnitPath(m.namespaceRoot(drivePath), db.StackName), "db-dumps/"+filepath.Base(result.FilePath))
|
||||
|
||||
// Persist validation result to settings.json
|
||||
if m.settings != nil && result.FilePath != "" {
|
||||
@@ -644,8 +660,9 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
strings.Join(failedSummaryLines(summary), "; "))
|
||||
}
|
||||
|
||||
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest).
|
||||
m.captureAllRecoveryUnits()
|
||||
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest). A DATA run:
|
||||
// the capture folds the stamps the legs above just wrote (v0.275.0).
|
||||
m.captureAllRecoveryUnits(true)
|
||||
|
||||
// F5 (CAMPAIGN-3): after the units are fresh on the CURRENT drives, prune any orphaned
|
||||
// backups/primary/<app> dir an app left on an OLD drive when its HDD_PATH moved — pure disk
|
||||
@@ -823,6 +840,8 @@ func (m *Manager) DumpAppVolumes(stackName string) error {
|
||||
if info, _ := os.Stat(tarPath); info != nil {
|
||||
m.logger.Printf("[INFO] [backup] Volume dump: %s/%s → %s", stackName, volName, humanizeBytes(info.Size()))
|
||||
}
|
||||
// v0.275.0 (R-696): the tar records the versions that wrote it.
|
||||
m.stampDataFile(stackName, RecoveryUnitPath(m.namespaceRoot(drivePath), stackName), "volume-dumps/"+volName+".tar")
|
||||
}
|
||||
|
||||
// Clean up tars (and any orphan `.tar.tmp` from a killed run) for volumes that no longer exist.
|
||||
@@ -1158,7 +1177,7 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
|
||||
func() {
|
||||
defer m.beginRunSummary(runKindRefresh, "")()
|
||||
defer m.emitRunSummary()
|
||||
m.captureAllRecoveryUnits()
|
||||
m.captureAllRecoveryUnits(false) // a REFRESH: never moves the unit's data time or its definition away from its data
|
||||
}()
|
||||
}
|
||||
|
||||
@@ -1439,3 +1458,11 @@ func (m *Manager) stackIsDeploying(name string) bool {
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// dumpOneOrDefault runs the dumpOne seam, or the real DumpOne.
|
||||
func (m *Manager) dumpOneOrDefault(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
|
||||
if m.dumpOne != nil {
|
||||
return m.dumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
|
||||
}
|
||||
return DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
|
||||
}
|
||||
|
||||
@@ -112,7 +112,7 @@ func TestFloor_RefusesTheAppAndLeavesItsPreviousUnitByteIdentical(t *testing.T)
|
||||
}
|
||||
before := checksumFile(t, prev)
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
|
||||
// The refused app must NOT have been attempted at all — the floor is checked BEFORE any write.
|
||||
for _, hit := range h.prov.infoHits {
|
||||
@@ -170,7 +170,7 @@ func TestFloor_NeverDeletesAnotherAppsUnit(t *testing.T) {
|
||||
}
|
||||
before := checksumFile(t, keep)
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
|
||||
if _, err := os.Stat(keep); err != nil {
|
||||
t.Fatalf("another app's unit was DELETED to make room: %v — nothing here is generational, so "+
|
||||
@@ -190,7 +190,7 @@ func TestFloor_LargeUnitWithAmpleSpaceIsCaptured(t *testing.T) {
|
||||
// A huge app on a huge, mostly-empty filesystem: 40% used, 600 GB free.
|
||||
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 600, TotalGB: 1000, UsedGB: 400}
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("a capture was refused on a filesystem with 600 GB free (%+v) — the floor has become "+
|
||||
@@ -209,7 +209,7 @@ func TestFloor_TheOld20GCeilingIsGone(t *testing.T) {
|
||||
// deliberately far above 20 so that a literal `UsedGB > 20` cap cannot survive this test: a
|
||||
// fixture sitting exactly on the old boundary would pass under the very shape it forbids.
|
||||
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 180, TotalGB: 300, UsedGB: 120}
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("refused with 180 GB free: %+v — a fixed per-area limit survives somewhere", h.events)
|
||||
}
|
||||
@@ -258,7 +258,7 @@ func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
// No entry → the injected reader returns nil, which is what system.GetDiskUsage does on error.
|
||||
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
|
||||
if len(h.events) != 0 {
|
||||
t.Fatalf("an UNREADABLE filesystem produced %d alert(s): %+v — an absent, unmounted or "+
|
||||
@@ -275,7 +275,7 @@ func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
|
||||
func TestErrCaptureFloor_IsMatchable(t *testing.T) {
|
||||
h := newFloorHarness(t, "immich")
|
||||
h.setSpace("immich", 99, 0.2)
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
if len(h.events) != 1 {
|
||||
t.Fatalf("want 1 event, got %d", len(h.events))
|
||||
}
|
||||
|
||||
@@ -0,0 +1,309 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// ── The backup's data and its version travel together (controller v0.275.0, R-696, `07` §6.6) ────────
|
||||
//
|
||||
// WHAT WENT WRONG. The recovery unit's DEFINITION (compose/, image_pins) was re-captured by the periodic
|
||||
// status refresh as soon as the app's pin moved, while its DATA (db-dumps/, volume-dumps/) is written
|
||||
// only by the backup legs. So for up to a day after an update the unit said "new version" and held the
|
||||
// old version's data. Measured on 9202 2026-09-26 (`audits/version-travel-2026-09-26/A1/`): after a
|
||||
// PostgreSQL 16 → 18 step the unit held the 18 definition over a 16 dump and a 16 datadir tar; a restore
|
||||
// poured the 16 datadir back, `postgres:18` refused it and the app was left down. And Tier 1's "proven
|
||||
// at" was the manifest's refresh time, so the kept pre-conversion copy was released on a backup taken
|
||||
// BEFORE the conversion (demo-hp, 2026-09-26 night).
|
||||
//
|
||||
// THE SHAPE. Every data file a backup leg writes gets a STAMP beside it, at the moment it is written:
|
||||
// its size and mtime (so a file rewritten by anything else is recognised as unstamped), the definition's
|
||||
// image pins, and what each service was running (`installed_images`, ref@digest). The unit capture folds
|
||||
// the stamps into the manifest's `data` block — the time of the unit's data (its OLDEST stamped file:
|
||||
// the copy is as fresh as its stalest part) and the versions that wrote it — and KEEPS the definition
|
||||
// the data belongs to: when the pins have moved since the data was written, compose/ is not rewritten
|
||||
// until the next data run replaces the data. The manifest's `image_pins` stays the app's CURRENT pins,
|
||||
// so it says both. Restores read `data` to start the data with its own definition (restore_unit.go),
|
||||
// and the update/release read `data.at` as the unit's time (restore_points.go).
|
||||
//
|
||||
// UNKNOWN IS NEVER CURRENT. A unit whose data files are not all validly stamped (written before
|
||||
// v0.275.0, or by a path that does not stamp) has NO `data` block, restores as before with a WARN, and
|
||||
// its time is the newest DATA file's mtime — never the manifest's.
|
||||
|
||||
// dataStampsFile sits in the unit root, beside manifest.json, so it travels with every copy of the unit
|
||||
// (the Tier-2 mirror copies the whole unit, the off-site snapshot includes it).
|
||||
const dataStampsFile = "data-stamps.json"
|
||||
|
||||
// DataStamp is one data file's record, written by the leg that wrote the file.
|
||||
type DataStamp struct {
|
||||
// At is the file's mtime when it was stamped, RFC3339Nano UTC — the stamp is valid only while the
|
||||
// file still has exactly this mtime and Size.
|
||||
At string `json:"at"`
|
||||
Size int64 `json:"size"`
|
||||
// Pins are the definition's image pins (the compose `image:` lines) when the file was written.
|
||||
Pins []string `json:"pins"`
|
||||
// Images is service -> ref@digest running when the file was written. Empty when not observed.
|
||||
Images map[string]string `json:"images,omitempty"`
|
||||
}
|
||||
|
||||
// UnitData is the manifest's account of the data the unit holds.
|
||||
type UnitData struct {
|
||||
// At is RFC3339 UTC: the unit's data time — the oldest data file's stamp, or, for a unit with no data
|
||||
// files, the data run that confirmed it. This is the unit's "proven at", never the manifest's time.
|
||||
At string `json:"at"`
|
||||
// ImagePins are the definition pins the data belongs to — what compose/ holds. Empty when Mixed.
|
||||
ImagePins []string `json:"image_pins"`
|
||||
// Images is the running set (service -> ref@digest) when the oldest file was written.
|
||||
Images map[string]string `json:"images,omitempty"`
|
||||
// Files are the stamps, keyed by the path relative to the unit ("db-dumps/x.sql").
|
||||
Files map[string]DataStamp `json:"files,omitempty"`
|
||||
// Mixed: the files were written under DIFFERENT pins — no one definition fits all of them, so a
|
||||
// restore of this unit is refused (a restore never starts data with a definition it does not belong to).
|
||||
Mixed bool `json:"mixed,omitempty"`
|
||||
}
|
||||
|
||||
// DataTime parses At. False when absent or unreadable.
|
||||
func (d *UnitData) DataTime() (time.Time, bool) {
|
||||
if d == nil || d.At == "" {
|
||||
return time.Time{}, false
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339, d.At)
|
||||
if err != nil {
|
||||
return time.Time{}, false
|
||||
}
|
||||
return t, true
|
||||
}
|
||||
|
||||
func readDataStamps(unitDir string) map[string]DataStamp {
|
||||
data, err := os.ReadFile(filepath.Join(unitDir, dataStampsFile))
|
||||
if err != nil {
|
||||
return map[string]DataStamp{}
|
||||
}
|
||||
var out map[string]DataStamp
|
||||
if json.Unmarshal(data, &out) != nil || out == nil {
|
||||
return map[string]DataStamp{}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// stampDataFile records the file at <unitDir>/<rel> as written NOW by the current definition and
|
||||
// running images. Called by the legs right after a dump or a tar is promoted to its final name. A failure
|
||||
// to stamp is a WARN and leaves the file unstamped — the unit then reads as "versions unknown", which is
|
||||
// the pre-v0.275.0 behaviour, never a false claim.
|
||||
func (m *Manager) stampDataFile(stackName, unitDir, rel string) {
|
||||
fi, err := os.Stat(filepath.Join(unitDir, rel))
|
||||
if err != nil {
|
||||
m.logger.Printf("[WARN] [backup] %s: cannot stamp %s (%v) — its versions will read as unknown", stackName, rel, err)
|
||||
return
|
||||
}
|
||||
var pins []string
|
||||
var images map[string]string
|
||||
if m.stackProvider != nil {
|
||||
if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok {
|
||||
pins, images = definitionPins(info), info.InstalledImages
|
||||
}
|
||||
}
|
||||
m.stampMu.Lock()
|
||||
defer m.stampMu.Unlock()
|
||||
stamps := readDataStamps(unitDir)
|
||||
stamps[rel] = DataStamp{At: fi.ModTime().UTC().Format(time.RFC3339Nano), Size: fi.Size(), Pins: pins, Images: images}
|
||||
// Entries whose file is gone (a volume the app no longer has) are dropped here, so the file stays small.
|
||||
for k := range stamps {
|
||||
if _, err := os.Stat(filepath.Join(unitDir, k)); err != nil {
|
||||
delete(stamps, k)
|
||||
}
|
||||
}
|
||||
body, err := json.MarshalIndent(stamps, "", " ")
|
||||
if err == nil {
|
||||
err = atomicWrite(filepath.Join(unitDir, dataStampsFile), append(body, '\n'), 0644)
|
||||
}
|
||||
if err != nil {
|
||||
m.logger.Printf("[WARN] [backup] %s: writing the data stamp for %s failed (%v) — its versions will read as unknown", stackName, rel, err)
|
||||
return
|
||||
}
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] [backup] %s: stamped %s (%d B) with pins %v", stackName, rel, fi.Size(), pins)
|
||||
}
|
||||
}
|
||||
|
||||
// foldUnitData builds the manifest's `data` block from the unit's data files and their stamps.
|
||||
//
|
||||
// - no data files: a data run confirms the unit NOW under the current pins; a refresh keeps `prev`;
|
||||
// - every file validly stamped: the oldest stamp's time; its pins when all agree, else Mixed;
|
||||
// - any file unstamped or re-written since its stamp: nil — the versions are UNKNOWN.
|
||||
func foldUnitData(unitDir string, dbDumps, volDumps []string, dataRun bool, now time.Time, pins []string, images map[string]string, prev *UnitData) *UnitData {
|
||||
var rels []string
|
||||
for _, n := range dbDumps {
|
||||
rels = append(rels, "db-dumps/"+n)
|
||||
}
|
||||
for _, n := range volDumps {
|
||||
rels = append(rels, "volume-dumps/"+n)
|
||||
}
|
||||
if len(rels) == 0 {
|
||||
if dataRun {
|
||||
return &UnitData{At: now.UTC().Format(time.RFC3339), ImagePins: pins, Images: images}
|
||||
}
|
||||
return prev
|
||||
}
|
||||
stamps := readDataStamps(unitDir)
|
||||
out := &UnitData{Files: map[string]DataStamp{}}
|
||||
var oldest time.Time
|
||||
for _, rel := range rels {
|
||||
st, ok := stamps[rel]
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
fi, err := os.Stat(filepath.Join(unitDir, rel))
|
||||
if err != nil || fi.Size() != st.Size || fi.ModTime().UTC().Format(time.RFC3339Nano) != st.At {
|
||||
return nil
|
||||
}
|
||||
t := fi.ModTime().UTC()
|
||||
if oldest.IsZero() || t.Before(oldest) {
|
||||
oldest = t
|
||||
out.ImagePins, out.Images = st.Pins, st.Images
|
||||
}
|
||||
out.Files[rel] = st
|
||||
}
|
||||
for _, st := range out.Files {
|
||||
if !samePins(st.Pins, out.ImagePins) {
|
||||
out.Mixed = true
|
||||
}
|
||||
}
|
||||
if out.Mixed {
|
||||
out.ImagePins, out.Images = nil, nil
|
||||
}
|
||||
out.At = oldest.Format(time.RFC3339)
|
||||
return out
|
||||
}
|
||||
|
||||
// samePins compares two pin lists as SETS (compose order is file order on both sides, but a set compare
|
||||
// cannot be fooled by it).
|
||||
func samePins(a, b []string) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
x := append([]string(nil), a...)
|
||||
y := append([]string(nil), b...)
|
||||
sort.Strings(x)
|
||||
sort.Strings(y)
|
||||
return stringSliceEqual(x, y)
|
||||
}
|
||||
|
||||
// unitDataEqual is the manifest-rewrite check's view of `data`.
|
||||
func unitDataEqual(a, b *UnitData) bool {
|
||||
if a == nil || b == nil {
|
||||
return a == nil && b == nil
|
||||
}
|
||||
if a.At != b.At || a.Mixed != b.Mixed || !stringSliceEqual(a.ImagePins, b.ImagePins) || len(a.Files) != len(b.Files) {
|
||||
return false
|
||||
}
|
||||
for k, v := range a.Files {
|
||||
w, ok := b.Files[k]
|
||||
if !ok || v.At != w.At || v.Size != w.Size {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// PinsVersion is the household's name for a set of pins: every image as `name:tag`, registry path and
|
||||
// digest stripped, in compose order ("docmost:0.96.0, postgres:16-alpine, redis:7-alpine"). All of them,
|
||||
// because a version step can move ANY service — a PostgreSQL 16 → 18 step leaves the app's own tag as it
|
||||
// was, and a label naming only the first image would read the same on both sides of it (seen on the
|
||||
// first draft of the mismatch sentence: „(0.96.0) … (0.96.0)").
|
||||
func PinsVersion(pins []string) string {
|
||||
out := make([]string, 0, len(pins))
|
||||
for _, ref := range pins {
|
||||
if i := strings.Index(ref, "@"); i >= 0 {
|
||||
ref = ref[:i]
|
||||
}
|
||||
if i := strings.LastIndex(ref, "/"); i >= 0 {
|
||||
ref = ref[i+1:]
|
||||
}
|
||||
out = append(out, ref)
|
||||
}
|
||||
return strings.Join(out, ", ")
|
||||
}
|
||||
|
||||
// recordOffsiteDataAt remembers the data time of the unit just pushed off-site (v0.275.0, R-696).
|
||||
func (m *Manager) recordOffsiteDataAt(stackName, unitDir string) {
|
||||
if m.settings == nil {
|
||||
return
|
||||
}
|
||||
t, ok := unitNewestArtifact(unitDir)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
now := time.Now().UTC().Format(time.RFC3339)
|
||||
if err := m.settings.SetOffsiteDataAt(stackName, settings.OffsiteDataRecord{PushedAt: now, DataAt: t.UTC().Format(time.RFC3339)}); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: recording the pushed copy's data time failed: %v — the off-site copy is dated by its snapshot", stackName, err)
|
||||
}
|
||||
}
|
||||
|
||||
// offsiteDataTime caps an off-site snapshot's time with the data time this box recorded when it pushed
|
||||
// it. A snapshot NEWER than the last recorded push (taken by another box, or a record lost) keeps its
|
||||
// own time: the cap applies only to a copy this box knows the content of.
|
||||
func (m *Manager) offsiteDataTime(stackName string, snapshotAt time.Time) time.Time {
|
||||
if m.settings == nil {
|
||||
return snapshotAt
|
||||
}
|
||||
rec, ok := m.settings.GetOffsiteDataAt(stackName)
|
||||
if !ok {
|
||||
return snapshotAt
|
||||
}
|
||||
pushed, perr := time.Parse(time.RFC3339, rec.PushedAt)
|
||||
data, derr := time.Parse(time.RFC3339, rec.DataAt)
|
||||
if perr != nil || derr != nil || snapshotAt.After(pushed) || !data.Before(snapshotAt) {
|
||||
return snapshotAt
|
||||
}
|
||||
return data
|
||||
}
|
||||
|
||||
// UnitDumpStamp is one stamped database dump of an app's own unit (v0.275.0).
|
||||
type UnitDumpStamp struct {
|
||||
File string
|
||||
At time.Time
|
||||
Images map[string]string
|
||||
}
|
||||
|
||||
// UnitDumpStamps returns the app's OWN unit's database dumps as its manifest records them — only when
|
||||
// the unit's data is known (a `data` block); an unstamped unit returns nothing, and a release that needs
|
||||
// one waits (fail closed).
|
||||
func (m *Manager) UnitDumpStamps(stackName string) []UnitDumpStamp {
|
||||
man := readManifest(UnitManifestFile(m.primaryUnitDirFor(stackName)))
|
||||
if man == nil || man.Data == nil {
|
||||
return nil
|
||||
}
|
||||
var out []UnitDumpStamp
|
||||
for rel, st := range man.Data.Files {
|
||||
if !strings.HasPrefix(rel, "db-dumps/") || !strings.HasSuffix(rel, ".sql") {
|
||||
continue
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339Nano, st.At)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
out = append(out, UnitDumpStamp{File: rel, At: t, Images: st.Images})
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].File < out[j].File })
|
||||
return out
|
||||
}
|
||||
|
||||
// definitionPins are the pins of the definition a capture copies into compose/: the stack dir's own
|
||||
// docker-compose.yml, parsed. ONE source for the stamps, the fold and the freeze, so "the data's pins"
|
||||
// and "the pins compose/ holds" are read from the same file the restore will start (production's
|
||||
// RecoveryInfo.ImagePins is that parse too; a provider that says otherwise cannot make them disagree).
|
||||
func definitionPins(info RecoveryInfo) []string {
|
||||
if info.StackDir != "" {
|
||||
if p := ParseComposeImages(filepath.Join(info.StackDir, "docker-compose.yml")); len(p) > 0 {
|
||||
return p
|
||||
}
|
||||
}
|
||||
return info.ImagePins
|
||||
}
|
||||
@@ -1372,6 +1372,9 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
|
||||
continue
|
||||
}
|
||||
res.backedUp++
|
||||
// v0.275.0 (R-696): what the snapshot just taken HOLDS is the unit's data, of the unit's data time —
|
||||
// recorded so the update's precondition dates the off-site copy by its data, not by the snapshot.
|
||||
m.recordOffsiteDataAt(stack, src)
|
||||
// R-412 leg 1 — A PUSH THAT CARRIED NOTHING MUST NOT READ AS A PLAIN SUCCESS.
|
||||
//
|
||||
// Measured on demo-hp 2026-08-31: a recovery unit was destroyed mid-run, the capture rebuilt it
|
||||
|
||||
@@ -91,6 +91,13 @@ type OffsiteReconstituteResult struct {
|
||||
// than reporting a bare success — a warning beside a success is read as a success, so the
|
||||
// difference has to survive into the message.
|
||||
Placement PlacementCheck
|
||||
// v0.275.0 (R-696, `07` §6.6) — the versions the snapshot's data belongs to (UnitRestoreResult's
|
||||
// fields, same meaning). VersionChanged: the app came back at the SNAPSHOT's version, its definition
|
||||
// written from the snapshot's unit, because the live one was another version.
|
||||
DataPins []string
|
||||
DataAt time.Time
|
||||
VersionChanged bool
|
||||
VersionsUnknown bool
|
||||
}
|
||||
|
||||
// fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite).
|
||||
@@ -693,12 +700,50 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
|
||||
return res, util.MsgError("err.backup.adatbazis_masolat_csonka_nem_indult", stack)
|
||||
}
|
||||
|
||||
// --- WHICH VERSION COMES BACK (v0.275.0, R-696, `07` §6.6) ---------------------------------
|
||||
// Until v0.275.0 this path never wrote the definition, so after any update the snapshot's data
|
||||
// (last night's version) was started by the app's NEW definition — for an engine step that is the
|
||||
// measured A1 failure (a PostgreSQL 16 datadir under 18: refused, the app left down). Now the
|
||||
// snapshot's data comes back with the definition it belongs to: when the snapshot records its data's
|
||||
// versions and they differ from what runs, the snapshot unit's definition is written into the stack
|
||||
// dir (and pinned) before anything is started, and the normal guarded update climbs from there. A
|
||||
// snapshot without that record restores as before, WARNed. Decided before the first mutation.
|
||||
scratchCompose := UnitComposeDir(scratchUnit)
|
||||
dataPins, verr := unitVersionCheck(stack, man, scratchCompose)
|
||||
if verr != nil {
|
||||
m.logger.Printf("[ERROR] [offbox] Restore REFUSED for %s: the snapshot's definition does not belong to its data — nothing was touched", stack)
|
||||
return res, verr
|
||||
}
|
||||
var snapEnv map[string]string
|
||||
defineFromSnapshot := false
|
||||
if dataPins == nil {
|
||||
res.VersionsUnknown = true
|
||||
m.logger.Printf("[WARN] [offbox] %s: snapshot %s does not record which versions wrote its data (taken before v0.275.0) — restoring into the app's current definition, as before", stack, id)
|
||||
} else {
|
||||
res.DataPins = dataPins
|
||||
res.DataAt, _ = man.Data.DataTime()
|
||||
if info, ok := m.stackProvider.GetStackRecoveryInfo(stack); ok && len(definitionPins(info)) > 0 && !samePins(definitionPins(info), dataPins) {
|
||||
env, _, eerr := m.unitRestoreEnv(stack, scratchCompose, man)
|
||||
if eerr != nil {
|
||||
return res, eerr
|
||||
}
|
||||
snapEnv, defineFromSnapshot, res.VersionChanged = env, true, true
|
||||
m.logger.Printf("[INFO] [offbox] %s: snapshot %s holds data of %v (written %s); the app runs %v — it comes back at the snapshot's version", stack, id, dataPins, man.Data.At, definitionPins(info))
|
||||
}
|
||||
}
|
||||
|
||||
// --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------
|
||||
// Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir,
|
||||
// so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first
|
||||
// mutation so the refusal below costs nothing.
|
||||
// Read from the compose that will RUN: the live one, or — when the app comes back at the snapshot's
|
||||
// version — the snapshot unit's, which is written into the stack dir before the first start. Resolved
|
||||
// BEFORE the first mutation so the refusal below costs nothing.
|
||||
var dbServices []string
|
||||
if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
|
||||
if defineFromSnapshot {
|
||||
svcs, dsErr := DBServiceNames(filepath.Join(scratchCompose, "docker-compose.yml"))
|
||||
if dsErr != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: could not read the snapshot's compose services: %v", stack, dsErr)
|
||||
}
|
||||
dbServices = svcs
|
||||
} else if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
|
||||
svcs, dsErr := DBServiceNames(composePath)
|
||||
if dsErr != nil {
|
||||
// "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse.
|
||||
@@ -754,6 +799,16 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
|
||||
if err := m.stackProvider.StopStack(stack); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err)
|
||||
}
|
||||
// v0.275.0: the snapshot's own definition, written while nothing of the app's data has been touched
|
||||
// yet — a failure here restarts the app as it was.
|
||||
if defineFromSnapshot {
|
||||
if err := m.stackProvider.RecreateStackDefinitionFromUnit(stack, scratchCompose, snapEnv); err != nil {
|
||||
if sErr := restartStack(); sErr != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s: restart after a failed definition write also failed: %v", stack, sErr)
|
||||
}
|
||||
return res, fmt.Errorf("restoring %s: writing the snapshot's definition failed: %w", stack, err)
|
||||
}
|
||||
}
|
||||
copier := m.fullPlaceCopier()
|
||||
for _, pl := range placements {
|
||||
if pl.isUnit {
|
||||
|
||||
@@ -69,6 +69,11 @@ type RecoveryManifest struct {
|
||||
// never claim a coherence it did not establish — it carries the prior stamp forward instead.
|
||||
OffsiteRunID string `json:"offsite_run_id,omitempty"`
|
||||
DumpsAt string `json:"dumps_at,omitempty"` // RFC3339 UTC — when this run's dump leg finished
|
||||
// Data (v0.275.0, R-696) is the account of the DATA this unit holds: its time and the versions that
|
||||
// wrote it (data_versions.go). compose/ holds the definition THESE pins name; ImagePins above are the
|
||||
// app's CURRENT pins — the two differ between an update and the next data run. Nil = unknown (a unit
|
||||
// written before v0.275.0, or one with an unstamped data file): it restores as before, with a WARN.
|
||||
Data *UnitData `json:"data,omitempty"`
|
||||
}
|
||||
|
||||
// SetVersion records the controller version stamped into recovery-unit manifests.
|
||||
@@ -92,7 +97,15 @@ func (m *Manager) SetTier2Notifier(fn func(stackName, destLabel string, dur time
|
||||
// Idempotent: it builds the captured content in memory first and SKIPS all writes when the unit is
|
||||
// already current (same config checksums, same dump set, same controller version) — so it can run on
|
||||
// the periodic status refresh without thrashing a spinning USB drive.
|
||||
//
|
||||
// v0.275.0 (R-696): CaptureRecoveryUnit is the capture AFTER a data run (it follows the legs in
|
||||
// RunAppBackupNow). The periodic refresh calls captureRecoveryUnit(…, false), which never moves the
|
||||
// unit's data time and never rewrites the definition away from the data it belongs to.
|
||||
func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
return m.captureRecoveryUnit(stackName, true)
|
||||
}
|
||||
|
||||
func (m *Manager) captureRecoveryUnit(stackName string, dataRun bool) error {
|
||||
if m.stackProvider == nil {
|
||||
return fmt.Errorf("no stack provider")
|
||||
}
|
||||
@@ -161,6 +174,38 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
|
||||
manifestPath := RecoveryUnitManifestPath(nsRoot, stackName)
|
||||
cur := readManifest(manifestPath)
|
||||
unitDir := RecoveryUnitPath(nsRoot, stackName)
|
||||
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
|
||||
|
||||
// v0.275.0 (R-696): the data's own account, folded from the stamps the legs wrote.
|
||||
var prevData *UnitData
|
||||
if cur != nil {
|
||||
prevData = cur.Data
|
||||
}
|
||||
curPins := definitionPins(info)
|
||||
data := foldUnitData(unitDir, dbDumps, volDumps, dataRun, time.Now(), curPins, info.InstalledImages, prevData)
|
||||
|
||||
// THE DEFINITION STAYS WITH ITS DATA. When the app's pins have moved since the data was written (an
|
||||
// update between two data runs), compose/ keeps the definition the data belongs to until the next data
|
||||
// run replaces the data — the refresh used to rewrite it here within five minutes of every update,
|
||||
// pairing the new version with the old data (A1: a 16 datadir under an 18 definition; the restore left
|
||||
// the app down). The checksums then describe what compose/ really holds, so the already-current check
|
||||
// below does not rewrite the manifest on every refresh.
|
||||
frozen := data != nil && !data.Mixed && !samePins(data.ImagePins, curPins)
|
||||
if frozen {
|
||||
files, checksums, configFiles = nil, map[string]string{}, nil
|
||||
for _, fname := range []string{"docker-compose.yml", ".felhom.yml", "app.yaml"} {
|
||||
b, err := os.ReadFile(filepath.Join(composeDir, fname))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
checksums[fname] = sha256Hex(b)
|
||||
configFiles = append(configFiles, fname)
|
||||
}
|
||||
if held := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")); !samePins(held, data.ImagePins) {
|
||||
m.logger.Printf("[WARN] [backup] %s: the unit's definition %v matches neither its data %v nor the app %v — kept as it is; a restore of this unit will refuse", stackName, held, data.ImagePins, curPins)
|
||||
}
|
||||
}
|
||||
|
||||
// R-43/R-44: the coherence stamp of the offsite run currently in flight ("" on the periodic
|
||||
// refresh and on the local dump run). When empty we CARRY THE PRIOR STAMP FORWARD rather than
|
||||
@@ -193,11 +238,16 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
stringMapEqual(cur.Checksums, checksums) &&
|
||||
stringSliceEqual(cur.DBDumps, dbDumps) &&
|
||||
stringSliceEqual(cur.VolumeDumps, volDumps) &&
|
||||
stringSliceEqual(cur.ImagePins, info.ImagePins) &&
|
||||
unitDataEqual(cur.Data, data) &&
|
||||
cur.OffsiteRunID == runID {
|
||||
return nil
|
||||
}
|
||||
|
||||
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
|
||||
if frozen && (cur == nil || cur.Data == nil || stringSliceEqual(cur.ImagePins, cur.Data.ImagePins)) {
|
||||
m.logger.Printf("[INFO] [backup] %s: the app now runs %v; the unit keeps the definition of its data (%v, written %s) until the next backup replaces the data",
|
||||
stackName, curPins, data.ImagePins, data.At)
|
||||
}
|
||||
if err := os.MkdirAll(composeDir, 0755); err != nil {
|
||||
return fmt.Errorf("creating recovery-unit compose dir: %w", err)
|
||||
}
|
||||
@@ -226,6 +276,10 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
|
||||
Checksums: checksums,
|
||||
OffsiteRunID: runID,
|
||||
DumpsAt: dumpsAt,
|
||||
Data: data,
|
||||
}
|
||||
if data == nil && (len(dbDumps)+len(volDumps)) > 0 && m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] [backup] %s: the unit's data files are not all stamped — its versions are unknown until the next backup", stackName)
|
||||
}
|
||||
if err := writeManifest(manifestPath, manifest); err != nil {
|
||||
return fmt.Errorf("writing manifest: %w", err)
|
||||
@@ -387,7 +441,7 @@ func (m *Manager) readUnitSpace(stackName string) *UnitSpace {
|
||||
// volume-dump legs of this run already consulted for this app. When a run is in flight the answer
|
||||
// here is a memo lookup — an app refused before its first write is refused here too, silently,
|
||||
// because it was already alerted once. Outside a run (the periodic status refresh) it decides fresh.
|
||||
func (m *Manager) captureAllRecoveryUnits() {
|
||||
func (m *Manager) captureAllRecoveryUnits(dataRun bool) {
|
||||
if m.stackProvider == nil {
|
||||
return
|
||||
}
|
||||
@@ -409,7 +463,7 @@ func (m *Manager) captureAllRecoveryUnits() {
|
||||
if !m.admitApp(stack.Name) {
|
||||
continue
|
||||
}
|
||||
if err := m.CaptureRecoveryUnit(stack.Name); err != nil {
|
||||
if err := m.captureRecoveryUnit(stack.Name, dataRun); err != nil {
|
||||
m.noteFailure(stack.Name, "recovery-unit capture", err.Error())
|
||||
m.logger.Printf("[WARN] [backup] Recovery unit capture failed for %s: %v", stack.Name, err)
|
||||
// R-158: per app, and the loop CONTINUES — one app's failure must not silence the
|
||||
|
||||
@@ -85,7 +85,7 @@ func newUnitNotifyManager(t *testing.T, stacks []string, fail map[string]bool) (
|
||||
func TestCaptureAll_FailureNotifiesOnceWithTheSpaceFigures(t *testing.T) {
|
||||
m, got := newUnitNotifyManager(t, []string{"immich"}, map[string]bool{"immich": true})
|
||||
|
||||
m.captureAllRecoveryUnits()
|
||||
m.captureAllRecoveryUnits(false)
|
||||
|
||||
if len(*got) != 1 {
|
||||
t.Fatalf("got %d unit-failure events, want exactly 1 — a per-app Tier-1 capture failure "+
|
||||
@@ -117,7 +117,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
|
||||
[]string{"homebox", "immich", "nextcloud"},
|
||||
map[string]bool{"immich": true})
|
||||
|
||||
m.captureAllRecoveryUnits()
|
||||
m.captureAllRecoveryUnits(false)
|
||||
|
||||
if len(*got) != 1 {
|
||||
t.Fatalf("got %d events, want exactly 1 — either the loop ABORTED on the middle app "+
|
||||
@@ -134,7 +134,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
|
||||
m2, got2 := newUnitNotifyManager(t,
|
||||
[]string{"homebox", "immich", "nextcloud"},
|
||||
map[string]bool{"immich": true, "nextcloud": true})
|
||||
m2.captureAllRecoveryUnits()
|
||||
m2.captureAllRecoveryUnits(false)
|
||||
if len(*got2) != 2 {
|
||||
t.Fatalf("got %d events, want 2 — the app AFTER the first failure was never reached, so the "+
|
||||
"loop is aborting rather than continuing: %+v", len(*got2), *got2)
|
||||
@@ -148,7 +148,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
|
||||
// to ignore.
|
||||
func TestCaptureAll_SuccessIsSilent(t *testing.T) {
|
||||
m, got := newUnitNotifyManager(t, []string{"homebox"}, nil)
|
||||
m.captureAllRecoveryUnits()
|
||||
m.captureAllRecoveryUnits(false)
|
||||
if len(*got) != 0 {
|
||||
t.Fatalf("a successful capture fired %d event(s): %+v", len(*got), *got)
|
||||
}
|
||||
@@ -163,7 +163,7 @@ func TestCaptureAll_UnwiredNotifyDoesNotPanic(t *testing.T) {
|
||||
systemDataPath: dir,
|
||||
stackProvider: &unitFailProvider{stacks: []string{"immich"}, fail: map[string]bool{"immich": true}, dir: dir},
|
||||
}
|
||||
m.captureAllRecoveryUnits() // no SetUnitNotify — must not panic
|
||||
m.captureAllRecoveryUnits(false) // no SetUnitNotify — must not panic
|
||||
}
|
||||
|
||||
// §8.4 in the failure direction: an unreadable target filesystem is reported as UNKNOWN, never as
|
||||
|
||||
@@ -66,17 +66,32 @@ func (m *Manager) driveLabelForRoot(root string) string {
|
||||
return ""
|
||||
}
|
||||
|
||||
// unitNewestArtifact is the unit's data time: the newest of its manifest, .sql dumps and .tar
|
||||
// volume dumps. ONE rule, shared with ListRestorePoints, so the two lists cannot date a unit
|
||||
// differently.
|
||||
// unitNewestArtifact is the unit's DATA time. ONE rule, shared by ListRestorePoints (Tier 1), the Tier-2
|
||||
// copy's date, the removed-app list and kept data, so no two of them can date a unit differently.
|
||||
//
|
||||
// v0.275.0 (R-696) — THE TIME OF THE DATA, NEVER OF THE MANIFEST. It used to be the newest of the
|
||||
// manifest, the .sql dumps and the .tar dumps; a refresh rewrites the manifest when the app's pins move,
|
||||
// so a unit re-captured two minutes after an update read as two minutes old over data from before the
|
||||
// update (9202 2026-09-25 11:06; demo-hp 2026-09-26 02:20, where it released the kept pre-conversion
|
||||
// copy). Now: the manifest's `data.at` when the data is stamped; else the newest DATA file (the undo
|
||||
// copies `pre-restore-*` excluded — an update's own safety dump is not a backup); the manifest's time
|
||||
// only for a unit that holds no data file at all, whose whole content is its definition.
|
||||
func unitNewestArtifact(unitDir string) (time.Time, bool) {
|
||||
fi, err := os.Stat(UnitManifestFile(unitDir))
|
||||
if err != nil {
|
||||
return time.Time{}, false
|
||||
}
|
||||
newest := fi.ModTime()
|
||||
newest = newestArtifact(UnitDBDumpDir(unitDir), ".sql", newest)
|
||||
newest = newestArtifact(UnitVolumeDumpDir(unitDir), ".tar", newest)
|
||||
if man := readManifest(UnitManifestFile(unitDir)); man != nil {
|
||||
if t, ok := man.Data.DataTime(); ok {
|
||||
return t, true
|
||||
}
|
||||
}
|
||||
var newest time.Time
|
||||
newest = newestDataFile(UnitDBDumpDir(unitDir), ".sql", newest)
|
||||
newest = newestDataFile(UnitVolumeDumpDir(unitDir), ".tar", newest)
|
||||
if newest.IsZero() {
|
||||
return fi.ModTime(), true
|
||||
}
|
||||
return newest, true
|
||||
}
|
||||
|
||||
|
||||
@@ -31,9 +31,7 @@ const restorePointShortID = "helyi"
|
||||
// known at all (found=false → the caller should 404). A known stack with no recovery unit on
|
||||
// disk returns an EMPTY list (a valid answer — "no backup yet"), not an error.
|
||||
//
|
||||
// The single point's Time is the newest mtime among the unit's artifacts (manifest.json,
|
||||
// db-dumps/*.sql, volume-dumps/*.tar): the manifest is only rewritten when the app's config
|
||||
// changes (checksum-skip), so the nightly-refreshed dumps are usually the freshest artifact.
|
||||
// The single point's Time is the unit's DATA time (unitNewestArtifact, v0.275.0 — R-696).
|
||||
func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, found bool) {
|
||||
if m.stackProvider == nil {
|
||||
return nil, false
|
||||
@@ -61,13 +59,11 @@ func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, fo
|
||||
return []RestorePoint{}, true
|
||||
}
|
||||
|
||||
fi, err := os.Stat(RecoveryUnitManifestPath(nsRoot, stackName))
|
||||
if err != nil {
|
||||
// v0.275.0 (R-696): the unit's DATA time (unitNewestArtifact), never the manifest's refresh time.
|
||||
newest, ok := unitNewestArtifact(RecoveryUnitPath(nsRoot, stackName))
|
||||
if !ok {
|
||||
return []RestorePoint{}, true // no recovery unit yet — "no backup" is a valid answer
|
||||
}
|
||||
newest := fi.ModTime()
|
||||
newest = newestArtifact(AppDBDumpPath(nsRoot, stackName), ".sql", newest)
|
||||
newest = newestArtifact(AppVolumeDumpPath(nsRoot, stackName), ".tar", newest)
|
||||
|
||||
return []RestorePoint{{
|
||||
Time: newest.UTC().Format(time.RFC3339),
|
||||
@@ -77,15 +73,16 @@ func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, fo
|
||||
}}, true
|
||||
}
|
||||
|
||||
// newestArtifact returns the newest mtime among cur and the files with the given extension in
|
||||
// dir (non-recursive; a missing dir contributes nothing).
|
||||
func newestArtifact(dir, ext string, cur time.Time) time.Time {
|
||||
// newestDataFile returns the newest mtime among cur and the DATA files with the given extension in dir
|
||||
// (non-recursive; a missing dir contributes nothing). The undo copies (`pre-restore-*`) are not data of
|
||||
// the unit (R-361) and never date it.
|
||||
func newestDataFile(dir, ext string, cur time.Time) time.Time {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return cur
|
||||
}
|
||||
for _, e := range entries {
|
||||
if e.IsDir() || !strings.HasSuffix(e.Name(), ext) {
|
||||
if e.IsDir() || !strings.HasSuffix(e.Name(), ext) || strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
|
||||
continue
|
||||
}
|
||||
if info, err := e.Info(); err == nil && info.ModTime().After(cur) {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||||
"os"
|
||||
@@ -166,6 +167,40 @@ type UnitRestoreResult struct {
|
||||
// statement, and precisely the R-88 failure direction (degrade to NO DATA rather than to UNKNOWN)
|
||||
// this whole change exists to remove. An unknown must be carried, never drawn as a zero.
|
||||
CountsUnknown bool
|
||||
// DataPins / DataAt (v0.275.0, R-696) — the versions the restored DATA belongs to and when it was
|
||||
// written, from the unit's `data` block; the definition started with it is the one they name. Empty
|
||||
// when the unit's versions are unknown (VersionsUnknown).
|
||||
DataPins []string
|
||||
DataAt time.Time
|
||||
// VersionChanged — the app now runs DataPins, which differ from what it ran before the restore (a
|
||||
// restore of an older version). The page then says which version came back (`07` §6.6).
|
||||
VersionChanged bool
|
||||
// VersionsUnknown — a unit written before v0.275.0 (or with an unstamped data file): restored as
|
||||
// before, with a WARN naming it.
|
||||
VersionsUnknown bool
|
||||
}
|
||||
|
||||
// ErrUnitVersionMismatch is the KIND of the refusal of a restore whose unit would start its data with a
|
||||
// definition the data does not belong to (v0.275.0, R-696): a unit whose files were written under different
|
||||
// pins (Mixed), or whose compose/ names other pins than its data. Raised BEFORE anything is touched;
|
||||
// branch with errors.Is. The customer sentence is the bundle's.
|
||||
var ErrUnitVersionMismatch = errors.New("the recovery unit's definition is not the one its data belongs to")
|
||||
|
||||
// unitVersionCheck is A3's rule, as a pure function of the manifest and the unit's own compose file:
|
||||
// known and matching → the pins to report; unknown → (nil, nil) and the caller WARNs; mixed or
|
||||
// mismatched → the refusal.
|
||||
func unitVersionCheck(stackName string, man *RecoveryManifest, composeDir string) ([]string, error) {
|
||||
if man == nil || man.Data == nil {
|
||||
return nil, nil
|
||||
}
|
||||
if man.Data.Mixed {
|
||||
return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_versions_mixed", stackName)
|
||||
}
|
||||
def := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
|
||||
if !samePins(def, man.Data.ImagePins) {
|
||||
return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_version_mismatch", stackName, PinsVersion(def), PinsVersion(man.Data.ImagePins))
|
||||
}
|
||||
return man.Data.ImagePins, nil
|
||||
}
|
||||
|
||||
// RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit.
|
||||
@@ -323,60 +358,35 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
|
||||
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
|
||||
|
||||
composeDir := UnitComposeDir(unitDir)
|
||||
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
|
||||
|
||||
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
|
||||
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
|
||||
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
|
||||
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
|
||||
// reconcileRestoreSecrets), then the fail-closed gate.
|
||||
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
|
||||
fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
|
||||
return res, err
|
||||
// v0.275.0 (R-696, `07` §6.6) — A RESTORE NEVER STARTS DATA WITH A DEFINITION IT DOES NOT BELONG TO.
|
||||
// The unit keeps the definition of its data (data_versions.go); this refuses, before anything is
|
||||
// touched, a unit where the two disagree. Measured before the fix (A1, 9202): the unit's PostgreSQL 18
|
||||
// definition over a 16 datadir tar — the tar replaced the live volume, 18 refused it, the app stayed down.
|
||||
dataPins, verr := unitVersionCheck(stackName, manifest, composeDir)
|
||||
if verr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: the unit's definition %v, its data %v (mixed=%v) — a restore never starts data with a definition it does not belong to; nothing was touched",
|
||||
stackName, ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")), manifest.Data.ImagePins, manifest.Data.Mixed)
|
||||
return res, verr
|
||||
}
|
||||
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
|
||||
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
|
||||
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
|
||||
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
|
||||
// never generated: the fail-closed gate above already refused if one was missing, and the
|
||||
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
|
||||
//
|
||||
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
|
||||
// generator run means the secret was empty at capture AND absent from the guest.
|
||||
//
|
||||
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
|
||||
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
|
||||
// regenerated value leaves the app unable to authenticate against its own restored rows while the
|
||||
// dump replay, which uses the container's local trust socket, still reports success. The old wording
|
||||
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
|
||||
// DB/root-password fields and is now scoped to what is actually true.
|
||||
if len(missing) > 0 {
|
||||
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
|
||||
for _, dk := range manifest.DataKeyEnvVars {
|
||||
dataKeySet[dk] = true
|
||||
}
|
||||
var generated, unresolved []string
|
||||
for _, name := range missing {
|
||||
if !dataKeySet[name] && m.generateSecret != nil {
|
||||
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
|
||||
fullEnv[name] = v
|
||||
generated = append(generated, name)
|
||||
continue
|
||||
}
|
||||
if dataPins == nil {
|
||||
res.VersionsUnknown = true
|
||||
m.logger.Printf("[WARN] [backup] Restore %s from %s: the unit does not record which versions wrote its data (written before v0.275.0, or an unstamped file) — restoring its definition as captured, as before", stackName, unitDir)
|
||||
} else {
|
||||
res.DataPins = dataPins
|
||||
res.DataAt, _ = manifest.Data.DataTime()
|
||||
if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok {
|
||||
if live := definitionPins(info); len(live) > 0 && !samePins(live, dataPins) {
|
||||
res.VersionChanged = true
|
||||
m.logger.Printf("[INFO] [backup] Restore %s: the data belongs to %v (written %s); the app ran %v — it comes back at its data's version, and the update climbs from there",
|
||||
stackName, dataPins, manifest.Data.At, live)
|
||||
}
|
||||
unresolved = append(unresolved, name)
|
||||
}
|
||||
if len(generated) > 0 {
|
||||
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
|
||||
stackName, generated)
|
||||
}
|
||||
if len(unresolved) > 0 {
|
||||
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
|
||||
stackName, len(unresolved), unresolved)
|
||||
}
|
||||
}
|
||||
fullEnv, missing, err := m.unitRestoreEnv(stackName, composeDir, manifest)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
// R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with
|
||||
// two answers, and "an absent log line is not evidence" — the drill reads this line to prove the
|
||||
// secondary mirror, not the primary unit, was the source. It is a path, never a secret.
|
||||
@@ -472,3 +482,66 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
|
||||
m.clearUpdateHoldAfterRestore(stackName)
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// unitRestoreEnv rebuilds the env a restore starts the unit's definition with: the unit's plain config,
|
||||
// its portable secrets (D5), the guest's for the withheld class, the fail-closed data-key gate, and a
|
||||
// generated replacement for a missing RESETTABLE secret (O4). ONE implementation for the unit restore and,
|
||||
// since v0.275.0, the off-site restore that brings an app back at its snapshot's version. Values are
|
||||
// never logged.
|
||||
func (m *Manager) unitRestoreEnv(stackName, composeDir string, manifest *RecoveryManifest) (fullEnv map[string]string, missing []string, err error) {
|
||||
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
|
||||
|
||||
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
|
||||
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
|
||||
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
|
||||
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
|
||||
// reconcileRestoreSecrets), then the fail-closed gate.
|
||||
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
|
||||
fullEnv, missing, err = reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
|
||||
if err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
|
||||
return nil, missing, err
|
||||
}
|
||||
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
|
||||
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
|
||||
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
|
||||
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
|
||||
// never generated: the fail-closed gate above already refused if one was missing, and the
|
||||
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
|
||||
//
|
||||
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
|
||||
// generator run means the secret was empty at capture AND absent from the guest.
|
||||
//
|
||||
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
|
||||
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
|
||||
// regenerated value leaves the app unable to authenticate against its own restored rows while the
|
||||
// dump replay, which uses the container's local trust socket, still reports success. The old wording
|
||||
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
|
||||
// DB/root-password fields and is now scoped to what is actually true.
|
||||
if len(missing) > 0 {
|
||||
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
|
||||
for _, dk := range manifest.DataKeyEnvVars {
|
||||
dataKeySet[dk] = true
|
||||
}
|
||||
var generated, unresolved []string
|
||||
for _, name := range missing {
|
||||
if !dataKeySet[name] && m.generateSecret != nil {
|
||||
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
|
||||
fullEnv[name] = v
|
||||
generated = append(generated, name)
|
||||
continue
|
||||
}
|
||||
}
|
||||
unresolved = append(unresolved, name)
|
||||
}
|
||||
if len(generated) > 0 {
|
||||
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
|
||||
stackName, generated)
|
||||
}
|
||||
if len(unresolved) > 0 {
|
||||
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
|
||||
stackName, len(unresolved), unresolved)
|
||||
}
|
||||
}
|
||||
return fullEnv, missing, nil
|
||||
}
|
||||
|
||||
@@ -24,7 +24,7 @@ func (h *admissionHarness) digestOf(kind string) *RunSummary {
|
||||
doneAdm := h.m.beginAdmissionRun()
|
||||
doneSum := h.m.beginRunSummary(kind, "run-test")
|
||||
h.m.runVolumeDumps()
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
h.m.emitRunSummary()
|
||||
doneSum()
|
||||
doneAdm()
|
||||
@@ -153,7 +153,7 @@ func TestRunSummary_RefreshSweepHasNoRunID(t *testing.T) {
|
||||
defer h.m.beginAdmissionRun()()
|
||||
defer h.m.beginRunSummary(runKindRefresh, "")()
|
||||
defer h.m.emitRunSummary()
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
}()
|
||||
if got == nil {
|
||||
t.Fatal("the periodic sweep emitted no digest — with the per-app event now record-only, a " +
|
||||
|
||||
@@ -148,7 +148,7 @@ func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) {
|
||||
if len(h.volDumped) != 1 || h.volDumped[0] != "free" {
|
||||
t.Errorf("positive control: the unheld app must still be dumped, got %v", h.volDumped)
|
||||
}
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
for _, n := range h.prov.infoHits {
|
||||
if n == "held" {
|
||||
t.Error("the capture must not rewrite a HELD app's restore point")
|
||||
@@ -178,7 +178,7 @@ func TestSlice4_NightlyLegsLeaveAnAppMidUpdateAlone(t *testing.T) {
|
||||
h.m.settings = slice4Settings(t)
|
||||
h.m.SetUpdatingCheck(func(name string) bool { return name == "updating" })
|
||||
h.m.runVolumeDumps()
|
||||
h.m.captureAllRecoveryUnits()
|
||||
h.m.captureAllRecoveryUnits(false)
|
||||
var mirrored []string
|
||||
h.m.perAppTier2 = func(name string) error { mirrored = append(mirrored, name); return nil }
|
||||
h.m.RunAllTier2()
|
||||
|
||||
@@ -43,6 +43,9 @@ type Tier2RestorePoint struct {
|
||||
PackagePreserved bool
|
||||
// CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded.
|
||||
CopyLastSuccess string
|
||||
// DataDate (v0.275.0, R-696) — the mirrored unit's DATA time (unitNewestArtifact on the mirror): when
|
||||
// the data the copy holds was written, which a mirror run copies but never makes newer. "" = unknown.
|
||||
DataDate string
|
||||
}
|
||||
|
||||
// restorePointFromCoverage is the pure half of the predicate.
|
||||
@@ -54,6 +57,7 @@ func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint {
|
||||
CopyDateProven: cov.CopyLastSuccess != "",
|
||||
PackagePreserved: preserved,
|
||||
CopyLastSuccess: cov.CopyLastSuccess,
|
||||
DataDate: cov.UnitDataDate,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -93,6 +97,12 @@ func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
|
||||
if err != nil {
|
||||
return time.Time{}, false
|
||||
}
|
||||
// v0.275.0 (R-696): a mirror run copies the unit's data; it never makes the data newer. When the
|
||||
// mirror's data time is known and older than the copy, the data time is the copy's age — a mirror
|
||||
// taken right after an update, of a unit whose dump is from before it, is as old as that dump.
|
||||
if d, derr := time.Parse(time.RFC3339, p.DataDate); derr == nil && d.Before(t) {
|
||||
return d, true
|
||||
}
|
||||
return t, true
|
||||
}
|
||||
|
||||
@@ -209,8 +219,9 @@ var updateOffsiteCheckTimeout = 15 * time.Second
|
||||
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
|
||||
type UpdateTierPoint struct {
|
||||
Tier int
|
||||
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
|
||||
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
|
||||
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime (capped by the
|
||||
// mirror's data time), Tier 1 the unit's DATA time (ListRestorePoints → unitNewestArtifact), Tier 3
|
||||
// the newest snapshot, capped by the data time the box recorded when it pushed it (v0.275.0, R-696).
|
||||
At time.Time
|
||||
}
|
||||
|
||||
@@ -281,7 +292,7 @@ func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier in
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
if at, ok := got[stackName]; ok && !at.IsZero() {
|
||||
return UpdateTierPoint{Tier: tier, At: at}, true
|
||||
return UpdateTierPoint{Tier: tier, At: m.offsiteDataTime(stackName, at)}, true
|
||||
}
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
@@ -391,11 +402,12 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
|
||||
if db.StackName != stackName {
|
||||
continue
|
||||
}
|
||||
res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug())
|
||||
res := m.dumpOneOrDefault(ctx, db, AppDBDumpPath(nsRoot, stackName))
|
||||
if res.Error != nil {
|
||||
return util.MsgError("err.backup.adatbazis_mentes_sikertelen", db.ContainerName, res.Error)
|
||||
}
|
||||
dumped++
|
||||
m.stampDataFile(stackName, RecoveryUnitPath(nsRoot, stackName), "db-dumps/"+filepath.Base(res.FilePath))
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size))
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user