v0.275.0: a backup's data and its version travel together (R-696, 07 §6.6, D4 option A); R-695, R-691, R-694
gates / gates (push) Successful in 23s

The unit's data files are stamped with the versions that wrote them; the capture keeps the
definition the data belongs to; a restore never starts data under another version's
definition (unit restores refuse a mismatch; the off-site restore writes the snapshot's
definition); every tier's time is its data's; the conversion-copy release needs a dump on
the new engine. File-browser sync single-flight + no empty kept folder (R-695); the kept
view joins the folder's owning group, language switch resyncs (R-691); a restore-generated
login is not shown as the password (R-694). Red-proofs in
felhom.eu/documentation/audits/version-travel-2026-09-26/.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-26 10:35:22 +02:00
parent fb2bcdd5d6
commit b6810f14ff
47 changed files with 3771 additions and 135 deletions
@@ -0,0 +1,576 @@
package backup
import (
"context"
"encoding/json"
"errors"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// Part A of the version-travel brief (controller v0.275.0, R-696, `07` §6.6): a backup's data and its
// version travel together. Measured before the fix on 9202 (`audits/version-travel-2026-09-26/A1/`): the
// periodic refresh re-captured the unit's DEFINITION two minutes after an update, over the previous
// version's DATA; Tier 1's time moved to the refresh; a restore in that window started a PostgreSQL 16
// datadir under the 18 definition and left the app down.
//
// Every test asserts a CONSEQUENCE a household or the update would see — which definition the restore
// starts, which time the precondition reads, which copy the release trusts — not the mechanism.
// vtStack is one app's stack dir + a provider whose ImagePins are what production derives them from:
// ParseComposeImages of the stack's compose (the adapter's GetStackRecoveryInfo does exactly that).
type vtStack struct {
t *testing.T
tmp string
drive string
stackDir string
fake *fakeRecoveryProvider
m *Manager
}
func vtCompose(pgMajor string) string {
return "services:\n app:\n image: example/app:0.96.0\n app-db:\n image: postgres:" + pgMajor + "-alpine\n"
}
func newVTStack(t *testing.T, pgMajor string) *vtStack {
t.Helper()
tmp := t.TempDir()
v := &vtStack{t: t, tmp: tmp, drive: filepath.Join(tmp, "drive"), stackDir: filepath.Join(tmp, "stack")}
if err := os.MkdirAll(v.stackDir, 0o755); err != nil {
t.Fatal(err)
}
v.fake = &fakeRecoveryProvider{hdd: v.drive, running: true}
v.m = &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(tmp, "system"),
stackProvider: v.fake,
version: "vtest",
}
v.setVersion(pgMajor)
return v
}
// setVersion is what an update does to the stack dir: the definition moves, and what runs moves with it.
func (v *vtStack) setVersion(pgMajor string) {
v.t.Helper()
mustWrite(v.t, filepath.Join(v.stackDir, "docker-compose.yml"), vtCompose(pgMajor))
mustWrite(v.t, filepath.Join(v.stackDir, ".felhom.yml"), "display_name: App "+pgMajor+"\n")
mustWrite(v.t, filepath.Join(v.stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: vt\n")
v.fake.info = RecoveryInfo{
StackDir: v.stackDir,
DisplayName: "App",
ImagePins: ParseComposeImages(filepath.Join(v.stackDir, "docker-compose.yml")),
NonSecretEnv: map[string]string{"SUBDOMAIN": "vt"},
InstalledImages: map[string]string{
"app": "example/app:0.96.0@sha256:aaa",
"app-db": "postgres:" + pgMajor + "-alpine@sha256:" + pgMajor + pgMajor,
},
}
}
func (v *vtStack) unitDir() string { return RecoveryUnitPath(v.drive, "app") }
// backupLegs is what a data run writes, through the SAME stamp call the legs make: a database dump and
// a volume tar, both dated `at` (so "the refresh ran later" is a fact of the clock, not of the test).
func (v *vtStack) backupLegs(at time.Time, marker string) {
v.t.Helper()
sql := filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql")
tar := filepath.Join(UnitVolumeDumpDir(v.unitDir()), "app_db.tar")
mustWrite(v.t, sql, pgDump(1)+"-- "+marker+"\n")
mustWrite(v.t, tar, "tar:"+marker)
for _, p := range []string{sql, tar} {
if err := os.Chtimes(p, at, at); err != nil {
v.t.Fatal(err)
}
}
v.m.stampDataFile("app", v.unitDir(), "db-dumps/app-postgres.sql")
v.m.stampDataFile("app", v.unitDir(), "volume-dumps/app_db.tar")
}
func (v *vtStack) manifest() *RecoveryManifest {
v.t.Helper()
man := readManifest(UnitManifestFile(v.unitDir()))
if man == nil {
v.t.Fatal("no readable manifest")
}
return man
}
func (v *vtStack) unitComposeImages() []string {
return ParseComposeImages(filepath.Join(UnitComposeDir(v.unitDir()), "docker-compose.yml"))
}
// theWindow builds the measured A1 state: a data run at PostgreSQL 16 two hours ago, then the update to
// 18, then the periodic refresh NOW.
func theWindow(t *testing.T) (*vtStack, time.Time) {
t.Helper()
v := newVTStack(t, "16")
dataAt := time.Now().Add(-2 * time.Hour).UTC().Truncate(time.Second)
v.backupLegs(dataAt, "written-by-16")
if err := v.m.CaptureRecoveryUnit("app"); err != nil { // the data run's capture
t.Fatal(err)
}
v.setVersion("18") // the guarded update moved the pin
if err := v.m.captureRecoveryUnit("app", false); err != nil { // the 5-minute refresh
t.Fatal(err)
}
return v, dataAt
}
// A5 red-proof 1 — the manifest refresh after a pin change no longer moves Tier 1's time.
// Pre-fix (v0.274.0): ListRestorePoints = newest of the manifest's and the dumps' mtimes → "now".
func TestA5_RefreshAfterAPinChangeDoesNotMoveTier1sTime(t *testing.T) {
v, dataAt := theWindow(t)
pts, found := v.m.ListRestorePoints("app")
if !found || len(pts) != 1 {
t.Fatalf("restore points = %v found=%v, want one", pts, found)
}
got, err := time.Parse(time.RFC3339, pts[0].Time)
if err != nil {
t.Fatal(err)
}
if !got.Equal(dataAt) {
t.Fatalf("Tier 1's time = %s, want the DATA's time %s — the refresh %s made a two-hour-old dump read as new",
got.Format(time.RFC3339), dataAt.Format(time.RFC3339), time.Since(got).Round(time.Second))
}
}
// The definition stays with its data: after the refresh, compose/ still names the 16 definition, the
// manifest says both (image_pins = what runs, data.image_pins = what wrote the data).
// Pre-fix: the refresh rewrote compose/ to 18 (A1 S3b).
func TestA2_TheUnitKeepsTheDefinitionItsDataBelongsTo(t *testing.T) {
v, _ := theWindow(t)
if got := v.unitComposeImages(); !samePins(got, ParseComposeImages(writeTmpCompose(t, vtCompose("16")))) {
t.Fatalf("the unit's compose/ names %v — the refresh paired the 18 definition with the 16 data", got)
}
man := v.manifest()
if man.Data == nil || !strings.Contains(strings.Join(man.Data.ImagePins, " "), "postgres:16-alpine") {
t.Fatalf("manifest data = %+v, want the 16 pins", man.Data)
}
if !strings.Contains(strings.Join(man.ImagePins, " "), "postgres:18-alpine") {
t.Fatalf("manifest image_pins = %v, want the app's CURRENT (18) pins", man.ImagePins)
}
if got := man.Data.Files["db-dumps/app-postgres.sql"].Images["app-db"]; !strings.HasPrefix(got, "postgres:16-alpine@sha256:") {
t.Fatalf("the dump's recorded engine = %q, want postgres:16 with its digest", got)
}
// And a SECOND refresh writes nothing: the frozen checksums describe what compose/ holds.
before, _ := os.Stat(UnitManifestFile(v.unitDir()))
time.Sleep(20 * time.Millisecond)
if err := v.m.captureRecoveryUnit("app", false); err != nil {
t.Fatal(err)
}
after, _ := os.Stat(UnitManifestFile(v.unitDir()))
if !after.ModTime().Equal(before.ModTime()) {
t.Fatal("a second refresh rewrote the manifest — the frozen unit thrashes the drive every five minutes")
}
}
func writeTmpCompose(t *testing.T, body string) string {
t.Helper()
p := filepath.Join(t.TempDir(), "docker-compose.yml")
mustWrite(t, p, body)
return p
}
// The next data run replaces the data AND the definition: both are 18 afterwards.
func TestA2_TheNextDataRunMovesDataAndDefinitionTogether(t *testing.T) {
v, _ := theWindow(t)
now := time.Now().UTC().Truncate(time.Second)
v.backupLegs(now, "written-by-18")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
if got := strings.Join(v.unitComposeImages(), " "); !strings.Contains(got, "postgres:18-alpine") {
t.Fatalf("after a data run on 18 the unit's compose/ names %s, want 18", got)
}
if man := v.manifest(); man.Data == nil || man.Data.At != now.Format(time.RFC3339) || man.Data.Mixed {
t.Fatalf("data = %+v, want at %s, not mixed", man.Data, now.Format(time.RFC3339))
}
}
// vtRestorer records the definition the restore STARTS the data with.
type vtRestorer struct {
*fakeRecoveryProvider
startedWith []string
}
func (f *vtRestorer) RecreateStackDefinitionFromUnit(name, composeDir string, env map[string]string) error {
f.startedWith = ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, env)
}
func vtRestoreSeams(m *Manager) (vols *[]string) {
var vd []string
m.volumeReplayFrom = func(_, dir string) (int, error) { vd = append(vd, dir); return 1, nil }
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
}
m.importDBDump = func(context.Context, DiscoveredDB, string) error { return nil }
return &vd
}
// A5 red-proof 3 — a restore in the window starts the OLD version with the OLD data, and says so.
// Pre-fix: the unit's definition was 18 (the refresh), so the 16 datadir was started under 18 (A1 S4).
func TestA5_ARestoreInTheWindowStartsTheOldVersionWithTheOldData(t *testing.T) {
v, dataAt := theWindow(t)
rec := &vtRestorer{fakeRecoveryProvider: v.fake}
v.m.stackProvider = rec
vtRestoreSeams(v.m)
res, err := v.m.RestoreFromRecoveryUnit("app")
if err != nil {
t.Fatalf("restore: %v", err)
}
if got := strings.Join(rec.startedWith, " "); !strings.Contains(got, "postgres:16-alpine") || strings.Contains(got, "postgres:18") {
t.Fatalf("the restore started the 16 data with the definition %s — a restore must never mix versions", got)
}
if !res.VersionChanged || !samePins(res.DataPins, ParseComposeImages(writeTmpCompose(t, vtCompose("16")))) || !res.DataAt.Equal(dataAt) {
t.Fatalf("result = changed %v pins %v at %s, want changed, the 16 pins, %s", res.VersionChanged, res.DataPins, res.DataAt, dataAt)
}
}
// A restore never starts data with a definition it does not belong to: a unit whose compose/ names other
// pins than its data, and a unit whose files were written by different versions, are refused BEFORE
// anything is touched (no stop, no volume, no recreate).
func TestA3_AMismatchedOrMixedUnitIsRefusedBeforeAnythingMoves(t *testing.T) {
t.Run("definition-not-the-data's", func(t *testing.T) {
v, _ := theWindow(t)
// What v0.274.0's refresh left behind: the NEW definition in compose/ over the old data.
mustWrite(t, filepath.Join(UnitComposeDir(v.unitDir()), "docker-compose.yml"), vtCompose("18"))
vols := vtRestoreSeams(v.m)
_, err := v.m.RestoreFromRecoveryUnit("app")
if !errors.Is(err, ErrUnitVersionMismatch) {
t.Fatalf("err = %v, want ErrUnitVersionMismatch", err)
}
if len(v.fake.calls) != 0 || len(*vols) != 0 {
t.Fatalf("the refusal touched the app: calls=%v volumes=%v", v.fake.calls, *vols)
}
})
t.Run("mixed", func(t *testing.T) {
v, _ := theWindow(t)
// The DB leg ran under 18, the volume leg failed and kept its 16 tar.
sql := filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql")
mustWrite(t, sql, pgDump(1))
v.m.stampDataFile("app", v.unitDir(), "db-dumps/app-postgres.sql")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
if man := v.manifest(); man.Data == nil || !man.Data.Mixed {
t.Fatalf("data = %+v, want Mixed", man.Data)
}
vols := vtRestoreSeams(v.m)
_, err := v.m.RestoreFromRecoveryUnit("app")
if !errors.Is(err, ErrUnitVersionMismatch) {
t.Fatalf("err = %v, want ErrUnitVersionMismatch", err)
}
if len(v.fake.calls) != 0 || len(*vols) != 0 {
t.Fatalf("the refusal touched the app: calls=%v volumes=%v", v.fake.calls, *vols)
}
})
}
// An older unit (no stamps) restores as before: its definition, VersionsUnknown, no refusal.
func TestA3_AnUnstampedUnitRestoresAsBefore(t *testing.T) {
v := newVTStack(t, "16")
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), pgDump(1))
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
if man := v.manifest(); man.Data != nil {
t.Fatalf("an unstamped dump produced data %+v — unknown must stay unknown", man.Data)
}
rec := &vtRestorer{fakeRecoveryProvider: v.fake}
v.m.stackProvider = rec
vtRestoreSeams(v.m)
res, err := v.m.RestoreFromRecoveryUnit("app")
if err != nil {
t.Fatalf("restore: %v", err)
}
if !res.VersionsUnknown || res.VersionChanged || len(rec.startedWith) == 0 {
t.Fatalf("result = %+v started=%v, want VersionsUnknown and the unit's definition started", res, rec.startedWith)
}
}
// A5 red-proof 4 — the update's precondition refuses a stale dump that a refresh made look new.
// Pre-fix: Tier 1 = the refresh's manifest time → "0m old" → accepted, and the update leaned on a copy
// of the previous version's data from two hours before.
func TestA5_ThePreconditionRefusesAStaleDumpARefreshMadeLookNew(t *testing.T) {
v, dataAt := theWindow(t)
now := time.Now()
fresh := func(p UpdateTierPoint) bool { return now.Sub(p.At) <= time.Hour }
p, ok, seen := v.m.UpdateRestorePoints(context.Background(), "app", fresh)
if ok {
t.Fatalf("the precondition ACCEPTED tier %d at %s (%s old) — the data is from %s",
p.Tier, p.At.Format(time.RFC3339), now.Sub(p.At).Round(time.Second), dataAt.Format(time.RFC3339))
}
if len(seen) != 1 || seen[0].Tier != UpdateTierLocal || !seen[0].At.Equal(dataAt) {
t.Fatalf("seen = %+v, want Tier 1 at the data's time %s", seen, dataAt.Format(time.RFC3339))
}
}
// A unit with no data file at all is dated by the data run that confirmed it, and a refresh keeps it.
func TestA4_AUnitWithoutDataFilesIsDatedByItsDataRun(t *testing.T) {
v := newVTStack(t, "16")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
first := v.manifest().Data
if first == nil || first.At == "" {
t.Fatalf("data = %+v, want the data run's time", first)
}
v.setVersion("18")
if err := v.m.captureRecoveryUnit("app", false); err != nil {
t.Fatal(err)
}
if got := v.manifest().Data; got == nil || got.At != first.At {
t.Fatalf("a refresh moved the data time: %+v → %+v", first, got)
}
}
// The undo copies are not data: an update's safety dump written after the backup must not date Tier 1.
func TestA4_AnUndoCopyNeverDatesTheUnit(t *testing.T) {
v := newVTStack(t, "16")
old := time.Now().Add(-3 * time.Hour).UTC().Truncate(time.Second)
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), pgDump(1))
_ = os.Chtimes(filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), old, old)
if err := v.m.captureRecoveryUnit("app", false); err != nil { // unstamped: the legacy rule
t.Fatal(err)
}
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), preRestoreDumpPrefix+"20260926T000000Z-app-postgres.sql"), pgDump(1))
pts, _ := v.m.ListRestorePoints("app")
if len(pts) != 1 || pts[0].Time != old.Format(time.RFC3339) {
t.Fatalf("Tier 1 = %+v, want the dump's %s (not the undo copy's, not the manifest's)", pts, old.Format(time.RFC3339))
}
}
// The stamps are written by the PRODUCTION leg (seam discipline): RunAppBackupNow → the dump seam → the
// stamp → the capture's `data`. A stamp helper nobody calls is the "seam built but never wired" shape.
func TestA2_TheUpdatesOwnBackupStampsItsDataThroughTheRealLeg(t *testing.T) {
v := newVTStack(t, "16")
v.m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
}
v.m.dumpOne = func(_ context.Context, db DiscoveredDB, dir string, _ *log.Logger, _ bool) DumpResult {
p := filepath.Join(dir, "app-postgres.sql")
mustWrite(t, p, pgDump(1))
return DumpResult{DB: db, FilePath: p}
}
v.m.perAppTier2 = func(string) error { return nil }
if err := v.m.RunAppBackupNow(context.Background(), "app"); err != nil {
t.Fatalf("RunAppBackupNow: %v", err)
}
man := v.manifest()
if man.Data == nil || man.Data.Mixed || len(man.Data.Files) != 1 {
t.Fatalf("data = %+v, want one stamped dump", man.Data)
}
st := man.Data.Files["db-dumps/app-postgres.sql"]
if st.Images["app-db"] != "postgres:16-alpine@sha256:1616" || !samePins(st.Pins, v.fake.info.ImagePins) {
t.Fatalf("stamp = %+v, want the running images and the definition's pins", st)
}
var raw map[string]interface{}
b, _ := os.ReadFile(UnitManifestFile(v.unitDir()))
_ = json.Unmarshal(b, &raw)
if _, ok := raw["data"]; !ok {
t.Fatal("manifest.json carries no `data` key")
}
}
// Tier 2: a mirror taken after the update of a unit whose data is older is as old as that data.
func TestA4_ATier2CopyIsAsOldAsItsData(t *testing.T) {
copyAt := time.Now().UTC().Truncate(time.Second)
dataAt := copyAt.Add(-2 * time.Hour)
p := Tier2RestorePoint{Restorable: true, CopyDateProven: true, CopyLastSuccess: copyAt.Format(time.RFC3339), DataDate: dataAt.Format(time.RFC3339)}
got, ok := p.ProvenCopyTime()
if !ok || !got.Equal(dataAt) {
t.Fatalf("Tier 2 proven at %s ok=%v, want the data's %s", got, ok, dataAt)
}
p.DataDate = ""
if got, _ := p.ProvenCopyTime(); !got.Equal(copyAt) {
t.Fatalf("with no data date, Tier 2 = %s, want the copy's %s (as before)", got, copyAt)
}
}
// The household's version label names every image — a PostgreSQL step is visible in it.
func TestA3_PinsVersionNamesEveryImage(t *testing.T) {
got := PinsVersion([]string{"docmost/docmost:0.96.0@sha256:b5", "postgres:16-alpine@sha256:72", "gitea.dooplex.hu/x/redis:7-alpine"})
if got != "docmost:0.96.0, postgres:16-alpine, redis:7-alpine" {
t.Fatalf("PinsVersion = %q", got)
}
}
// vtReconProvider is the reconstitution fixture's provider with a LIVE version (18) and a recorder for
// the definition the restore writes.
type vtReconProvider struct {
*recordingProvider
livePins []string
wroteDef []string
wroteAtCall int
}
func (p *vtReconProvider) GetStackRecoveryInfo(string) (RecoveryInfo, bool) {
return RecoveryInfo{DisplayName: "Immich", ImagePins: p.livePins}, true
}
func (p *vtReconProvider) RecreateStackDefinitionFromUnit(_, composeDir string, _ map[string]string) error {
p.wroteDef = ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
p.wroteAtCall = len(p.calls)
p.calls = append(p.calls, "recreate")
return nil
}
// vtSnapshotUnit gives the fixture's scratch unit a definition and a `data` block (a snapshot taken by
// v0.275.0), and returns the unit dir.
func vtSnapshotUnit(t *testing.T, m *Manager, pgMajor string, withData bool) string {
t.Helper()
scratch, _, err := m.offboxRestoreScratchDir("immich")
if err != nil {
t.Fatal(err)
}
var unit string
_ = filepath.Walk(scratch, func(p string, fi os.FileInfo, _ error) error {
if fi != nil && !fi.IsDir() && fi.Name() == "manifest.json" {
unit = filepath.Dir(p)
}
return nil
})
if unit == "" {
t.Fatal("no scratch unit")
}
mustWrite(t, filepath.Join(UnitComposeDir(unit), "docker-compose.yml"), vtCompose(pgMajor))
mustWrite(t, filepath.Join(UnitComposeDir(unit), "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: vt\n")
man := readManifest(UnitManifestFile(unit))
if withData {
man.Data = &UnitData{At: "2026-09-26T02:15:01Z", ImagePins: ParseComposeImages(filepath.Join(UnitComposeDir(unit), "docker-compose.yml"))}
}
if err := writeManifest(UnitManifestFile(unit), man); err != nil {
t.Fatal(err)
}
return unit
}
// Off-site (Tier 3): v0.274.0 never wrote the definition — last night's snapshot data went under the
// app's NEW definition (A1, read from source). Now the snapshot's own definition is written BEFORE any
// file, volume or database is touched, and the database service is resolved from it.
func TestA3_TheOffsiteRestoreBringsTheSnapshotsVersionBack(t *testing.T) {
m, prov, imported := reconFixture(t, "20260926T021500Z", "2026-09-26T02:15:01Z", pgDump(1))
vp := &vtReconProvider{recordingProvider: prov, livePins: ParseComposeImages(writeTmpCompose(t, vtCompose("18")))}
m.SetStackProvider(vp)
vtSnapshotUnit(t, m, "16", true)
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err != nil {
t.Fatalf("reconstitute: %v", err)
}
if got := strings.Join(vp.wroteDef, " "); !strings.Contains(got, "postgres:16-alpine") {
t.Fatalf("the restore wrote the definition %q — want the snapshot's 16", got)
}
if vp.wroteAtCall != 1 || vp.calls[0] != "stop" {
t.Fatalf("calls = %v — the definition must be written right after the stop, before any data", vp.calls)
}
if !res.VersionChanged || res.DataAt.Format(time.RFC3339) != "2026-09-26T02:15:01Z" || len(*imported) != 1 {
t.Fatalf("result changed=%v at=%s imported=%v", res.VersionChanged, res.DataAt, *imported)
}
if got := strings.Join(vp.gotServices, ","); got != "app-db" {
t.Fatalf("the DB-only start was %q — the database service must come from the definition that RUNS (the snapshot's)", got)
}
}
// Same version, or a snapshot from before v0.275.0: nothing is written — the path of every earlier release.
func TestA3_TheOffsiteRestoreLeavesTheDefinitionWhenNothingDiffers(t *testing.T) {
for _, c := range []struct {
name string
live string
withData bool
unknown bool
}{{"same-version", "16", true, false}, {"pre-v0.275.0-snapshot", "18", false, true}} {
t.Run(c.name, func(t *testing.T) {
m, prov, _ := reconFixture(t, "20260926T021500Z", "2026-09-26T02:15:01Z", pgDump(1))
vp := &vtReconProvider{recordingProvider: prov, livePins: ParseComposeImages(writeTmpCompose(t, vtCompose(c.live)))}
m.SetStackProvider(vp)
vtSnapshotUnit(t, m, "16", c.withData)
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err != nil {
t.Fatalf("reconstitute: %v", err)
}
if vp.wroteDef != nil || res.VersionChanged || res.VersionsUnknown != c.unknown {
t.Fatalf("wrote %v changed=%v unknown=%v", vp.wroteDef, res.VersionChanged, res.VersionsUnknown)
}
})
}
}
// Tier 3: a snapshot pushed after a failed dump leg carries older data than its own time — the recorded
// push caps it; a snapshot newer than the last recorded push (another box) keeps its own time.
func TestA4_AnOffsiteCopyIsAsOldAsTheDataItWasPushedWith(t *testing.T) {
m, sett := newOffboxManager(t)
snap := time.Date(2026, 9, 26, 2, 15, 0, 0, time.UTC)
if got := m.offsiteDataTime("app", snap); !got.Equal(snap) {
t.Fatalf("no record: %s, want the snapshot's own time", got)
}
data := snap.Add(-24 * time.Hour)
if err := sett.SetOffsiteDataAt("app", settings.OffsiteDataRecord{PushedAt: snap.Add(time.Minute).Format(time.RFC3339), DataAt: data.Format(time.RFC3339)}); err != nil {
t.Fatal(err)
}
if got := m.offsiteDataTime("app", snap); !got.Equal(data) {
t.Fatalf("recorded push: %s, want the data's %s", got, data)
}
later := snap.Add(2 * time.Hour)
if got := m.offsiteDataTime("app", later); !got.Equal(later) {
t.Fatalf("a snapshot newer than the recorded push: %s, want its own %s", got, later)
}
}
// The push records the pushed unit's DATA time (the production call in runOffboxInternal).
func TestA4_ThePushRecordsTheUnitsDataTime(t *testing.T) {
m, sett := newOffboxManager(t)
v := newVTStack(t, "16")
dataAt := time.Now().Add(-26 * time.Hour).UTC().Truncate(time.Second)
v.backupLegs(dataAt, "x")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
m.recordOffsiteDataAt("app", v.unitDir())
rec, ok := sett.GetOffsiteDataAt("app")
if !ok || rec.DataAt != dataAt.Format(time.RFC3339) || rec.PushedAt == "" {
t.Fatalf("record = %+v ok=%v, want data at %s", rec, ok, dataAt.Format(time.RFC3339))
}
}
// The production push path writes the record (seam discipline: recordOffsiteDataAt is CALLED by
// runOffboxInternal after a successful snapshot, not only testable on its own).
func TestA4_TheOffsiteRunRecordsThePushedDataTime(t *testing.T) {
drive := t.TempDir()
m, sett, prov := classifiedOffboxManager(t, drive)
u := mkUnit(t, drive, "immich")
if err := writeManifest(UnitManifestFile(u), &RecoveryManifest{AppName: "immich", Data: &UnitData{At: "2026-09-25T02:15:01Z"}}); err != nil {
t.Fatal(err)
}
prov.hdd["immich"] = drive
prov.has["immich"] = true
_ = sett.SetAppOffbox("immich", true)
m.SetOffsitePreDumpFn(func(context.Context) error { return nil })
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
switch {
case contains(args, "cat") && contains(args, "config"):
return []byte(`{"version":2}`), nil
case contains(args, "snapshots"):
return []byte(`[]`), nil
case contains(args, "stats"):
return []byte(`{"total_size":123}`), nil
}
return nil, nil
})
if err := m.RunOffboxBackup(context.Background()); err != nil {
t.Fatalf("run: %v", err)
}
rec, ok := sett.GetOffsiteDataAt("immich")
if !ok || rec.DataAt != "2026-09-25T02:15:01Z" || rec.PushedAt == "" {
t.Fatalf("after a successful push the data-time record is %+v ok=%v, want the unit's data time", rec, ok)
}
}
+3 -3
View File
@@ -219,7 +219,7 @@ func (h *admissionHarness) runOneBackupRun() {
done := h.m.beginAdmissionRun()
defer done()
h.m.runVolumeDumps()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
}
// ── The instrument: a checksum of the whole backup tree ──────────────────────────────────────────
@@ -664,7 +664,7 @@ func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
// 2. The DB leg consults it BEFORE the dump. Order is the whole point: a gate after the write is
// the defect, relocated.
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "DumpOne",
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "dumpOneOrDefault", // v0.275.0: the dump goes through the seam wrapper
"the DATABASE leg dumps before consulting the reserve")
// 3. The volume leg consults it BEFORE the dump seam — which stops the stack as its first act.
@@ -691,7 +691,7 @@ func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
}
return true
})
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "CaptureRecoveryUnit",
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "captureRecoveryUnit", // v0.275.0: the data-run flag is passed through
"the CAPTURE leg captures before consulting the reserve")
}
+32 -5
View File
@@ -172,6 +172,14 @@ type Manager struct {
// disconnected) can be unit-tested without Docker. Nil → the real DumpAppVolumesSafe.
dumpVolumesSafe func(stackName string) error
// dumpOne (v0.275.0) — the per-database dump seam, nil → the real DumpOne. It lets a test drive the
// REAL legs (and so the stamps they write) without a database container.
dumpOne func(ctx context.Context, db DiscoveredDB, dumpDir string, logger *log.Logger, debug bool) DumpResult
// stampMu serialises writes of a unit's data-stamps.json (v0.275.0, data_versions.go). The legs run
// under the running flag already; this keeps a stray concurrent caller from losing a stamp.
stampMu sync.Mutex
// updatingCheck (slice 4) — nil-safe; see isHeld / SetUpdatingCheck.
updatingCheck func(stackName string) bool
// undoCopyRemover (R-671, v0.272.0) deletes an app's leftover undo copies — stacks.Manager.RemoveUndoCopies,
@@ -539,7 +547,13 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
defer m.beginRunSummary(kind, newRunID())()
defer m.emitRunSummary()
dbs, err := DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
discover := m.discoverDBs
if discover == nil {
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
}
}
dbs, err := discover(ctx)
if err != nil {
m.logger.Printf("[ERROR] [backup] Database discovery failed: %v", err)
return err
@@ -586,7 +600,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), db.StackName)
result := DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
result := m.dumpOneOrDefault(ctx, db, dumpDir)
results = append(results, result)
if result.Error != nil {
@@ -597,6 +611,8 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
} else {
totalSize += result.Size
summary = append(summary, fmt.Sprintf("OK %s (%s)", result.DB.ContainerName, humanizeBytes(result.Size)))
// v0.275.0 (R-696): the dump records the versions that wrote it, at the moment it is written.
m.stampDataFile(db.StackName, RecoveryUnitPath(m.namespaceRoot(drivePath), db.StackName), "db-dumps/"+filepath.Base(result.FilePath))
// Persist validation result to settings.json
if m.settings != nil && result.FilePath != "" {
@@ -644,8 +660,9 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
strings.Join(failedSummaryLines(summary), "; "))
}
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest).
m.captureAllRecoveryUnits()
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest). A DATA run:
// the capture folds the stamps the legs above just wrote (v0.275.0).
m.captureAllRecoveryUnits(true)
// F5 (CAMPAIGN-3): after the units are fresh on the CURRENT drives, prune any orphaned
// backups/primary/<app> dir an app left on an OLD drive when its HDD_PATH moved — pure disk
@@ -823,6 +840,8 @@ func (m *Manager) DumpAppVolumes(stackName string) error {
if info, _ := os.Stat(tarPath); info != nil {
m.logger.Printf("[INFO] [backup] Volume dump: %s/%s → %s", stackName, volName, humanizeBytes(info.Size()))
}
// v0.275.0 (R-696): the tar records the versions that wrote it.
m.stampDataFile(stackName, RecoveryUnitPath(m.namespaceRoot(drivePath), stackName), "volume-dumps/"+volName+".tar")
}
// Clean up tars (and any orphan `.tar.tmp` from a killed run) for volumes that no longer exist.
@@ -1158,7 +1177,7 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
func() {
defer m.beginRunSummary(runKindRefresh, "")()
defer m.emitRunSummary()
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false) // a REFRESH: never moves the unit's data time or its definition away from its data
}()
}
@@ -1439,3 +1458,11 @@ func (m *Manager) stackIsDeploying(name string) bool {
}
return false
}
// dumpOneOrDefault runs the dumpOne seam, or the real DumpOne.
func (m *Manager) dumpOneOrDefault(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
if m.dumpOne != nil {
return m.dumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
}
return DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
}
@@ -112,7 +112,7 @@ func TestFloor_RefusesTheAppAndLeavesItsPreviousUnitByteIdentical(t *testing.T)
}
before := checksumFile(t, prev)
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
// The refused app must NOT have been attempted at all — the floor is checked BEFORE any write.
for _, hit := range h.prov.infoHits {
@@ -170,7 +170,7 @@ func TestFloor_NeverDeletesAnotherAppsUnit(t *testing.T) {
}
before := checksumFile(t, keep)
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if _, err := os.Stat(keep); err != nil {
t.Fatalf("another app's unit was DELETED to make room: %v — nothing here is generational, so "+
@@ -190,7 +190,7 @@ func TestFloor_LargeUnitWithAmpleSpaceIsCaptured(t *testing.T) {
// A huge app on a huge, mostly-empty filesystem: 40% used, 600 GB free.
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 600, TotalGB: 1000, UsedGB: 400}
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 0 {
t.Fatalf("a capture was refused on a filesystem with 600 GB free (%+v) — the floor has become "+
@@ -209,7 +209,7 @@ func TestFloor_TheOld20GCeilingIsGone(t *testing.T) {
// deliberately far above 20 so that a literal `UsedGB > 20` cap cannot survive this test: a
// fixture sitting exactly on the old boundary would pass under the very shape it forbids.
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 180, TotalGB: 300, UsedGB: 120}
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 0 {
t.Fatalf("refused with 180 GB free: %+v — a fixed per-area limit survives somewhere", h.events)
}
@@ -258,7 +258,7 @@ func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
h := newFloorHarness(t, "immich")
// No entry → the injected reader returns nil, which is what system.GetDiskUsage does on error.
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 0 {
t.Fatalf("an UNREADABLE filesystem produced %d alert(s): %+v — an absent, unmounted or "+
@@ -275,7 +275,7 @@ func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
func TestErrCaptureFloor_IsMatchable(t *testing.T) {
h := newFloorHarness(t, "immich")
h.setSpace("immich", 99, 0.2)
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 1 {
t.Fatalf("want 1 event, got %d", len(h.events))
}
+309
View File
@@ -0,0 +1,309 @@
package backup
import (
"encoding/json"
"os"
"path/filepath"
"sort"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// ── The backup's data and its version travel together (controller v0.275.0, R-696, `07` §6.6) ────────
//
// WHAT WENT WRONG. The recovery unit's DEFINITION (compose/, image_pins) was re-captured by the periodic
// status refresh as soon as the app's pin moved, while its DATA (db-dumps/, volume-dumps/) is written
// only by the backup legs. So for up to a day after an update the unit said "new version" and held the
// old version's data. Measured on 9202 2026-09-26 (`audits/version-travel-2026-09-26/A1/`): after a
// PostgreSQL 16 → 18 step the unit held the 18 definition over a 16 dump and a 16 datadir tar; a restore
// poured the 16 datadir back, `postgres:18` refused it and the app was left down. And Tier 1's "proven
// at" was the manifest's refresh time, so the kept pre-conversion copy was released on a backup taken
// BEFORE the conversion (demo-hp, 2026-09-26 night).
//
// THE SHAPE. Every data file a backup leg writes gets a STAMP beside it, at the moment it is written:
// its size and mtime (so a file rewritten by anything else is recognised as unstamped), the definition's
// image pins, and what each service was running (`installed_images`, ref@digest). The unit capture folds
// the stamps into the manifest's `data` block — the time of the unit's data (its OLDEST stamped file:
// the copy is as fresh as its stalest part) and the versions that wrote it — and KEEPS the definition
// the data belongs to: when the pins have moved since the data was written, compose/ is not rewritten
// until the next data run replaces the data. The manifest's `image_pins` stays the app's CURRENT pins,
// so it says both. Restores read `data` to start the data with its own definition (restore_unit.go),
// and the update/release read `data.at` as the unit's time (restore_points.go).
//
// UNKNOWN IS NEVER CURRENT. A unit whose data files are not all validly stamped (written before
// v0.275.0, or by a path that does not stamp) has NO `data` block, restores as before with a WARN, and
// its time is the newest DATA file's mtime — never the manifest's.
// dataStampsFile sits in the unit root, beside manifest.json, so it travels with every copy of the unit
// (the Tier-2 mirror copies the whole unit, the off-site snapshot includes it).
const dataStampsFile = "data-stamps.json"
// DataStamp is one data file's record, written by the leg that wrote the file.
type DataStamp struct {
// At is the file's mtime when it was stamped, RFC3339Nano UTC — the stamp is valid only while the
// file still has exactly this mtime and Size.
At string `json:"at"`
Size int64 `json:"size"`
// Pins are the definition's image pins (the compose `image:` lines) when the file was written.
Pins []string `json:"pins"`
// Images is service -> ref@digest running when the file was written. Empty when not observed.
Images map[string]string `json:"images,omitempty"`
}
// UnitData is the manifest's account of the data the unit holds.
type UnitData struct {
// At is RFC3339 UTC: the unit's data time — the oldest data file's stamp, or, for a unit with no data
// files, the data run that confirmed it. This is the unit's "proven at", never the manifest's time.
At string `json:"at"`
// ImagePins are the definition pins the data belongs to — what compose/ holds. Empty when Mixed.
ImagePins []string `json:"image_pins"`
// Images is the running set (service -> ref@digest) when the oldest file was written.
Images map[string]string `json:"images,omitempty"`
// Files are the stamps, keyed by the path relative to the unit ("db-dumps/x.sql").
Files map[string]DataStamp `json:"files,omitempty"`
// Mixed: the files were written under DIFFERENT pins — no one definition fits all of them, so a
// restore of this unit is refused (a restore never starts data with a definition it does not belong to).
Mixed bool `json:"mixed,omitempty"`
}
// DataTime parses At. False when absent or unreadable.
func (d *UnitData) DataTime() (time.Time, bool) {
if d == nil || d.At == "" {
return time.Time{}, false
}
t, err := time.Parse(time.RFC3339, d.At)
if err != nil {
return time.Time{}, false
}
return t, true
}
func readDataStamps(unitDir string) map[string]DataStamp {
data, err := os.ReadFile(filepath.Join(unitDir, dataStampsFile))
if err != nil {
return map[string]DataStamp{}
}
var out map[string]DataStamp
if json.Unmarshal(data, &out) != nil || out == nil {
return map[string]DataStamp{}
}
return out
}
// stampDataFile records the file at <unitDir>/<rel> as written NOW by the current definition and
// running images. Called by the legs right after a dump or a tar is promoted to its final name. A failure
// to stamp is a WARN and leaves the file unstamped — the unit then reads as "versions unknown", which is
// the pre-v0.275.0 behaviour, never a false claim.
func (m *Manager) stampDataFile(stackName, unitDir, rel string) {
fi, err := os.Stat(filepath.Join(unitDir, rel))
if err != nil {
m.logger.Printf("[WARN] [backup] %s: cannot stamp %s (%v) — its versions will read as unknown", stackName, rel, err)
return
}
var pins []string
var images map[string]string
if m.stackProvider != nil {
if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok {
pins, images = definitionPins(info), info.InstalledImages
}
}
m.stampMu.Lock()
defer m.stampMu.Unlock()
stamps := readDataStamps(unitDir)
stamps[rel] = DataStamp{At: fi.ModTime().UTC().Format(time.RFC3339Nano), Size: fi.Size(), Pins: pins, Images: images}
// Entries whose file is gone (a volume the app no longer has) are dropped here, so the file stays small.
for k := range stamps {
if _, err := os.Stat(filepath.Join(unitDir, k)); err != nil {
delete(stamps, k)
}
}
body, err := json.MarshalIndent(stamps, "", " ")
if err == nil {
err = atomicWrite(filepath.Join(unitDir, dataStampsFile), append(body, '\n'), 0644)
}
if err != nil {
m.logger.Printf("[WARN] [backup] %s: writing the data stamp for %s failed (%v) — its versions will read as unknown", stackName, rel, err)
return
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] %s: stamped %s (%d B) with pins %v", stackName, rel, fi.Size(), pins)
}
}
// foldUnitData builds the manifest's `data` block from the unit's data files and their stamps.
//
// - no data files: a data run confirms the unit NOW under the current pins; a refresh keeps `prev`;
// - every file validly stamped: the oldest stamp's time; its pins when all agree, else Mixed;
// - any file unstamped or re-written since its stamp: nil — the versions are UNKNOWN.
func foldUnitData(unitDir string, dbDumps, volDumps []string, dataRun bool, now time.Time, pins []string, images map[string]string, prev *UnitData) *UnitData {
var rels []string
for _, n := range dbDumps {
rels = append(rels, "db-dumps/"+n)
}
for _, n := range volDumps {
rels = append(rels, "volume-dumps/"+n)
}
if len(rels) == 0 {
if dataRun {
return &UnitData{At: now.UTC().Format(time.RFC3339), ImagePins: pins, Images: images}
}
return prev
}
stamps := readDataStamps(unitDir)
out := &UnitData{Files: map[string]DataStamp{}}
var oldest time.Time
for _, rel := range rels {
st, ok := stamps[rel]
if !ok {
return nil
}
fi, err := os.Stat(filepath.Join(unitDir, rel))
if err != nil || fi.Size() != st.Size || fi.ModTime().UTC().Format(time.RFC3339Nano) != st.At {
return nil
}
t := fi.ModTime().UTC()
if oldest.IsZero() || t.Before(oldest) {
oldest = t
out.ImagePins, out.Images = st.Pins, st.Images
}
out.Files[rel] = st
}
for _, st := range out.Files {
if !samePins(st.Pins, out.ImagePins) {
out.Mixed = true
}
}
if out.Mixed {
out.ImagePins, out.Images = nil, nil
}
out.At = oldest.Format(time.RFC3339)
return out
}
// samePins compares two pin lists as SETS (compose order is file order on both sides, but a set compare
// cannot be fooled by it).
func samePins(a, b []string) bool {
if len(a) != len(b) {
return false
}
x := append([]string(nil), a...)
y := append([]string(nil), b...)
sort.Strings(x)
sort.Strings(y)
return stringSliceEqual(x, y)
}
// unitDataEqual is the manifest-rewrite check's view of `data`.
func unitDataEqual(a, b *UnitData) bool {
if a == nil || b == nil {
return a == nil && b == nil
}
if a.At != b.At || a.Mixed != b.Mixed || !stringSliceEqual(a.ImagePins, b.ImagePins) || len(a.Files) != len(b.Files) {
return false
}
for k, v := range a.Files {
w, ok := b.Files[k]
if !ok || v.At != w.At || v.Size != w.Size {
return false
}
}
return true
}
// PinsVersion is the household's name for a set of pins: every image as `name:tag`, registry path and
// digest stripped, in compose order ("docmost:0.96.0, postgres:16-alpine, redis:7-alpine"). All of them,
// because a version step can move ANY service — a PostgreSQL 16 → 18 step leaves the app's own tag as it
// was, and a label naming only the first image would read the same on both sides of it (seen on the
// first draft of the mismatch sentence: „(0.96.0) … (0.96.0)").
func PinsVersion(pins []string) string {
out := make([]string, 0, len(pins))
for _, ref := range pins {
if i := strings.Index(ref, "@"); i >= 0 {
ref = ref[:i]
}
if i := strings.LastIndex(ref, "/"); i >= 0 {
ref = ref[i+1:]
}
out = append(out, ref)
}
return strings.Join(out, ", ")
}
// recordOffsiteDataAt remembers the data time of the unit just pushed off-site (v0.275.0, R-696).
func (m *Manager) recordOffsiteDataAt(stackName, unitDir string) {
if m.settings == nil {
return
}
t, ok := unitNewestArtifact(unitDir)
if !ok {
return
}
now := time.Now().UTC().Format(time.RFC3339)
if err := m.settings.SetOffsiteDataAt(stackName, settings.OffsiteDataRecord{PushedAt: now, DataAt: t.UTC().Format(time.RFC3339)}); err != nil {
m.logger.Printf("[WARN] [offbox] %s: recording the pushed copy's data time failed: %v — the off-site copy is dated by its snapshot", stackName, err)
}
}
// offsiteDataTime caps an off-site snapshot's time with the data time this box recorded when it pushed
// it. A snapshot NEWER than the last recorded push (taken by another box, or a record lost) keeps its
// own time: the cap applies only to a copy this box knows the content of.
func (m *Manager) offsiteDataTime(stackName string, snapshotAt time.Time) time.Time {
if m.settings == nil {
return snapshotAt
}
rec, ok := m.settings.GetOffsiteDataAt(stackName)
if !ok {
return snapshotAt
}
pushed, perr := time.Parse(time.RFC3339, rec.PushedAt)
data, derr := time.Parse(time.RFC3339, rec.DataAt)
if perr != nil || derr != nil || snapshotAt.After(pushed) || !data.Before(snapshotAt) {
return snapshotAt
}
return data
}
// UnitDumpStamp is one stamped database dump of an app's own unit (v0.275.0).
type UnitDumpStamp struct {
File string
At time.Time
Images map[string]string
}
// UnitDumpStamps returns the app's OWN unit's database dumps as its manifest records them — only when
// the unit's data is known (a `data` block); an unstamped unit returns nothing, and a release that needs
// one waits (fail closed).
func (m *Manager) UnitDumpStamps(stackName string) []UnitDumpStamp {
man := readManifest(UnitManifestFile(m.primaryUnitDirFor(stackName)))
if man == nil || man.Data == nil {
return nil
}
var out []UnitDumpStamp
for rel, st := range man.Data.Files {
if !strings.HasPrefix(rel, "db-dumps/") || !strings.HasSuffix(rel, ".sql") {
continue
}
t, err := time.Parse(time.RFC3339Nano, st.At)
if err != nil {
continue
}
out = append(out, UnitDumpStamp{File: rel, At: t, Images: st.Images})
}
sort.Slice(out, func(i, j int) bool { return out[i].File < out[j].File })
return out
}
// definitionPins are the pins of the definition a capture copies into compose/: the stack dir's own
// docker-compose.yml, parsed. ONE source for the stamps, the fold and the freeze, so "the data's pins"
// and "the pins compose/ holds" are read from the same file the restore will start (production's
// RecoveryInfo.ImagePins is that parse too; a provider that says otherwise cannot make them disagree).
func definitionPins(info RecoveryInfo) []string {
if info.StackDir != "" {
if p := ParseComposeImages(filepath.Join(info.StackDir, "docker-compose.yml")); len(p) > 0 {
return p
}
}
return info.ImagePins
}
+3
View File
@@ -1372,6 +1372,9 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
continue
}
res.backedUp++
// v0.275.0 (R-696): what the snapshot just taken HOLDS is the unit's data, of the unit's data time —
// recorded so the update's precondition dates the off-site copy by its data, not by the snapshot.
m.recordOffsiteDataAt(stack, src)
// R-412 leg 1 — A PUSH THAT CARRIED NOTHING MUST NOT READ AS A PLAIN SUCCESS.
//
// Measured on demo-hp 2026-08-31: a recovery unit was destroyed mid-run, the capture rebuilt it
@@ -91,6 +91,13 @@ type OffsiteReconstituteResult struct {
// than reporting a bare success — a warning beside a success is read as a success, so the
// difference has to survive into the message.
Placement PlacementCheck
// v0.275.0 (R-696, `07` §6.6) — the versions the snapshot's data belongs to (UnitRestoreResult's
// fields, same meaning). VersionChanged: the app came back at the SNAPSHOT's version, its definition
// written from the snapshot's unit, because the live one was another version.
DataPins []string
DataAt time.Time
VersionChanged bool
VersionsUnknown bool
}
// fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite).
@@ -693,12 +700,50 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
return res, util.MsgError("err.backup.adatbazis_masolat_csonka_nem_indult", stack)
}
// --- WHICH VERSION COMES BACK (v0.275.0, R-696, `07` §6.6) ---------------------------------
// Until v0.275.0 this path never wrote the definition, so after any update the snapshot's data
// (last night's version) was started by the app's NEW definition — for an engine step that is the
// measured A1 failure (a PostgreSQL 16 datadir under 18: refused, the app left down). Now the
// snapshot's data comes back with the definition it belongs to: when the snapshot records its data's
// versions and they differ from what runs, the snapshot unit's definition is written into the stack
// dir (and pinned) before anything is started, and the normal guarded update climbs from there. A
// snapshot without that record restores as before, WARNed. Decided before the first mutation.
scratchCompose := UnitComposeDir(scratchUnit)
dataPins, verr := unitVersionCheck(stack, man, scratchCompose)
if verr != nil {
m.logger.Printf("[ERROR] [offbox] Restore REFUSED for %s: the snapshot's definition does not belong to its data — nothing was touched", stack)
return res, verr
}
var snapEnv map[string]string
defineFromSnapshot := false
if dataPins == nil {
res.VersionsUnknown = true
m.logger.Printf("[WARN] [offbox] %s: snapshot %s does not record which versions wrote its data (taken before v0.275.0) — restoring into the app's current definition, as before", stack, id)
} else {
res.DataPins = dataPins
res.DataAt, _ = man.Data.DataTime()
if info, ok := m.stackProvider.GetStackRecoveryInfo(stack); ok && len(definitionPins(info)) > 0 && !samePins(definitionPins(info), dataPins) {
env, _, eerr := m.unitRestoreEnv(stack, scratchCompose, man)
if eerr != nil {
return res, eerr
}
snapEnv, defineFromSnapshot, res.VersionChanged = env, true, true
m.logger.Printf("[INFO] [offbox] %s: snapshot %s holds data of %v (written %s); the app runs %v — it comes back at the snapshot's version", stack, id, dataPins, man.Data.At, definitionPins(info))
}
}
// --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------
// Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir,
// so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first
// mutation so the refusal below costs nothing.
// Read from the compose that will RUN: the live one, or — when the app comes back at the snapshot's
// version — the snapshot unit's, which is written into the stack dir before the first start. Resolved
// BEFORE the first mutation so the refusal below costs nothing.
var dbServices []string
if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
if defineFromSnapshot {
svcs, dsErr := DBServiceNames(filepath.Join(scratchCompose, "docker-compose.yml"))
if dsErr != nil {
m.logger.Printf("[WARN] [offbox] %s: could not read the snapshot's compose services: %v", stack, dsErr)
}
dbServices = svcs
} else if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
svcs, dsErr := DBServiceNames(composePath)
if dsErr != nil {
// "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse.
@@ -754,6 +799,16 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
if err := m.stackProvider.StopStack(stack); err != nil {
m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err)
}
// v0.275.0: the snapshot's own definition, written while nothing of the app's data has been touched
// yet — a failure here restarts the app as it was.
if defineFromSnapshot {
if err := m.stackProvider.RecreateStackDefinitionFromUnit(stack, scratchCompose, snapEnv); err != nil {
if sErr := restartStack(); sErr != nil {
m.logger.Printf("[WARN] [offbox] %s: restart after a failed definition write also failed: %v", stack, sErr)
}
return res, fmt.Errorf("restoring %s: writing the snapshot's definition failed: %w", stack, err)
}
}
copier := m.fullPlaceCopier()
for _, pl := range placements {
if pl.isUnit {
+57 -3
View File
@@ -69,6 +69,11 @@ type RecoveryManifest struct {
// never claim a coherence it did not establish — it carries the prior stamp forward instead.
OffsiteRunID string `json:"offsite_run_id,omitempty"`
DumpsAt string `json:"dumps_at,omitempty"` // RFC3339 UTC — when this run's dump leg finished
// Data (v0.275.0, R-696) is the account of the DATA this unit holds: its time and the versions that
// wrote it (data_versions.go). compose/ holds the definition THESE pins name; ImagePins above are the
// app's CURRENT pins — the two differ between an update and the next data run. Nil = unknown (a unit
// written before v0.275.0, or one with an unstamped data file): it restores as before, with a WARN.
Data *UnitData `json:"data,omitempty"`
}
// SetVersion records the controller version stamped into recovery-unit manifests.
@@ -92,7 +97,15 @@ func (m *Manager) SetTier2Notifier(fn func(stackName, destLabel string, dur time
// Idempotent: it builds the captured content in memory first and SKIPS all writes when the unit is
// already current (same config checksums, same dump set, same controller version) — so it can run on
// the periodic status refresh without thrashing a spinning USB drive.
//
// v0.275.0 (R-696): CaptureRecoveryUnit is the capture AFTER a data run (it follows the legs in
// RunAppBackupNow). The periodic refresh calls captureRecoveryUnit(…, false), which never moves the
// unit's data time and never rewrites the definition away from the data it belongs to.
func (m *Manager) CaptureRecoveryUnit(stackName string) error {
return m.captureRecoveryUnit(stackName, true)
}
func (m *Manager) captureRecoveryUnit(stackName string, dataRun bool) error {
if m.stackProvider == nil {
return fmt.Errorf("no stack provider")
}
@@ -161,6 +174,38 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
manifestPath := RecoveryUnitManifestPath(nsRoot, stackName)
cur := readManifest(manifestPath)
unitDir := RecoveryUnitPath(nsRoot, stackName)
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
// v0.275.0 (R-696): the data's own account, folded from the stamps the legs wrote.
var prevData *UnitData
if cur != nil {
prevData = cur.Data
}
curPins := definitionPins(info)
data := foldUnitData(unitDir, dbDumps, volDumps, dataRun, time.Now(), curPins, info.InstalledImages, prevData)
// THE DEFINITION STAYS WITH ITS DATA. When the app's pins have moved since the data was written (an
// update between two data runs), compose/ keeps the definition the data belongs to until the next data
// run replaces the data — the refresh used to rewrite it here within five minutes of every update,
// pairing the new version with the old data (A1: a 16 datadir under an 18 definition; the restore left
// the app down). The checksums then describe what compose/ really holds, so the already-current check
// below does not rewrite the manifest on every refresh.
frozen := data != nil && !data.Mixed && !samePins(data.ImagePins, curPins)
if frozen {
files, checksums, configFiles = nil, map[string]string{}, nil
for _, fname := range []string{"docker-compose.yml", ".felhom.yml", "app.yaml"} {
b, err := os.ReadFile(filepath.Join(composeDir, fname))
if err != nil {
continue
}
checksums[fname] = sha256Hex(b)
configFiles = append(configFiles, fname)
}
if held := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")); !samePins(held, data.ImagePins) {
m.logger.Printf("[WARN] [backup] %s: the unit's definition %v matches neither its data %v nor the app %v — kept as it is; a restore of this unit will refuse", stackName, held, data.ImagePins, curPins)
}
}
// R-43/R-44: the coherence stamp of the offsite run currently in flight ("" on the periodic
// refresh and on the local dump run). When empty we CARRY THE PRIOR STAMP FORWARD rather than
@@ -193,11 +238,16 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
stringMapEqual(cur.Checksums, checksums) &&
stringSliceEqual(cur.DBDumps, dbDumps) &&
stringSliceEqual(cur.VolumeDumps, volDumps) &&
stringSliceEqual(cur.ImagePins, info.ImagePins) &&
unitDataEqual(cur.Data, data) &&
cur.OffsiteRunID == runID {
return nil
}
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
if frozen && (cur == nil || cur.Data == nil || stringSliceEqual(cur.ImagePins, cur.Data.ImagePins)) {
m.logger.Printf("[INFO] [backup] %s: the app now runs %v; the unit keeps the definition of its data (%v, written %s) until the next backup replaces the data",
stackName, curPins, data.ImagePins, data.At)
}
if err := os.MkdirAll(composeDir, 0755); err != nil {
return fmt.Errorf("creating recovery-unit compose dir: %w", err)
}
@@ -226,6 +276,10 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
Checksums: checksums,
OffsiteRunID: runID,
DumpsAt: dumpsAt,
Data: data,
}
if data == nil && (len(dbDumps)+len(volDumps)) > 0 && m.isDebug() {
m.logger.Printf("[DEBUG] [backup] %s: the unit's data files are not all stamped — its versions are unknown until the next backup", stackName)
}
if err := writeManifest(manifestPath, manifest); err != nil {
return fmt.Errorf("writing manifest: %w", err)
@@ -387,7 +441,7 @@ func (m *Manager) readUnitSpace(stackName string) *UnitSpace {
// volume-dump legs of this run already consulted for this app. When a run is in flight the answer
// here is a memo lookup — an app refused before its first write is refused here too, silently,
// because it was already alerted once. Outside a run (the periodic status refresh) it decides fresh.
func (m *Manager) captureAllRecoveryUnits() {
func (m *Manager) captureAllRecoveryUnits(dataRun bool) {
if m.stackProvider == nil {
return
}
@@ -409,7 +463,7 @@ func (m *Manager) captureAllRecoveryUnits() {
if !m.admitApp(stack.Name) {
continue
}
if err := m.CaptureRecoveryUnit(stack.Name); err != nil {
if err := m.captureRecoveryUnit(stack.Name, dataRun); err != nil {
m.noteFailure(stack.Name, "recovery-unit capture", err.Error())
m.logger.Printf("[WARN] [backup] Recovery unit capture failed for %s: %v", stack.Name, err)
// R-158: per app, and the loop CONTINUES — one app's failure must not silence the
@@ -85,7 +85,7 @@ func newUnitNotifyManager(t *testing.T, stacks []string, fail map[string]bool) (
func TestCaptureAll_FailureNotifiesOnceWithTheSpaceFigures(t *testing.T) {
m, got := newUnitNotifyManager(t, []string{"immich"}, map[string]bool{"immich": true})
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false)
if len(*got) != 1 {
t.Fatalf("got %d unit-failure events, want exactly 1 — a per-app Tier-1 capture failure "+
@@ -117,7 +117,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
[]string{"homebox", "immich", "nextcloud"},
map[string]bool{"immich": true})
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false)
if len(*got) != 1 {
t.Fatalf("got %d events, want exactly 1 — either the loop ABORTED on the middle app "+
@@ -134,7 +134,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
m2, got2 := newUnitNotifyManager(t,
[]string{"homebox", "immich", "nextcloud"},
map[string]bool{"immich": true, "nextcloud": true})
m2.captureAllRecoveryUnits()
m2.captureAllRecoveryUnits(false)
if len(*got2) != 2 {
t.Fatalf("got %d events, want 2 — the app AFTER the first failure was never reached, so the "+
"loop is aborting rather than continuing: %+v", len(*got2), *got2)
@@ -148,7 +148,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
// to ignore.
func TestCaptureAll_SuccessIsSilent(t *testing.T) {
m, got := newUnitNotifyManager(t, []string{"homebox"}, nil)
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false)
if len(*got) != 0 {
t.Fatalf("a successful capture fired %d event(s): %+v", len(*got), *got)
}
@@ -163,7 +163,7 @@ func TestCaptureAll_UnwiredNotifyDoesNotPanic(t *testing.T) {
systemDataPath: dir,
stackProvider: &unitFailProvider{stacks: []string{"immich"}, fail: map[string]bool{"immich": true}, dir: dir},
}
m.captureAllRecoveryUnits() // no SetUnitNotify — must not panic
m.captureAllRecoveryUnits(false) // no SetUnitNotify — must not panic
}
// §8.4 in the failure direction: an unreadable target filesystem is reported as UNKNOWN, never as
+21 -6
View File
@@ -66,17 +66,32 @@ func (m *Manager) driveLabelForRoot(root string) string {
return ""
}
// unitNewestArtifact is the unit's data time: the newest of its manifest, .sql dumps and .tar
// volume dumps. ONE rule, shared with ListRestorePoints, so the two lists cannot date a unit
// differently.
// unitNewestArtifact is the unit's DATA time. ONE rule, shared by ListRestorePoints (Tier 1), the Tier-2
// copy's date, the removed-app list and kept data, so no two of them can date a unit differently.
//
// v0.275.0 (R-696) — THE TIME OF THE DATA, NEVER OF THE MANIFEST. It used to be the newest of the
// manifest, the .sql dumps and the .tar dumps; a refresh rewrites the manifest when the app's pins move,
// so a unit re-captured two minutes after an update read as two minutes old over data from before the
// update (9202 2026-09-25 11:06; demo-hp 2026-09-26 02:20, where it released the kept pre-conversion
// copy). Now: the manifest's `data.at` when the data is stamped; else the newest DATA file (the undo
// copies `pre-restore-*` excluded — an update's own safety dump is not a backup); the manifest's time
// only for a unit that holds no data file at all, whose whole content is its definition.
func unitNewestArtifact(unitDir string) (time.Time, bool) {
fi, err := os.Stat(UnitManifestFile(unitDir))
if err != nil {
return time.Time{}, false
}
newest := fi.ModTime()
newest = newestArtifact(UnitDBDumpDir(unitDir), ".sql", newest)
newest = newestArtifact(UnitVolumeDumpDir(unitDir), ".tar", newest)
if man := readManifest(UnitManifestFile(unitDir)); man != nil {
if t, ok := man.Data.DataTime(); ok {
return t, true
}
}
var newest time.Time
newest = newestDataFile(UnitDBDumpDir(unitDir), ".sql", newest)
newest = newestDataFile(UnitVolumeDumpDir(unitDir), ".tar", newest)
if newest.IsZero() {
return fi.ModTime(), true
}
return newest, true
}
+9 -12
View File
@@ -31,9 +31,7 @@ const restorePointShortID = "helyi"
// known at all (found=false → the caller should 404). A known stack with no recovery unit on
// disk returns an EMPTY list (a valid answer — "no backup yet"), not an error.
//
// The single point's Time is the newest mtime among the unit's artifacts (manifest.json,
// db-dumps/*.sql, volume-dumps/*.tar): the manifest is only rewritten when the app's config
// changes (checksum-skip), so the nightly-refreshed dumps are usually the freshest artifact.
// The single point's Time is the unit's DATA time (unitNewestArtifact, v0.275.0 — R-696).
func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, found bool) {
if m.stackProvider == nil {
return nil, false
@@ -61,13 +59,11 @@ func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, fo
return []RestorePoint{}, true
}
fi, err := os.Stat(RecoveryUnitManifestPath(nsRoot, stackName))
if err != nil {
// v0.275.0 (R-696): the unit's DATA time (unitNewestArtifact), never the manifest's refresh time.
newest, ok := unitNewestArtifact(RecoveryUnitPath(nsRoot, stackName))
if !ok {
return []RestorePoint{}, true // no recovery unit yet — "no backup" is a valid answer
}
newest := fi.ModTime()
newest = newestArtifact(AppDBDumpPath(nsRoot, stackName), ".sql", newest)
newest = newestArtifact(AppVolumeDumpPath(nsRoot, stackName), ".tar", newest)
return []RestorePoint{{
Time: newest.UTC().Format(time.RFC3339),
@@ -77,15 +73,16 @@ func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, fo
}}, true
}
// newestArtifact returns the newest mtime among cur and the files with the given extension in
// dir (non-recursive; a missing dir contributes nothing).
func newestArtifact(dir, ext string, cur time.Time) time.Time {
// newestDataFile returns the newest mtime among cur and the DATA files with the given extension in dir
// (non-recursive; a missing dir contributes nothing). The undo copies (`pre-restore-*`) are not data of
// the unit (R-361) and never date it.
func newestDataFile(dir, ext string, cur time.Time) time.Time {
entries, err := os.ReadDir(dir)
if err != nil {
return cur
}
for _, e := range entries {
if e.IsDir() || !strings.HasSuffix(e.Name(), ext) {
if e.IsDir() || !strings.HasSuffix(e.Name(), ext) || strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
continue
}
if info, err := e.Info(); err == nil && info.ModTime().After(cur) {
+122 -49
View File
@@ -1,6 +1,7 @@
package backup
import (
"errors"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"os"
@@ -166,6 +167,40 @@ type UnitRestoreResult struct {
// statement, and precisely the R-88 failure direction (degrade to NO DATA rather than to UNKNOWN)
// this whole change exists to remove. An unknown must be carried, never drawn as a zero.
CountsUnknown bool
// DataPins / DataAt (v0.275.0, R-696) — the versions the restored DATA belongs to and when it was
// written, from the unit's `data` block; the definition started with it is the one they name. Empty
// when the unit's versions are unknown (VersionsUnknown).
DataPins []string
DataAt time.Time
// VersionChanged — the app now runs DataPins, which differ from what it ran before the restore (a
// restore of an older version). The page then says which version came back (`07` §6.6).
VersionChanged bool
// VersionsUnknown — a unit written before v0.275.0 (or with an unstamped data file): restored as
// before, with a WARN naming it.
VersionsUnknown bool
}
// ErrUnitVersionMismatch is the KIND of the refusal of a restore whose unit would start its data with a
// definition the data does not belong to (v0.275.0, R-696): a unit whose files were written under different
// pins (Mixed), or whose compose/ names other pins than its data. Raised BEFORE anything is touched;
// branch with errors.Is. The customer sentence is the bundle's.
var ErrUnitVersionMismatch = errors.New("the recovery unit's definition is not the one its data belongs to")
// unitVersionCheck is A3's rule, as a pure function of the manifest and the unit's own compose file:
// known and matching → the pins to report; unknown → (nil, nil) and the caller WARNs; mixed or
// mismatched → the refusal.
func unitVersionCheck(stackName string, man *RecoveryManifest, composeDir string) ([]string, error) {
if man == nil || man.Data == nil {
return nil, nil
}
if man.Data.Mixed {
return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_versions_mixed", stackName)
}
def := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
if !samePins(def, man.Data.ImagePins) {
return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_version_mismatch", stackName, PinsVersion(def), PinsVersion(man.Data.ImagePins))
}
return man.Data.ImagePins, nil
}
// RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit.
@@ -323,60 +358,35 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
composeDir := UnitComposeDir(unitDir)
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
// reconcileRestoreSecrets), then the fail-closed gate.
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
if err != nil {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
return res, err
// v0.275.0 (R-696, `07` §6.6) — A RESTORE NEVER STARTS DATA WITH A DEFINITION IT DOES NOT BELONG TO.
// The unit keeps the definition of its data (data_versions.go); this refuses, before anything is
// touched, a unit where the two disagree. Measured before the fix (A1, 9202): the unit's PostgreSQL 18
// definition over a 16 datadir tar — the tar replaced the live volume, 18 refused it, the app stayed down.
dataPins, verr := unitVersionCheck(stackName, manifest, composeDir)
if verr != nil {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: the unit's definition %v, its data %v (mixed=%v) — a restore never starts data with a definition it does not belong to; nothing was touched",
stackName, ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")), manifest.Data.ImagePins, manifest.Data.Mixed)
return res, verr
}
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
// never generated: the fail-closed gate above already refused if one was missing, and the
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
//
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
// generator run means the secret was empty at capture AND absent from the guest.
//
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
// regenerated value leaves the app unable to authenticate against its own restored rows while the
// dump replay, which uses the container's local trust socket, still reports success. The old wording
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
// DB/root-password fields and is now scoped to what is actually true.
if len(missing) > 0 {
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
for _, dk := range manifest.DataKeyEnvVars {
dataKeySet[dk] = true
}
var generated, unresolved []string
for _, name := range missing {
if !dataKeySet[name] && m.generateSecret != nil {
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
fullEnv[name] = v
generated = append(generated, name)
continue
}
if dataPins == nil {
res.VersionsUnknown = true
m.logger.Printf("[WARN] [backup] Restore %s from %s: the unit does not record which versions wrote its data (written before v0.275.0, or an unstamped file) — restoring its definition as captured, as before", stackName, unitDir)
} else {
res.DataPins = dataPins
res.DataAt, _ = manifest.Data.DataTime()
if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok {
if live := definitionPins(info); len(live) > 0 && !samePins(live, dataPins) {
res.VersionChanged = true
m.logger.Printf("[INFO] [backup] Restore %s: the data belongs to %v (written %s); the app ran %v — it comes back at its data's version, and the update climbs from there",
stackName, dataPins, manifest.Data.At, live)
}
unresolved = append(unresolved, name)
}
if len(generated) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
stackName, generated)
}
if len(unresolved) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
stackName, len(unresolved), unresolved)
}
}
fullEnv, missing, err := m.unitRestoreEnv(stackName, composeDir, manifest)
if err != nil {
return res, err
}
// R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with
// two answers, and "an absent log line is not evidence" — the drill reads this line to prove the
// secondary mirror, not the primary unit, was the source. It is a path, never a secret.
@@ -472,3 +482,66 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
m.clearUpdateHoldAfterRestore(stackName)
return res, nil
}
// unitRestoreEnv rebuilds the env a restore starts the unit's definition with: the unit's plain config,
// its portable secrets (D5), the guest's for the withheld class, the fail-closed data-key gate, and a
// generated replacement for a missing RESETTABLE secret (O4). ONE implementation for the unit restore and,
// since v0.275.0, the off-site restore that brings an app back at its snapshot's version. Values are
// never logged.
func (m *Manager) unitRestoreEnv(stackName, composeDir string, manifest *RecoveryManifest) (fullEnv map[string]string, missing []string, err error) {
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
// reconcileRestoreSecrets), then the fail-closed gate.
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
fullEnv, missing, err = reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
if err != nil {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
return nil, missing, err
}
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
// never generated: the fail-closed gate above already refused if one was missing, and the
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
//
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
// generator run means the secret was empty at capture AND absent from the guest.
//
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
// regenerated value leaves the app unable to authenticate against its own restored rows while the
// dump replay, which uses the container's local trust socket, still reports success. The old wording
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
// DB/root-password fields and is now scoped to what is actually true.
if len(missing) > 0 {
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
for _, dk := range manifest.DataKeyEnvVars {
dataKeySet[dk] = true
}
var generated, unresolved []string
for _, name := range missing {
if !dataKeySet[name] && m.generateSecret != nil {
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
fullEnv[name] = v
generated = append(generated, name)
continue
}
}
unresolved = append(unresolved, name)
}
if len(generated) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
stackName, generated)
}
if len(unresolved) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
stackName, len(unresolved), unresolved)
}
}
return fullEnv, missing, nil
}
@@ -24,7 +24,7 @@ func (h *admissionHarness) digestOf(kind string) *RunSummary {
doneAdm := h.m.beginAdmissionRun()
doneSum := h.m.beginRunSummary(kind, "run-test")
h.m.runVolumeDumps()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
h.m.emitRunSummary()
doneSum()
doneAdm()
@@ -153,7 +153,7 @@ func TestRunSummary_RefreshSweepHasNoRunID(t *testing.T) {
defer h.m.beginAdmissionRun()()
defer h.m.beginRunSummary(runKindRefresh, "")()
defer h.m.emitRunSummary()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
}()
if got == nil {
t.Fatal("the periodic sweep emitted no digest — with the per-app event now record-only, a " +
@@ -148,7 +148,7 @@ func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) {
if len(h.volDumped) != 1 || h.volDumped[0] != "free" {
t.Errorf("positive control: the unheld app must still be dumped, got %v", h.volDumped)
}
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
for _, n := range h.prov.infoHits {
if n == "held" {
t.Error("the capture must not rewrite a HELD app's restore point")
@@ -178,7 +178,7 @@ func TestSlice4_NightlyLegsLeaveAnAppMidUpdateAlone(t *testing.T) {
h.m.settings = slice4Settings(t)
h.m.SetUpdatingCheck(func(name string) bool { return name == "updating" })
h.m.runVolumeDumps()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
var mirrored []string
h.m.perAppTier2 = func(name string) error { mirrored = append(mirrored, name); return nil }
h.m.RunAllTier2()
+16 -4
View File
@@ -43,6 +43,9 @@ type Tier2RestorePoint struct {
PackagePreserved bool
// CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded.
CopyLastSuccess string
// DataDate (v0.275.0, R-696) — the mirrored unit's DATA time (unitNewestArtifact on the mirror): when
// the data the copy holds was written, which a mirror run copies but never makes newer. "" = unknown.
DataDate string
}
// restorePointFromCoverage is the pure half of the predicate.
@@ -54,6 +57,7 @@ func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint {
CopyDateProven: cov.CopyLastSuccess != "",
PackagePreserved: preserved,
CopyLastSuccess: cov.CopyLastSuccess,
DataDate: cov.UnitDataDate,
}
}
@@ -93,6 +97,12 @@ func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
if err != nil {
return time.Time{}, false
}
// v0.275.0 (R-696): a mirror run copies the unit's data; it never makes the data newer. When the
// mirror's data time is known and older than the copy, the data time is the copy's age — a mirror
// taken right after an update, of a unit whose dump is from before it, is as old as that dump.
if d, derr := time.Parse(time.RFC3339, p.DataDate); derr == nil && d.Before(t) {
return d, true
}
return t, true
}
@@ -209,8 +219,9 @@ var updateOffsiteCheckTimeout = 15 * time.Second
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
type UpdateTierPoint struct {
Tier int
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime (capped by the
// mirror's data time), Tier 1 the unit's DATA time (ListRestorePoints → unitNewestArtifact), Tier 3
// the newest snapshot, capped by the data time the box recorded when it pushed it (v0.275.0, R-696).
At time.Time
}
@@ -281,7 +292,7 @@ func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier in
return UpdateTierPoint{}, false
}
if at, ok := got[stackName]; ok && !at.IsZero() {
return UpdateTierPoint{Tier: tier, At: at}, true
return UpdateTierPoint{Tier: tier, At: m.offsiteDataTime(stackName, at)}, true
}
}
return UpdateTierPoint{}, false
@@ -391,11 +402,12 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
if db.StackName != stackName {
continue
}
res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug())
res := m.dumpOneOrDefault(ctx, db, AppDBDumpPath(nsRoot, stackName))
if res.Error != nil {
return util.MsgError("err.backup.adatbazis_mentes_sikertelen", db.ContainerName, res.Error)
}
dumped++
m.stampDataFile(stackName, RecoveryUnitPath(nsRoot, stackName), "db-dumps/"+filepath.Base(res.FilePath))
m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size))
}