v0.275.0: a backup's data and its version travel together (R-696, 07 §6.6, D4 option A); R-695, R-691, R-694
gates / gates (push) Successful in 23s

The unit's data files are stamped with the versions that wrote them; the capture keeps the
definition the data belongs to; a restore never starts data under another version's
definition (unit restores refuse a mismatch; the off-site restore writes the snapshot's
definition); every tier's time is its data's; the conversion-copy release needs a dump on
the new engine. File-browser sync single-flight + no empty kept folder (R-695); the kept
view joins the folder's owning group, language switch resyncs (R-691); a restore-generated
login is not shown as the password (R-694). Red-proofs in
felhom.eu/documentation/audits/version-travel-2026-09-26/.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-26 10:35:22 +02:00
parent fb2bcdd5d6
commit b6810f14ff
47 changed files with 3771 additions and 135 deletions
+4
View File
@@ -83,6 +83,10 @@ type RecoveryInfo struct {
// 0600 app.yaml.
PortableSecretEnvVars []string
PortableSecrets map[string]string
// InstalledImages (v0.275.0, R-696) is what each compose service is RUNNING — app.yaml's
// `installed_images`, as `ref@digest` (the plain ref when no digest was recorded). Stamped beside each
// data file the backup writes, so the data says which versions wrote it. Nil = unknown, never current.
InstalledImages map[string]string
}
// ParseComposeImages extracts the pinned image references (`image: repo:tag`) from a
@@ -0,0 +1,576 @@
package backup
import (
"context"
"encoding/json"
"errors"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// Part A of the version-travel brief (controller v0.275.0, R-696, `07` §6.6): a backup's data and its
// version travel together. Measured before the fix on 9202 (`audits/version-travel-2026-09-26/A1/`): the
// periodic refresh re-captured the unit's DEFINITION two minutes after an update, over the previous
// version's DATA; Tier 1's time moved to the refresh; a restore in that window started a PostgreSQL 16
// datadir under the 18 definition and left the app down.
//
// Every test asserts a CONSEQUENCE a household or the update would see — which definition the restore
// starts, which time the precondition reads, which copy the release trusts — not the mechanism.
// vtStack is one app's stack dir + a provider whose ImagePins are what production derives them from:
// ParseComposeImages of the stack's compose (the adapter's GetStackRecoveryInfo does exactly that).
type vtStack struct {
t *testing.T
tmp string
drive string
stackDir string
fake *fakeRecoveryProvider
m *Manager
}
func vtCompose(pgMajor string) string {
return "services:\n app:\n image: example/app:0.96.0\n app-db:\n image: postgres:" + pgMajor + "-alpine\n"
}
func newVTStack(t *testing.T, pgMajor string) *vtStack {
t.Helper()
tmp := t.TempDir()
v := &vtStack{t: t, tmp: tmp, drive: filepath.Join(tmp, "drive"), stackDir: filepath.Join(tmp, "stack")}
if err := os.MkdirAll(v.stackDir, 0o755); err != nil {
t.Fatal(err)
}
v.fake = &fakeRecoveryProvider{hdd: v.drive, running: true}
v.m = &Manager{
logger: log.New(io.Discard, "", 0),
systemDataPath: filepath.Join(tmp, "system"),
stackProvider: v.fake,
version: "vtest",
}
v.setVersion(pgMajor)
return v
}
// setVersion is what an update does to the stack dir: the definition moves, and what runs moves with it.
func (v *vtStack) setVersion(pgMajor string) {
v.t.Helper()
mustWrite(v.t, filepath.Join(v.stackDir, "docker-compose.yml"), vtCompose(pgMajor))
mustWrite(v.t, filepath.Join(v.stackDir, ".felhom.yml"), "display_name: App "+pgMajor+"\n")
mustWrite(v.t, filepath.Join(v.stackDir, "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: vt\n")
v.fake.info = RecoveryInfo{
StackDir: v.stackDir,
DisplayName: "App",
ImagePins: ParseComposeImages(filepath.Join(v.stackDir, "docker-compose.yml")),
NonSecretEnv: map[string]string{"SUBDOMAIN": "vt"},
InstalledImages: map[string]string{
"app": "example/app:0.96.0@sha256:aaa",
"app-db": "postgres:" + pgMajor + "-alpine@sha256:" + pgMajor + pgMajor,
},
}
}
func (v *vtStack) unitDir() string { return RecoveryUnitPath(v.drive, "app") }
// backupLegs is what a data run writes, through the SAME stamp call the legs make: a database dump and
// a volume tar, both dated `at` (so "the refresh ran later" is a fact of the clock, not of the test).
func (v *vtStack) backupLegs(at time.Time, marker string) {
v.t.Helper()
sql := filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql")
tar := filepath.Join(UnitVolumeDumpDir(v.unitDir()), "app_db.tar")
mustWrite(v.t, sql, pgDump(1)+"-- "+marker+"\n")
mustWrite(v.t, tar, "tar:"+marker)
for _, p := range []string{sql, tar} {
if err := os.Chtimes(p, at, at); err != nil {
v.t.Fatal(err)
}
}
v.m.stampDataFile("app", v.unitDir(), "db-dumps/app-postgres.sql")
v.m.stampDataFile("app", v.unitDir(), "volume-dumps/app_db.tar")
}
func (v *vtStack) manifest() *RecoveryManifest {
v.t.Helper()
man := readManifest(UnitManifestFile(v.unitDir()))
if man == nil {
v.t.Fatal("no readable manifest")
}
return man
}
func (v *vtStack) unitComposeImages() []string {
return ParseComposeImages(filepath.Join(UnitComposeDir(v.unitDir()), "docker-compose.yml"))
}
// theWindow builds the measured A1 state: a data run at PostgreSQL 16 two hours ago, then the update to
// 18, then the periodic refresh NOW.
func theWindow(t *testing.T) (*vtStack, time.Time) {
t.Helper()
v := newVTStack(t, "16")
dataAt := time.Now().Add(-2 * time.Hour).UTC().Truncate(time.Second)
v.backupLegs(dataAt, "written-by-16")
if err := v.m.CaptureRecoveryUnit("app"); err != nil { // the data run's capture
t.Fatal(err)
}
v.setVersion("18") // the guarded update moved the pin
if err := v.m.captureRecoveryUnit("app", false); err != nil { // the 5-minute refresh
t.Fatal(err)
}
return v, dataAt
}
// A5 red-proof 1 — the manifest refresh after a pin change no longer moves Tier 1's time.
// Pre-fix (v0.274.0): ListRestorePoints = newest of the manifest's and the dumps' mtimes → "now".
func TestA5_RefreshAfterAPinChangeDoesNotMoveTier1sTime(t *testing.T) {
v, dataAt := theWindow(t)
pts, found := v.m.ListRestorePoints("app")
if !found || len(pts) != 1 {
t.Fatalf("restore points = %v found=%v, want one", pts, found)
}
got, err := time.Parse(time.RFC3339, pts[0].Time)
if err != nil {
t.Fatal(err)
}
if !got.Equal(dataAt) {
t.Fatalf("Tier 1's time = %s, want the DATA's time %s — the refresh %s made a two-hour-old dump read as new",
got.Format(time.RFC3339), dataAt.Format(time.RFC3339), time.Since(got).Round(time.Second))
}
}
// The definition stays with its data: after the refresh, compose/ still names the 16 definition, the
// manifest says both (image_pins = what runs, data.image_pins = what wrote the data).
// Pre-fix: the refresh rewrote compose/ to 18 (A1 S3b).
func TestA2_TheUnitKeepsTheDefinitionItsDataBelongsTo(t *testing.T) {
v, _ := theWindow(t)
if got := v.unitComposeImages(); !samePins(got, ParseComposeImages(writeTmpCompose(t, vtCompose("16")))) {
t.Fatalf("the unit's compose/ names %v — the refresh paired the 18 definition with the 16 data", got)
}
man := v.manifest()
if man.Data == nil || !strings.Contains(strings.Join(man.Data.ImagePins, " "), "postgres:16-alpine") {
t.Fatalf("manifest data = %+v, want the 16 pins", man.Data)
}
if !strings.Contains(strings.Join(man.ImagePins, " "), "postgres:18-alpine") {
t.Fatalf("manifest image_pins = %v, want the app's CURRENT (18) pins", man.ImagePins)
}
if got := man.Data.Files["db-dumps/app-postgres.sql"].Images["app-db"]; !strings.HasPrefix(got, "postgres:16-alpine@sha256:") {
t.Fatalf("the dump's recorded engine = %q, want postgres:16 with its digest", got)
}
// And a SECOND refresh writes nothing: the frozen checksums describe what compose/ holds.
before, _ := os.Stat(UnitManifestFile(v.unitDir()))
time.Sleep(20 * time.Millisecond)
if err := v.m.captureRecoveryUnit("app", false); err != nil {
t.Fatal(err)
}
after, _ := os.Stat(UnitManifestFile(v.unitDir()))
if !after.ModTime().Equal(before.ModTime()) {
t.Fatal("a second refresh rewrote the manifest — the frozen unit thrashes the drive every five minutes")
}
}
func writeTmpCompose(t *testing.T, body string) string {
t.Helper()
p := filepath.Join(t.TempDir(), "docker-compose.yml")
mustWrite(t, p, body)
return p
}
// The next data run replaces the data AND the definition: both are 18 afterwards.
func TestA2_TheNextDataRunMovesDataAndDefinitionTogether(t *testing.T) {
v, _ := theWindow(t)
now := time.Now().UTC().Truncate(time.Second)
v.backupLegs(now, "written-by-18")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
if got := strings.Join(v.unitComposeImages(), " "); !strings.Contains(got, "postgres:18-alpine") {
t.Fatalf("after a data run on 18 the unit's compose/ names %s, want 18", got)
}
if man := v.manifest(); man.Data == nil || man.Data.At != now.Format(time.RFC3339) || man.Data.Mixed {
t.Fatalf("data = %+v, want at %s, not mixed", man.Data, now.Format(time.RFC3339))
}
}
// vtRestorer records the definition the restore STARTS the data with.
type vtRestorer struct {
*fakeRecoveryProvider
startedWith []string
}
func (f *vtRestorer) RecreateStackDefinitionFromUnit(name, composeDir string, env map[string]string) error {
f.startedWith = ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
return f.fakeRecoveryProvider.RecreateStackDefinitionFromUnit(name, composeDir, env)
}
func vtRestoreSeams(m *Manager) (vols *[]string) {
var vd []string
m.volumeReplayFrom = func(_, dir string) (int, error) { vd = append(vd, dir); return 1, nil }
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
}
m.importDBDump = func(context.Context, DiscoveredDB, string) error { return nil }
return &vd
}
// A5 red-proof 3 — a restore in the window starts the OLD version with the OLD data, and says so.
// Pre-fix: the unit's definition was 18 (the refresh), so the 16 datadir was started under 18 (A1 S4).
func TestA5_ARestoreInTheWindowStartsTheOldVersionWithTheOldData(t *testing.T) {
v, dataAt := theWindow(t)
rec := &vtRestorer{fakeRecoveryProvider: v.fake}
v.m.stackProvider = rec
vtRestoreSeams(v.m)
res, err := v.m.RestoreFromRecoveryUnit("app")
if err != nil {
t.Fatalf("restore: %v", err)
}
if got := strings.Join(rec.startedWith, " "); !strings.Contains(got, "postgres:16-alpine") || strings.Contains(got, "postgres:18") {
t.Fatalf("the restore started the 16 data with the definition %s — a restore must never mix versions", got)
}
if !res.VersionChanged || !samePins(res.DataPins, ParseComposeImages(writeTmpCompose(t, vtCompose("16")))) || !res.DataAt.Equal(dataAt) {
t.Fatalf("result = changed %v pins %v at %s, want changed, the 16 pins, %s", res.VersionChanged, res.DataPins, res.DataAt, dataAt)
}
}
// A restore never starts data with a definition it does not belong to: a unit whose compose/ names other
// pins than its data, and a unit whose files were written by different versions, are refused BEFORE
// anything is touched (no stop, no volume, no recreate).
func TestA3_AMismatchedOrMixedUnitIsRefusedBeforeAnythingMoves(t *testing.T) {
t.Run("definition-not-the-data's", func(t *testing.T) {
v, _ := theWindow(t)
// What v0.274.0's refresh left behind: the NEW definition in compose/ over the old data.
mustWrite(t, filepath.Join(UnitComposeDir(v.unitDir()), "docker-compose.yml"), vtCompose("18"))
vols := vtRestoreSeams(v.m)
_, err := v.m.RestoreFromRecoveryUnit("app")
if !errors.Is(err, ErrUnitVersionMismatch) {
t.Fatalf("err = %v, want ErrUnitVersionMismatch", err)
}
if len(v.fake.calls) != 0 || len(*vols) != 0 {
t.Fatalf("the refusal touched the app: calls=%v volumes=%v", v.fake.calls, *vols)
}
})
t.Run("mixed", func(t *testing.T) {
v, _ := theWindow(t)
// The DB leg ran under 18, the volume leg failed and kept its 16 tar.
sql := filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql")
mustWrite(t, sql, pgDump(1))
v.m.stampDataFile("app", v.unitDir(), "db-dumps/app-postgres.sql")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
if man := v.manifest(); man.Data == nil || !man.Data.Mixed {
t.Fatalf("data = %+v, want Mixed", man.Data)
}
vols := vtRestoreSeams(v.m)
_, err := v.m.RestoreFromRecoveryUnit("app")
if !errors.Is(err, ErrUnitVersionMismatch) {
t.Fatalf("err = %v, want ErrUnitVersionMismatch", err)
}
if len(v.fake.calls) != 0 || len(*vols) != 0 {
t.Fatalf("the refusal touched the app: calls=%v volumes=%v", v.fake.calls, *vols)
}
})
}
// An older unit (no stamps) restores as before: its definition, VersionsUnknown, no refusal.
func TestA3_AnUnstampedUnitRestoresAsBefore(t *testing.T) {
v := newVTStack(t, "16")
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), pgDump(1))
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
if man := v.manifest(); man.Data != nil {
t.Fatalf("an unstamped dump produced data %+v — unknown must stay unknown", man.Data)
}
rec := &vtRestorer{fakeRecoveryProvider: v.fake}
v.m.stackProvider = rec
vtRestoreSeams(v.m)
res, err := v.m.RestoreFromRecoveryUnit("app")
if err != nil {
t.Fatalf("restore: %v", err)
}
if !res.VersionsUnknown || res.VersionChanged || len(rec.startedWith) == 0 {
t.Fatalf("result = %+v started=%v, want VersionsUnknown and the unit's definition started", res, rec.startedWith)
}
}
// A5 red-proof 4 — the update's precondition refuses a stale dump that a refresh made look new.
// Pre-fix: Tier 1 = the refresh's manifest time → "0m old" → accepted, and the update leaned on a copy
// of the previous version's data from two hours before.
func TestA5_ThePreconditionRefusesAStaleDumpARefreshMadeLookNew(t *testing.T) {
v, dataAt := theWindow(t)
now := time.Now()
fresh := func(p UpdateTierPoint) bool { return now.Sub(p.At) <= time.Hour }
p, ok, seen := v.m.UpdateRestorePoints(context.Background(), "app", fresh)
if ok {
t.Fatalf("the precondition ACCEPTED tier %d at %s (%s old) — the data is from %s",
p.Tier, p.At.Format(time.RFC3339), now.Sub(p.At).Round(time.Second), dataAt.Format(time.RFC3339))
}
if len(seen) != 1 || seen[0].Tier != UpdateTierLocal || !seen[0].At.Equal(dataAt) {
t.Fatalf("seen = %+v, want Tier 1 at the data's time %s", seen, dataAt.Format(time.RFC3339))
}
}
// A unit with no data file at all is dated by the data run that confirmed it, and a refresh keeps it.
func TestA4_AUnitWithoutDataFilesIsDatedByItsDataRun(t *testing.T) {
v := newVTStack(t, "16")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
first := v.manifest().Data
if first == nil || first.At == "" {
t.Fatalf("data = %+v, want the data run's time", first)
}
v.setVersion("18")
if err := v.m.captureRecoveryUnit("app", false); err != nil {
t.Fatal(err)
}
if got := v.manifest().Data; got == nil || got.At != first.At {
t.Fatalf("a refresh moved the data time: %+v → %+v", first, got)
}
}
// The undo copies are not data: an update's safety dump written after the backup must not date Tier 1.
func TestA4_AnUndoCopyNeverDatesTheUnit(t *testing.T) {
v := newVTStack(t, "16")
old := time.Now().Add(-3 * time.Hour).UTC().Truncate(time.Second)
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), pgDump(1))
_ = os.Chtimes(filepath.Join(UnitDBDumpDir(v.unitDir()), "app-postgres.sql"), old, old)
if err := v.m.captureRecoveryUnit("app", false); err != nil { // unstamped: the legacy rule
t.Fatal(err)
}
mustWrite(t, filepath.Join(UnitDBDumpDir(v.unitDir()), preRestoreDumpPrefix+"20260926T000000Z-app-postgres.sql"), pgDump(1))
pts, _ := v.m.ListRestorePoints("app")
if len(pts) != 1 || pts[0].Time != old.Format(time.RFC3339) {
t.Fatalf("Tier 1 = %+v, want the dump's %s (not the undo copy's, not the manifest's)", pts, old.Format(time.RFC3339))
}
}
// The stamps are written by the PRODUCTION leg (seam discipline): RunAppBackupNow → the dump seam → the
// stamp → the capture's `data`. A stamp helper nobody calls is the "seam built but never wired" shape.
func TestA2_TheUpdatesOwnBackupStampsItsDataThroughTheRealLeg(t *testing.T) {
v := newVTStack(t, "16")
v.m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
return []DiscoveredDB{{StackName: "app", ContainerName: "app-db", DBType: DBTypePostgres}}, nil
}
v.m.dumpOne = func(_ context.Context, db DiscoveredDB, dir string, _ *log.Logger, _ bool) DumpResult {
p := filepath.Join(dir, "app-postgres.sql")
mustWrite(t, p, pgDump(1))
return DumpResult{DB: db, FilePath: p}
}
v.m.perAppTier2 = func(string) error { return nil }
if err := v.m.RunAppBackupNow(context.Background(), "app"); err != nil {
t.Fatalf("RunAppBackupNow: %v", err)
}
man := v.manifest()
if man.Data == nil || man.Data.Mixed || len(man.Data.Files) != 1 {
t.Fatalf("data = %+v, want one stamped dump", man.Data)
}
st := man.Data.Files["db-dumps/app-postgres.sql"]
if st.Images["app-db"] != "postgres:16-alpine@sha256:1616" || !samePins(st.Pins, v.fake.info.ImagePins) {
t.Fatalf("stamp = %+v, want the running images and the definition's pins", st)
}
var raw map[string]interface{}
b, _ := os.ReadFile(UnitManifestFile(v.unitDir()))
_ = json.Unmarshal(b, &raw)
if _, ok := raw["data"]; !ok {
t.Fatal("manifest.json carries no `data` key")
}
}
// Tier 2: a mirror taken after the update of a unit whose data is older is as old as that data.
func TestA4_ATier2CopyIsAsOldAsItsData(t *testing.T) {
copyAt := time.Now().UTC().Truncate(time.Second)
dataAt := copyAt.Add(-2 * time.Hour)
p := Tier2RestorePoint{Restorable: true, CopyDateProven: true, CopyLastSuccess: copyAt.Format(time.RFC3339), DataDate: dataAt.Format(time.RFC3339)}
got, ok := p.ProvenCopyTime()
if !ok || !got.Equal(dataAt) {
t.Fatalf("Tier 2 proven at %s ok=%v, want the data's %s", got, ok, dataAt)
}
p.DataDate = ""
if got, _ := p.ProvenCopyTime(); !got.Equal(copyAt) {
t.Fatalf("with no data date, Tier 2 = %s, want the copy's %s (as before)", got, copyAt)
}
}
// The household's version label names every image — a PostgreSQL step is visible in it.
func TestA3_PinsVersionNamesEveryImage(t *testing.T) {
got := PinsVersion([]string{"docmost/docmost:0.96.0@sha256:b5", "postgres:16-alpine@sha256:72", "gitea.dooplex.hu/x/redis:7-alpine"})
if got != "docmost:0.96.0, postgres:16-alpine, redis:7-alpine" {
t.Fatalf("PinsVersion = %q", got)
}
}
// vtReconProvider is the reconstitution fixture's provider with a LIVE version (18) and a recorder for
// the definition the restore writes.
type vtReconProvider struct {
*recordingProvider
livePins []string
wroteDef []string
wroteAtCall int
}
func (p *vtReconProvider) GetStackRecoveryInfo(string) (RecoveryInfo, bool) {
return RecoveryInfo{DisplayName: "Immich", ImagePins: p.livePins}, true
}
func (p *vtReconProvider) RecreateStackDefinitionFromUnit(_, composeDir string, _ map[string]string) error {
p.wroteDef = ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
p.wroteAtCall = len(p.calls)
p.calls = append(p.calls, "recreate")
return nil
}
// vtSnapshotUnit gives the fixture's scratch unit a definition and a `data` block (a snapshot taken by
// v0.275.0), and returns the unit dir.
func vtSnapshotUnit(t *testing.T, m *Manager, pgMajor string, withData bool) string {
t.Helper()
scratch, _, err := m.offboxRestoreScratchDir("immich")
if err != nil {
t.Fatal(err)
}
var unit string
_ = filepath.Walk(scratch, func(p string, fi os.FileInfo, _ error) error {
if fi != nil && !fi.IsDir() && fi.Name() == "manifest.json" {
unit = filepath.Dir(p)
}
return nil
})
if unit == "" {
t.Fatal("no scratch unit")
}
mustWrite(t, filepath.Join(UnitComposeDir(unit), "docker-compose.yml"), vtCompose(pgMajor))
mustWrite(t, filepath.Join(UnitComposeDir(unit), "app.yaml"), "deployed: true\nenv:\n SUBDOMAIN: vt\n")
man := readManifest(UnitManifestFile(unit))
if withData {
man.Data = &UnitData{At: "2026-09-26T02:15:01Z", ImagePins: ParseComposeImages(filepath.Join(UnitComposeDir(unit), "docker-compose.yml"))}
}
if err := writeManifest(UnitManifestFile(unit), man); err != nil {
t.Fatal(err)
}
return unit
}
// Off-site (Tier 3): v0.274.0 never wrote the definition — last night's snapshot data went under the
// app's NEW definition (A1, read from source). Now the snapshot's own definition is written BEFORE any
// file, volume or database is touched, and the database service is resolved from it.
func TestA3_TheOffsiteRestoreBringsTheSnapshotsVersionBack(t *testing.T) {
m, prov, imported := reconFixture(t, "20260926T021500Z", "2026-09-26T02:15:01Z", pgDump(1))
vp := &vtReconProvider{recordingProvider: prov, livePins: ParseComposeImages(writeTmpCompose(t, vtCompose("18")))}
m.SetStackProvider(vp)
vtSnapshotUnit(t, m, "16", true)
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err != nil {
t.Fatalf("reconstitute: %v", err)
}
if got := strings.Join(vp.wroteDef, " "); !strings.Contains(got, "postgres:16-alpine") {
t.Fatalf("the restore wrote the definition %q — want the snapshot's 16", got)
}
if vp.wroteAtCall != 1 || vp.calls[0] != "stop" {
t.Fatalf("calls = %v — the definition must be written right after the stop, before any data", vp.calls)
}
if !res.VersionChanged || res.DataAt.Format(time.RFC3339) != "2026-09-26T02:15:01Z" || len(*imported) != 1 {
t.Fatalf("result changed=%v at=%s imported=%v", res.VersionChanged, res.DataAt, *imported)
}
if got := strings.Join(vp.gotServices, ","); got != "app-db" {
t.Fatalf("the DB-only start was %q — the database service must come from the definition that RUNS (the snapshot's)", got)
}
}
// Same version, or a snapshot from before v0.275.0: nothing is written — the path of every earlier release.
func TestA3_TheOffsiteRestoreLeavesTheDefinitionWhenNothingDiffers(t *testing.T) {
for _, c := range []struct {
name string
live string
withData bool
unknown bool
}{{"same-version", "16", true, false}, {"pre-v0.275.0-snapshot", "18", false, true}} {
t.Run(c.name, func(t *testing.T) {
m, prov, _ := reconFixture(t, "20260926T021500Z", "2026-09-26T02:15:01Z", pgDump(1))
vp := &vtReconProvider{recordingProvider: prov, livePins: ParseComposeImages(writeTmpCompose(t, vtCompose(c.live)))}
m.SetStackProvider(vp)
vtSnapshotUnit(t, m, "16", c.withData)
res, err := m.ReconstituteFromOffsite(context.Background(), "immich", false)
if err != nil {
t.Fatalf("reconstitute: %v", err)
}
if vp.wroteDef != nil || res.VersionChanged || res.VersionsUnknown != c.unknown {
t.Fatalf("wrote %v changed=%v unknown=%v", vp.wroteDef, res.VersionChanged, res.VersionsUnknown)
}
})
}
}
// Tier 3: a snapshot pushed after a failed dump leg carries older data than its own time — the recorded
// push caps it; a snapshot newer than the last recorded push (another box) keeps its own time.
func TestA4_AnOffsiteCopyIsAsOldAsTheDataItWasPushedWith(t *testing.T) {
m, sett := newOffboxManager(t)
snap := time.Date(2026, 9, 26, 2, 15, 0, 0, time.UTC)
if got := m.offsiteDataTime("app", snap); !got.Equal(snap) {
t.Fatalf("no record: %s, want the snapshot's own time", got)
}
data := snap.Add(-24 * time.Hour)
if err := sett.SetOffsiteDataAt("app", settings.OffsiteDataRecord{PushedAt: snap.Add(time.Minute).Format(time.RFC3339), DataAt: data.Format(time.RFC3339)}); err != nil {
t.Fatal(err)
}
if got := m.offsiteDataTime("app", snap); !got.Equal(data) {
t.Fatalf("recorded push: %s, want the data's %s", got, data)
}
later := snap.Add(2 * time.Hour)
if got := m.offsiteDataTime("app", later); !got.Equal(later) {
t.Fatalf("a snapshot newer than the recorded push: %s, want its own %s", got, later)
}
}
// The push records the pushed unit's DATA time (the production call in runOffboxInternal).
func TestA4_ThePushRecordsTheUnitsDataTime(t *testing.T) {
m, sett := newOffboxManager(t)
v := newVTStack(t, "16")
dataAt := time.Now().Add(-26 * time.Hour).UTC().Truncate(time.Second)
v.backupLegs(dataAt, "x")
if err := v.m.CaptureRecoveryUnit("app"); err != nil {
t.Fatal(err)
}
m.recordOffsiteDataAt("app", v.unitDir())
rec, ok := sett.GetOffsiteDataAt("app")
if !ok || rec.DataAt != dataAt.Format(time.RFC3339) || rec.PushedAt == "" {
t.Fatalf("record = %+v ok=%v, want data at %s", rec, ok, dataAt.Format(time.RFC3339))
}
}
// The production push path writes the record (seam discipline: recordOffsiteDataAt is CALLED by
// runOffboxInternal after a successful snapshot, not only testable on its own).
func TestA4_TheOffsiteRunRecordsThePushedDataTime(t *testing.T) {
drive := t.TempDir()
m, sett, prov := classifiedOffboxManager(t, drive)
u := mkUnit(t, drive, "immich")
if err := writeManifest(UnitManifestFile(u), &RecoveryManifest{AppName: "immich", Data: &UnitData{At: "2026-09-25T02:15:01Z"}}); err != nil {
t.Fatal(err)
}
prov.hdd["immich"] = drive
prov.has["immich"] = true
_ = sett.SetAppOffbox("immich", true)
m.SetOffsitePreDumpFn(func(context.Context) error { return nil })
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
switch {
case contains(args, "cat") && contains(args, "config"):
return []byte(`{"version":2}`), nil
case contains(args, "snapshots"):
return []byte(`[]`), nil
case contains(args, "stats"):
return []byte(`{"total_size":123}`), nil
}
return nil, nil
})
if err := m.RunOffboxBackup(context.Background()); err != nil {
t.Fatalf("run: %v", err)
}
rec, ok := sett.GetOffsiteDataAt("immich")
if !ok || rec.DataAt != "2026-09-25T02:15:01Z" || rec.PushedAt == "" {
t.Fatalf("after a successful push the data-time record is %+v ok=%v, want the unit's data time", rec, ok)
}
}
+3 -3
View File
@@ -219,7 +219,7 @@ func (h *admissionHarness) runOneBackupRun() {
done := h.m.beginAdmissionRun()
defer done()
h.m.runVolumeDumps()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
}
// ── The instrument: a checksum of the whole backup tree ──────────────────────────────────────────
@@ -664,7 +664,7 @@ func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
// 2. The DB leg consults it BEFORE the dump. Order is the whole point: a gate after the write is
// the defect, relocated.
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "DumpOne",
assertGateBefore(t, calls["runDBDumpsInternal"], "admitApp", "dumpOneOrDefault", // v0.275.0: the dump goes through the seam wrapper
"the DATABASE leg dumps before consulting the reserve")
// 3. The volume leg consults it BEFORE the dump seam — which stops the stack as its first act.
@@ -691,7 +691,7 @@ func TestAdmission_IsWiredIntoEveryProductionWriteLeg(t *testing.T) {
}
return true
})
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "CaptureRecoveryUnit",
assertGateBefore(t, capCalls["captureAllRecoveryUnits"], "admitApp", "captureRecoveryUnit", // v0.275.0: the data-run flag is passed through
"the CAPTURE leg captures before consulting the reserve")
}
+32 -5
View File
@@ -172,6 +172,14 @@ type Manager struct {
// disconnected) can be unit-tested without Docker. Nil → the real DumpAppVolumesSafe.
dumpVolumesSafe func(stackName string) error
// dumpOne (v0.275.0) — the per-database dump seam, nil → the real DumpOne. It lets a test drive the
// REAL legs (and so the stamps they write) without a database container.
dumpOne func(ctx context.Context, db DiscoveredDB, dumpDir string, logger *log.Logger, debug bool) DumpResult
// stampMu serialises writes of a unit's data-stamps.json (v0.275.0, data_versions.go). The legs run
// under the running flag already; this keeps a stray concurrent caller from losing a stamp.
stampMu sync.Mutex
// updatingCheck (slice 4) — nil-safe; see isHeld / SetUpdatingCheck.
updatingCheck func(stackName string) bool
// undoCopyRemover (R-671, v0.272.0) deletes an app's leftover undo copies — stacks.Manager.RemoveUndoCopies,
@@ -539,7 +547,13 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
defer m.beginRunSummary(kind, newRunID())()
defer m.emitRunSummary()
dbs, err := DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
discover := m.discoverDBs
if discover == nil {
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
}
}
dbs, err := discover(ctx)
if err != nil {
m.logger.Printf("[ERROR] [backup] Database discovery failed: %v", err)
return err
@@ -586,7 +600,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
dumpDir := AppDBDumpPath(m.namespaceRoot(drivePath), db.StackName)
result := DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
result := m.dumpOneOrDefault(ctx, db, dumpDir)
results = append(results, result)
if result.Error != nil {
@@ -597,6 +611,8 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
} else {
totalSize += result.Size
summary = append(summary, fmt.Sprintf("OK %s (%s)", result.DB.ContainerName, humanizeBytes(result.Size)))
// v0.275.0 (R-696): the dump records the versions that wrote it, at the moment it is written.
m.stampDataFile(db.StackName, RecoveryUnitPath(m.namespaceRoot(drivePath), db.StackName), "db-dumps/"+filepath.Base(result.FilePath))
// Persist validation result to settings.json
if m.settings != nil && result.FilePath != "" {
@@ -644,8 +660,9 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
strings.Join(failedSummaryLines(summary), "; "))
}
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest).
m.captureAllRecoveryUnits()
// Phase 2: refresh each deployed app's self-contained recovery unit (compose + manifest). A DATA run:
// the capture folds the stamps the legs above just wrote (v0.275.0).
m.captureAllRecoveryUnits(true)
// F5 (CAMPAIGN-3): after the units are fresh on the CURRENT drives, prune any orphaned
// backups/primary/<app> dir an app left on an OLD drive when its HDD_PATH moved — pure disk
@@ -823,6 +840,8 @@ func (m *Manager) DumpAppVolumes(stackName string) error {
if info, _ := os.Stat(tarPath); info != nil {
m.logger.Printf("[INFO] [backup] Volume dump: %s/%s → %s", stackName, volName, humanizeBytes(info.Size()))
}
// v0.275.0 (R-696): the tar records the versions that wrote it.
m.stampDataFile(stackName, RecoveryUnitPath(m.namespaceRoot(drivePath), stackName), "volume-dumps/"+volName+".tar")
}
// Clean up tars (and any orphan `.tar.tmp` from a killed run) for volumes that no longer exist.
@@ -1158,7 +1177,7 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
func() {
defer m.beginRunSummary(runKindRefresh, "")()
defer m.emitRunSummary()
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false) // a REFRESH: never moves the unit's data time or its definition away from its data
}()
}
@@ -1439,3 +1458,11 @@ func (m *Manager) stackIsDeploying(name string) bool {
}
return false
}
// dumpOneOrDefault runs the dumpOne seam, or the real DumpOne.
func (m *Manager) dumpOneOrDefault(ctx context.Context, db DiscoveredDB, dumpDir string) DumpResult {
if m.dumpOne != nil {
return m.dumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
}
return DumpOne(ctx, db, dumpDir, m.logger, m.isDebug())
}
@@ -112,7 +112,7 @@ func TestFloor_RefusesTheAppAndLeavesItsPreviousUnitByteIdentical(t *testing.T)
}
before := checksumFile(t, prev)
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
// The refused app must NOT have been attempted at all — the floor is checked BEFORE any write.
for _, hit := range h.prov.infoHits {
@@ -170,7 +170,7 @@ func TestFloor_NeverDeletesAnotherAppsUnit(t *testing.T) {
}
before := checksumFile(t, keep)
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if _, err := os.Stat(keep); err != nil {
t.Fatalf("another app's unit was DELETED to make room: %v — nothing here is generational, so "+
@@ -190,7 +190,7 @@ func TestFloor_LargeUnitWithAmpleSpaceIsCaptured(t *testing.T) {
// A huge app on a huge, mostly-empty filesystem: 40% used, 600 GB free.
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 600, TotalGB: 1000, UsedGB: 400}
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 0 {
t.Fatalf("a capture was refused on a filesystem with 600 GB free (%+v) — the floor has become "+
@@ -209,7 +209,7 @@ func TestFloor_TheOld20GCeilingIsGone(t *testing.T) {
// deliberately far above 20 so that a literal `UsedGB > 20` cap cannot survive this test: a
// fixture sitting exactly on the old boundary would pass under the very shape it forbids.
h.usage["immich"] = &UnitSpace{Path: h.dir, UsedPercent: 40, AvailGB: 180, TotalGB: 300, UsedGB: 120}
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 0 {
t.Fatalf("refused with 180 GB free: %+v — a fixed per-area limit survives somewhere", h.events)
}
@@ -258,7 +258,7 @@ func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
h := newFloorHarness(t, "immich")
// No entry → the injected reader returns nil, which is what system.GetDiskUsage does on error.
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 0 {
t.Fatalf("an UNREADABLE filesystem produced %d alert(s): %+v — an absent, unmounted or "+
@@ -275,7 +275,7 @@ func TestFloor_UnreadableFilesystemNeitherRefusesNorWarns(t *testing.T) {
func TestErrCaptureFloor_IsMatchable(t *testing.T) {
h := newFloorHarness(t, "immich")
h.setSpace("immich", 99, 0.2)
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
if len(h.events) != 1 {
t.Fatalf("want 1 event, got %d", len(h.events))
}
+309
View File
@@ -0,0 +1,309 @@
package backup
import (
"encoding/json"
"os"
"path/filepath"
"sort"
"strings"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// ── The backup's data and its version travel together (controller v0.275.0, R-696, `07` §6.6) ────────
//
// WHAT WENT WRONG. The recovery unit's DEFINITION (compose/, image_pins) was re-captured by the periodic
// status refresh as soon as the app's pin moved, while its DATA (db-dumps/, volume-dumps/) is written
// only by the backup legs. So for up to a day after an update the unit said "new version" and held the
// old version's data. Measured on 9202 2026-09-26 (`audits/version-travel-2026-09-26/A1/`): after a
// PostgreSQL 16 → 18 step the unit held the 18 definition over a 16 dump and a 16 datadir tar; a restore
// poured the 16 datadir back, `postgres:18` refused it and the app was left down. And Tier 1's "proven
// at" was the manifest's refresh time, so the kept pre-conversion copy was released on a backup taken
// BEFORE the conversion (demo-hp, 2026-09-26 night).
//
// THE SHAPE. Every data file a backup leg writes gets a STAMP beside it, at the moment it is written:
// its size and mtime (so a file rewritten by anything else is recognised as unstamped), the definition's
// image pins, and what each service was running (`installed_images`, ref@digest). The unit capture folds
// the stamps into the manifest's `data` block — the time of the unit's data (its OLDEST stamped file:
// the copy is as fresh as its stalest part) and the versions that wrote it — and KEEPS the definition
// the data belongs to: when the pins have moved since the data was written, compose/ is not rewritten
// until the next data run replaces the data. The manifest's `image_pins` stays the app's CURRENT pins,
// so it says both. Restores read `data` to start the data with its own definition (restore_unit.go),
// and the update/release read `data.at` as the unit's time (restore_points.go).
//
// UNKNOWN IS NEVER CURRENT. A unit whose data files are not all validly stamped (written before
// v0.275.0, or by a path that does not stamp) has NO `data` block, restores as before with a WARN, and
// its time is the newest DATA file's mtime — never the manifest's.
// dataStampsFile sits in the unit root, beside manifest.json, so it travels with every copy of the unit
// (the Tier-2 mirror copies the whole unit, the off-site snapshot includes it).
const dataStampsFile = "data-stamps.json"
// DataStamp is one data file's record, written by the leg that wrote the file.
type DataStamp struct {
// At is the file's mtime when it was stamped, RFC3339Nano UTC — the stamp is valid only while the
// file still has exactly this mtime and Size.
At string `json:"at"`
Size int64 `json:"size"`
// Pins are the definition's image pins (the compose `image:` lines) when the file was written.
Pins []string `json:"pins"`
// Images is service -> ref@digest running when the file was written. Empty when not observed.
Images map[string]string `json:"images,omitempty"`
}
// UnitData is the manifest's account of the data the unit holds.
type UnitData struct {
// At is RFC3339 UTC: the unit's data time — the oldest data file's stamp, or, for a unit with no data
// files, the data run that confirmed it. This is the unit's "proven at", never the manifest's time.
At string `json:"at"`
// ImagePins are the definition pins the data belongs to — what compose/ holds. Empty when Mixed.
ImagePins []string `json:"image_pins"`
// Images is the running set (service -> ref@digest) when the oldest file was written.
Images map[string]string `json:"images,omitempty"`
// Files are the stamps, keyed by the path relative to the unit ("db-dumps/x.sql").
Files map[string]DataStamp `json:"files,omitempty"`
// Mixed: the files were written under DIFFERENT pins — no one definition fits all of them, so a
// restore of this unit is refused (a restore never starts data with a definition it does not belong to).
Mixed bool `json:"mixed,omitempty"`
}
// DataTime parses At. False when absent or unreadable.
func (d *UnitData) DataTime() (time.Time, bool) {
if d == nil || d.At == "" {
return time.Time{}, false
}
t, err := time.Parse(time.RFC3339, d.At)
if err != nil {
return time.Time{}, false
}
return t, true
}
func readDataStamps(unitDir string) map[string]DataStamp {
data, err := os.ReadFile(filepath.Join(unitDir, dataStampsFile))
if err != nil {
return map[string]DataStamp{}
}
var out map[string]DataStamp
if json.Unmarshal(data, &out) != nil || out == nil {
return map[string]DataStamp{}
}
return out
}
// stampDataFile records the file at <unitDir>/<rel> as written NOW by the current definition and
// running images. Called by the legs right after a dump or a tar is promoted to its final name. A failure
// to stamp is a WARN and leaves the file unstamped — the unit then reads as "versions unknown", which is
// the pre-v0.275.0 behaviour, never a false claim.
func (m *Manager) stampDataFile(stackName, unitDir, rel string) {
fi, err := os.Stat(filepath.Join(unitDir, rel))
if err != nil {
m.logger.Printf("[WARN] [backup] %s: cannot stamp %s (%v) — its versions will read as unknown", stackName, rel, err)
return
}
var pins []string
var images map[string]string
if m.stackProvider != nil {
if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok {
pins, images = definitionPins(info), info.InstalledImages
}
}
m.stampMu.Lock()
defer m.stampMu.Unlock()
stamps := readDataStamps(unitDir)
stamps[rel] = DataStamp{At: fi.ModTime().UTC().Format(time.RFC3339Nano), Size: fi.Size(), Pins: pins, Images: images}
// Entries whose file is gone (a volume the app no longer has) are dropped here, so the file stays small.
for k := range stamps {
if _, err := os.Stat(filepath.Join(unitDir, k)); err != nil {
delete(stamps, k)
}
}
body, err := json.MarshalIndent(stamps, "", " ")
if err == nil {
err = atomicWrite(filepath.Join(unitDir, dataStampsFile), append(body, '\n'), 0644)
}
if err != nil {
m.logger.Printf("[WARN] [backup] %s: writing the data stamp for %s failed (%v) — its versions will read as unknown", stackName, rel, err)
return
}
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] %s: stamped %s (%d B) with pins %v", stackName, rel, fi.Size(), pins)
}
}
// foldUnitData builds the manifest's `data` block from the unit's data files and their stamps.
//
// - no data files: a data run confirms the unit NOW under the current pins; a refresh keeps `prev`;
// - every file validly stamped: the oldest stamp's time; its pins when all agree, else Mixed;
// - any file unstamped or re-written since its stamp: nil — the versions are UNKNOWN.
func foldUnitData(unitDir string, dbDumps, volDumps []string, dataRun bool, now time.Time, pins []string, images map[string]string, prev *UnitData) *UnitData {
var rels []string
for _, n := range dbDumps {
rels = append(rels, "db-dumps/"+n)
}
for _, n := range volDumps {
rels = append(rels, "volume-dumps/"+n)
}
if len(rels) == 0 {
if dataRun {
return &UnitData{At: now.UTC().Format(time.RFC3339), ImagePins: pins, Images: images}
}
return prev
}
stamps := readDataStamps(unitDir)
out := &UnitData{Files: map[string]DataStamp{}}
var oldest time.Time
for _, rel := range rels {
st, ok := stamps[rel]
if !ok {
return nil
}
fi, err := os.Stat(filepath.Join(unitDir, rel))
if err != nil || fi.Size() != st.Size || fi.ModTime().UTC().Format(time.RFC3339Nano) != st.At {
return nil
}
t := fi.ModTime().UTC()
if oldest.IsZero() || t.Before(oldest) {
oldest = t
out.ImagePins, out.Images = st.Pins, st.Images
}
out.Files[rel] = st
}
for _, st := range out.Files {
if !samePins(st.Pins, out.ImagePins) {
out.Mixed = true
}
}
if out.Mixed {
out.ImagePins, out.Images = nil, nil
}
out.At = oldest.Format(time.RFC3339)
return out
}
// samePins compares two pin lists as SETS (compose order is file order on both sides, but a set compare
// cannot be fooled by it).
func samePins(a, b []string) bool {
if len(a) != len(b) {
return false
}
x := append([]string(nil), a...)
y := append([]string(nil), b...)
sort.Strings(x)
sort.Strings(y)
return stringSliceEqual(x, y)
}
// unitDataEqual is the manifest-rewrite check's view of `data`.
func unitDataEqual(a, b *UnitData) bool {
if a == nil || b == nil {
return a == nil && b == nil
}
if a.At != b.At || a.Mixed != b.Mixed || !stringSliceEqual(a.ImagePins, b.ImagePins) || len(a.Files) != len(b.Files) {
return false
}
for k, v := range a.Files {
w, ok := b.Files[k]
if !ok || v.At != w.At || v.Size != w.Size {
return false
}
}
return true
}
// PinsVersion is the household's name for a set of pins: every image as `name:tag`, registry path and
// digest stripped, in compose order ("docmost:0.96.0, postgres:16-alpine, redis:7-alpine"). All of them,
// because a version step can move ANY service — a PostgreSQL 16 → 18 step leaves the app's own tag as it
// was, and a label naming only the first image would read the same on both sides of it (seen on the
// first draft of the mismatch sentence: „(0.96.0) … (0.96.0)").
func PinsVersion(pins []string) string {
out := make([]string, 0, len(pins))
for _, ref := range pins {
if i := strings.Index(ref, "@"); i >= 0 {
ref = ref[:i]
}
if i := strings.LastIndex(ref, "/"); i >= 0 {
ref = ref[i+1:]
}
out = append(out, ref)
}
return strings.Join(out, ", ")
}
// recordOffsiteDataAt remembers the data time of the unit just pushed off-site (v0.275.0, R-696).
func (m *Manager) recordOffsiteDataAt(stackName, unitDir string) {
if m.settings == nil {
return
}
t, ok := unitNewestArtifact(unitDir)
if !ok {
return
}
now := time.Now().UTC().Format(time.RFC3339)
if err := m.settings.SetOffsiteDataAt(stackName, settings.OffsiteDataRecord{PushedAt: now, DataAt: t.UTC().Format(time.RFC3339)}); err != nil {
m.logger.Printf("[WARN] [offbox] %s: recording the pushed copy's data time failed: %v — the off-site copy is dated by its snapshot", stackName, err)
}
}
// offsiteDataTime caps an off-site snapshot's time with the data time this box recorded when it pushed
// it. A snapshot NEWER than the last recorded push (taken by another box, or a record lost) keeps its
// own time: the cap applies only to a copy this box knows the content of.
func (m *Manager) offsiteDataTime(stackName string, snapshotAt time.Time) time.Time {
if m.settings == nil {
return snapshotAt
}
rec, ok := m.settings.GetOffsiteDataAt(stackName)
if !ok {
return snapshotAt
}
pushed, perr := time.Parse(time.RFC3339, rec.PushedAt)
data, derr := time.Parse(time.RFC3339, rec.DataAt)
if perr != nil || derr != nil || snapshotAt.After(pushed) || !data.Before(snapshotAt) {
return snapshotAt
}
return data
}
// UnitDumpStamp is one stamped database dump of an app's own unit (v0.275.0).
type UnitDumpStamp struct {
File string
At time.Time
Images map[string]string
}
// UnitDumpStamps returns the app's OWN unit's database dumps as its manifest records them — only when
// the unit's data is known (a `data` block); an unstamped unit returns nothing, and a release that needs
// one waits (fail closed).
func (m *Manager) UnitDumpStamps(stackName string) []UnitDumpStamp {
man := readManifest(UnitManifestFile(m.primaryUnitDirFor(stackName)))
if man == nil || man.Data == nil {
return nil
}
var out []UnitDumpStamp
for rel, st := range man.Data.Files {
if !strings.HasPrefix(rel, "db-dumps/") || !strings.HasSuffix(rel, ".sql") {
continue
}
t, err := time.Parse(time.RFC3339Nano, st.At)
if err != nil {
continue
}
out = append(out, UnitDumpStamp{File: rel, At: t, Images: st.Images})
}
sort.Slice(out, func(i, j int) bool { return out[i].File < out[j].File })
return out
}
// definitionPins are the pins of the definition a capture copies into compose/: the stack dir's own
// docker-compose.yml, parsed. ONE source for the stamps, the fold and the freeze, so "the data's pins"
// and "the pins compose/ holds" are read from the same file the restore will start (production's
// RecoveryInfo.ImagePins is that parse too; a provider that says otherwise cannot make them disagree).
func definitionPins(info RecoveryInfo) []string {
if info.StackDir != "" {
if p := ParseComposeImages(filepath.Join(info.StackDir, "docker-compose.yml")); len(p) > 0 {
return p
}
}
return info.ImagePins
}
+3
View File
@@ -1372,6 +1372,9 @@ func (m *Manager) runOffboxInternal(ctx context.Context, apps, base, env []strin
continue
}
res.backedUp++
// v0.275.0 (R-696): what the snapshot just taken HOLDS is the unit's data, of the unit's data time —
// recorded so the update's precondition dates the off-site copy by its data, not by the snapshot.
m.recordOffsiteDataAt(stack, src)
// R-412 leg 1 — A PUSH THAT CARRIED NOTHING MUST NOT READ AS A PLAIN SUCCESS.
//
// Measured on demo-hp 2026-08-31: a recovery unit was destroyed mid-run, the capture rebuilt it
@@ -91,6 +91,13 @@ type OffsiteReconstituteResult struct {
// than reporting a bare success — a warning beside a success is read as a success, so the
// difference has to survive into the message.
Placement PlacementCheck
// v0.275.0 (R-696, `07` §6.6) — the versions the snapshot's data belongs to (UnitRestoreResult's
// fields, same meaning). VersionChanged: the app came back at the SNAPSHOT's version, its definition
// written from the snapshot's unit, because the live one was another version.
DataPins []string
DataAt time.Time
VersionChanged bool
VersionsUnknown bool
}
// fullPlaceCopier returns the FULL-restore file copier (nil seam → rsyncRestoreOverwrite).
@@ -693,12 +700,50 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
return res, util.MsgError("err.backup.adatbazis_masolat_csonka_nem_indult", stack)
}
// --- WHICH VERSION COMES BACK (v0.275.0, R-696, `07` §6.6) ---------------------------------
// Until v0.275.0 this path never wrote the definition, so after any update the snapshot's data
// (last night's version) was started by the app's NEW definition — for an engine step that is the
// measured A1 failure (a PostgreSQL 16 datadir under 18: refused, the app left down). Now the
// snapshot's data comes back with the definition it belongs to: when the snapshot records its data's
// versions and they differ from what runs, the snapshot unit's definition is written into the stack
// dir (and pinned) before anything is started, and the normal guarded update climbs from there. A
// snapshot without that record restores as before, WARNed. Decided before the first mutation.
scratchCompose := UnitComposeDir(scratchUnit)
dataPins, verr := unitVersionCheck(stack, man, scratchCompose)
if verr != nil {
m.logger.Printf("[ERROR] [offbox] Restore REFUSED for %s: the snapshot's definition does not belong to its data — nothing was touched", stack)
return res, verr
}
var snapEnv map[string]string
defineFromSnapshot := false
if dataPins == nil {
res.VersionsUnknown = true
m.logger.Printf("[WARN] [offbox] %s: snapshot %s does not record which versions wrote its data (taken before v0.275.0) — restoring into the app's current definition, as before", stack, id)
} else {
res.DataPins = dataPins
res.DataAt, _ = man.Data.DataTime()
if info, ok := m.stackProvider.GetStackRecoveryInfo(stack); ok && len(definitionPins(info)) > 0 && !samePins(definitionPins(info), dataPins) {
env, _, eerr := m.unitRestoreEnv(stack, scratchCompose, man)
if eerr != nil {
return res, eerr
}
snapEnv, defineFromSnapshot, res.VersionChanged = env, true, true
m.logger.Printf("[INFO] [offbox] %s: snapshot %s holds data of %v (written %s); the app runs %v — it comes back at the snapshot's version", stack, id, dataPins, man.Data.At, definitionPins(info))
}
}
// --- WHICH SERVICE HOLDS THE DATABASE (R-47) ------------------------------------------------
// Read from the LIVE compose, not the scratch one: reconstitution never overwrites the stack dir,
// so the live file is what `docker compose up` will actually act on. Resolved BEFORE the first
// mutation so the refusal below costs nothing.
// Read from the compose that will RUN: the live one, or — when the app comes back at the snapshot's
// version — the snapshot unit's, which is written into the stack dir before the first start. Resolved
// BEFORE the first mutation so the refusal below costs nothing.
var dbServices []string
if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
if defineFromSnapshot {
svcs, dsErr := DBServiceNames(filepath.Join(scratchCompose, "docker-compose.yml"))
if dsErr != nil {
m.logger.Printf("[WARN] [offbox] %s: could not read the snapshot's compose services: %v", stack, dsErr)
}
dbServices = svcs
} else if composePath, cOK := m.stackProvider.GetStackComposePath(stack); cOK && composePath != "" {
svcs, dsErr := DBServiceNames(composePath)
if dsErr != nil {
// "cannot tell" is not "no database" — leave dbServices empty and let the gate refuse.
@@ -754,6 +799,16 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
if err := m.stackProvider.StopStack(stack); err != nil {
m.logger.Printf("[WARN] [offbox] could not stop %s before reconstitution: %v (continuing)", stack, err)
}
// v0.275.0: the snapshot's own definition, written while nothing of the app's data has been touched
// yet — a failure here restarts the app as it was.
if defineFromSnapshot {
if err := m.stackProvider.RecreateStackDefinitionFromUnit(stack, scratchCompose, snapEnv); err != nil {
if sErr := restartStack(); sErr != nil {
m.logger.Printf("[WARN] [offbox] %s: restart after a failed definition write also failed: %v", stack, sErr)
}
return res, fmt.Errorf("restoring %s: writing the snapshot's definition failed: %w", stack, err)
}
}
copier := m.fullPlaceCopier()
for _, pl := range placements {
if pl.isUnit {
+57 -3
View File
@@ -69,6 +69,11 @@ type RecoveryManifest struct {
// never claim a coherence it did not establish — it carries the prior stamp forward instead.
OffsiteRunID string `json:"offsite_run_id,omitempty"`
DumpsAt string `json:"dumps_at,omitempty"` // RFC3339 UTC — when this run's dump leg finished
// Data (v0.275.0, R-696) is the account of the DATA this unit holds: its time and the versions that
// wrote it (data_versions.go). compose/ holds the definition THESE pins name; ImagePins above are the
// app's CURRENT pins — the two differ between an update and the next data run. Nil = unknown (a unit
// written before v0.275.0, or one with an unstamped data file): it restores as before, with a WARN.
Data *UnitData `json:"data,omitempty"`
}
// SetVersion records the controller version stamped into recovery-unit manifests.
@@ -92,7 +97,15 @@ func (m *Manager) SetTier2Notifier(fn func(stackName, destLabel string, dur time
// Idempotent: it builds the captured content in memory first and SKIPS all writes when the unit is
// already current (same config checksums, same dump set, same controller version) — so it can run on
// the periodic status refresh without thrashing a spinning USB drive.
//
// v0.275.0 (R-696): CaptureRecoveryUnit is the capture AFTER a data run (it follows the legs in
// RunAppBackupNow). The periodic refresh calls captureRecoveryUnit(…, false), which never moves the
// unit's data time and never rewrites the definition away from the data it belongs to.
func (m *Manager) CaptureRecoveryUnit(stackName string) error {
return m.captureRecoveryUnit(stackName, true)
}
func (m *Manager) captureRecoveryUnit(stackName string, dataRun bool) error {
if m.stackProvider == nil {
return fmt.Errorf("no stack provider")
}
@@ -161,6 +174,38 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
manifestPath := RecoveryUnitManifestPath(nsRoot, stackName)
cur := readManifest(manifestPath)
unitDir := RecoveryUnitPath(nsRoot, stackName)
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
// v0.275.0 (R-696): the data's own account, folded from the stamps the legs wrote.
var prevData *UnitData
if cur != nil {
prevData = cur.Data
}
curPins := definitionPins(info)
data := foldUnitData(unitDir, dbDumps, volDumps, dataRun, time.Now(), curPins, info.InstalledImages, prevData)
// THE DEFINITION STAYS WITH ITS DATA. When the app's pins have moved since the data was written (an
// update between two data runs), compose/ keeps the definition the data belongs to until the next data
// run replaces the data — the refresh used to rewrite it here within five minutes of every update,
// pairing the new version with the old data (A1: a 16 datadir under an 18 definition; the restore left
// the app down). The checksums then describe what compose/ really holds, so the already-current check
// below does not rewrite the manifest on every refresh.
frozen := data != nil && !data.Mixed && !samePins(data.ImagePins, curPins)
if frozen {
files, checksums, configFiles = nil, map[string]string{}, nil
for _, fname := range []string{"docker-compose.yml", ".felhom.yml", "app.yaml"} {
b, err := os.ReadFile(filepath.Join(composeDir, fname))
if err != nil {
continue
}
checksums[fname] = sha256Hex(b)
configFiles = append(configFiles, fname)
}
if held := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")); !samePins(held, data.ImagePins) {
m.logger.Printf("[WARN] [backup] %s: the unit's definition %v matches neither its data %v nor the app %v — kept as it is; a restore of this unit will refuse", stackName, held, data.ImagePins, curPins)
}
}
// R-43/R-44: the coherence stamp of the offsite run currently in flight ("" on the periodic
// refresh and on the local dump run). When empty we CARRY THE PRIOR STAMP FORWARD rather than
@@ -193,11 +238,16 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
stringMapEqual(cur.Checksums, checksums) &&
stringSliceEqual(cur.DBDumps, dbDumps) &&
stringSliceEqual(cur.VolumeDumps, volDumps) &&
stringSliceEqual(cur.ImagePins, info.ImagePins) &&
unitDataEqual(cur.Data, data) &&
cur.OffsiteRunID == runID {
return nil
}
composeDir := RecoveryUnitComposePath(nsRoot, stackName)
if frozen && (cur == nil || cur.Data == nil || stringSliceEqual(cur.ImagePins, cur.Data.ImagePins)) {
m.logger.Printf("[INFO] [backup] %s: the app now runs %v; the unit keeps the definition of its data (%v, written %s) until the next backup replaces the data",
stackName, curPins, data.ImagePins, data.At)
}
if err := os.MkdirAll(composeDir, 0755); err != nil {
return fmt.Errorf("creating recovery-unit compose dir: %w", err)
}
@@ -226,6 +276,10 @@ func (m *Manager) CaptureRecoveryUnit(stackName string) error {
Checksums: checksums,
OffsiteRunID: runID,
DumpsAt: dumpsAt,
Data: data,
}
if data == nil && (len(dbDumps)+len(volDumps)) > 0 && m.isDebug() {
m.logger.Printf("[DEBUG] [backup] %s: the unit's data files are not all stamped — its versions are unknown until the next backup", stackName)
}
if err := writeManifest(manifestPath, manifest); err != nil {
return fmt.Errorf("writing manifest: %w", err)
@@ -387,7 +441,7 @@ func (m *Manager) readUnitSpace(stackName string) *UnitSpace {
// volume-dump legs of this run already consulted for this app. When a run is in flight the answer
// here is a memo lookup — an app refused before its first write is refused here too, silently,
// because it was already alerted once. Outside a run (the periodic status refresh) it decides fresh.
func (m *Manager) captureAllRecoveryUnits() {
func (m *Manager) captureAllRecoveryUnits(dataRun bool) {
if m.stackProvider == nil {
return
}
@@ -409,7 +463,7 @@ func (m *Manager) captureAllRecoveryUnits() {
if !m.admitApp(stack.Name) {
continue
}
if err := m.CaptureRecoveryUnit(stack.Name); err != nil {
if err := m.captureRecoveryUnit(stack.Name, dataRun); err != nil {
m.noteFailure(stack.Name, "recovery-unit capture", err.Error())
m.logger.Printf("[WARN] [backup] Recovery unit capture failed for %s: %v", stack.Name, err)
// R-158: per app, and the loop CONTINUES — one app's failure must not silence the
@@ -85,7 +85,7 @@ func newUnitNotifyManager(t *testing.T, stacks []string, fail map[string]bool) (
func TestCaptureAll_FailureNotifiesOnceWithTheSpaceFigures(t *testing.T) {
m, got := newUnitNotifyManager(t, []string{"immich"}, map[string]bool{"immich": true})
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false)
if len(*got) != 1 {
t.Fatalf("got %d unit-failure events, want exactly 1 — a per-app Tier-1 capture failure "+
@@ -117,7 +117,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
[]string{"homebox", "immich", "nextcloud"},
map[string]bool{"immich": true})
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false)
if len(*got) != 1 {
t.Fatalf("got %d events, want exactly 1 — either the loop ABORTED on the middle app "+
@@ -134,7 +134,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
m2, got2 := newUnitNotifyManager(t,
[]string{"homebox", "immich", "nextcloud"},
map[string]bool{"immich": true, "nextcloud": true})
m2.captureAllRecoveryUnits()
m2.captureAllRecoveryUnits(false)
if len(*got2) != 2 {
t.Fatalf("got %d events, want 2 — the app AFTER the first failure was never reached, so the "+
"loop is aborting rather than continuing: %+v", len(*got2), *got2)
@@ -148,7 +148,7 @@ func TestCaptureAll_OneFailureDoesNotAbortOrDuplicate(t *testing.T) {
// to ignore.
func TestCaptureAll_SuccessIsSilent(t *testing.T) {
m, got := newUnitNotifyManager(t, []string{"homebox"}, nil)
m.captureAllRecoveryUnits()
m.captureAllRecoveryUnits(false)
if len(*got) != 0 {
t.Fatalf("a successful capture fired %d event(s): %+v", len(*got), *got)
}
@@ -163,7 +163,7 @@ func TestCaptureAll_UnwiredNotifyDoesNotPanic(t *testing.T) {
systemDataPath: dir,
stackProvider: &unitFailProvider{stacks: []string{"immich"}, fail: map[string]bool{"immich": true}, dir: dir},
}
m.captureAllRecoveryUnits() // no SetUnitNotify — must not panic
m.captureAllRecoveryUnits(false) // no SetUnitNotify — must not panic
}
// §8.4 in the failure direction: an unreadable target filesystem is reported as UNKNOWN, never as
+21 -6
View File
@@ -66,17 +66,32 @@ func (m *Manager) driveLabelForRoot(root string) string {
return ""
}
// unitNewestArtifact is the unit's data time: the newest of its manifest, .sql dumps and .tar
// volume dumps. ONE rule, shared with ListRestorePoints, so the two lists cannot date a unit
// differently.
// unitNewestArtifact is the unit's DATA time. ONE rule, shared by ListRestorePoints (Tier 1), the Tier-2
// copy's date, the removed-app list and kept data, so no two of them can date a unit differently.
//
// v0.275.0 (R-696) — THE TIME OF THE DATA, NEVER OF THE MANIFEST. It used to be the newest of the
// manifest, the .sql dumps and the .tar dumps; a refresh rewrites the manifest when the app's pins move,
// so a unit re-captured two minutes after an update read as two minutes old over data from before the
// update (9202 2026-09-25 11:06; demo-hp 2026-09-26 02:20, where it released the kept pre-conversion
// copy). Now: the manifest's `data.at` when the data is stamped; else the newest DATA file (the undo
// copies `pre-restore-*` excluded — an update's own safety dump is not a backup); the manifest's time
// only for a unit that holds no data file at all, whose whole content is its definition.
func unitNewestArtifact(unitDir string) (time.Time, bool) {
fi, err := os.Stat(UnitManifestFile(unitDir))
if err != nil {
return time.Time{}, false
}
newest := fi.ModTime()
newest = newestArtifact(UnitDBDumpDir(unitDir), ".sql", newest)
newest = newestArtifact(UnitVolumeDumpDir(unitDir), ".tar", newest)
if man := readManifest(UnitManifestFile(unitDir)); man != nil {
if t, ok := man.Data.DataTime(); ok {
return t, true
}
}
var newest time.Time
newest = newestDataFile(UnitDBDumpDir(unitDir), ".sql", newest)
newest = newestDataFile(UnitVolumeDumpDir(unitDir), ".tar", newest)
if newest.IsZero() {
return fi.ModTime(), true
}
return newest, true
}
+9 -12
View File
@@ -31,9 +31,7 @@ const restorePointShortID = "helyi"
// known at all (found=false → the caller should 404). A known stack with no recovery unit on
// disk returns an EMPTY list (a valid answer — "no backup yet"), not an error.
//
// The single point's Time is the newest mtime among the unit's artifacts (manifest.json,
// db-dumps/*.sql, volume-dumps/*.tar): the manifest is only rewritten when the app's config
// changes (checksum-skip), so the nightly-refreshed dumps are usually the freshest artifact.
// The single point's Time is the unit's DATA time (unitNewestArtifact, v0.275.0 — R-696).
func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, found bool) {
if m.stackProvider == nil {
return nil, false
@@ -61,13 +59,11 @@ func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, fo
return []RestorePoint{}, true
}
fi, err := os.Stat(RecoveryUnitManifestPath(nsRoot, stackName))
if err != nil {
// v0.275.0 (R-696): the unit's DATA time (unitNewestArtifact), never the manifest's refresh time.
newest, ok := unitNewestArtifact(RecoveryUnitPath(nsRoot, stackName))
if !ok {
return []RestorePoint{}, true // no recovery unit yet — "no backup" is a valid answer
}
newest := fi.ModTime()
newest = newestArtifact(AppDBDumpPath(nsRoot, stackName), ".sql", newest)
newest = newestArtifact(AppVolumeDumpPath(nsRoot, stackName), ".tar", newest)
return []RestorePoint{{
Time: newest.UTC().Format(time.RFC3339),
@@ -77,15 +73,16 @@ func (m *Manager) ListRestorePoints(stackName string) (points []RestorePoint, fo
}}, true
}
// newestArtifact returns the newest mtime among cur and the files with the given extension in
// dir (non-recursive; a missing dir contributes nothing).
func newestArtifact(dir, ext string, cur time.Time) time.Time {
// newestDataFile returns the newest mtime among cur and the DATA files with the given extension in dir
// (non-recursive; a missing dir contributes nothing). The undo copies (`pre-restore-*`) are not data of
// the unit (R-361) and never date it.
func newestDataFile(dir, ext string, cur time.Time) time.Time {
entries, err := os.ReadDir(dir)
if err != nil {
return cur
}
for _, e := range entries {
if e.IsDir() || !strings.HasSuffix(e.Name(), ext) {
if e.IsDir() || !strings.HasSuffix(e.Name(), ext) || strings.HasPrefix(e.Name(), preRestoreDumpPrefix) {
continue
}
if info, err := e.Info(); err == nil && info.ModTime().After(cur) {
+122 -49
View File
@@ -1,6 +1,7 @@
package backup
import (
"errors"
"fmt"
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
"os"
@@ -166,6 +167,40 @@ type UnitRestoreResult struct {
// statement, and precisely the R-88 failure direction (degrade to NO DATA rather than to UNKNOWN)
// this whole change exists to remove. An unknown must be carried, never drawn as a zero.
CountsUnknown bool
// DataPins / DataAt (v0.275.0, R-696) — the versions the restored DATA belongs to and when it was
// written, from the unit's `data` block; the definition started with it is the one they name. Empty
// when the unit's versions are unknown (VersionsUnknown).
DataPins []string
DataAt time.Time
// VersionChanged — the app now runs DataPins, which differ from what it ran before the restore (a
// restore of an older version). The page then says which version came back (`07` §6.6).
VersionChanged bool
// VersionsUnknown — a unit written before v0.275.0 (or with an unstamped data file): restored as
// before, with a WARN naming it.
VersionsUnknown bool
}
// ErrUnitVersionMismatch is the KIND of the refusal of a restore whose unit would start its data with a
// definition the data does not belong to (v0.275.0, R-696): a unit whose files were written under different
// pins (Mixed), or whose compose/ names other pins than its data. Raised BEFORE anything is touched;
// branch with errors.Is. The customer sentence is the bundle's.
var ErrUnitVersionMismatch = errors.New("the recovery unit's definition is not the one its data belongs to")
// unitVersionCheck is A3's rule, as a pure function of the manifest and the unit's own compose file:
// known and matching → the pins to report; unknown → (nil, nil) and the caller WARNs; mixed or
// mismatched → the refusal.
func unitVersionCheck(stackName string, man *RecoveryManifest, composeDir string) ([]string, error) {
if man == nil || man.Data == nil {
return nil, nil
}
if man.Data.Mixed {
return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_versions_mixed", stackName)
}
def := ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml"))
if !samePins(def, man.Data.ImagePins) {
return nil, util.MsgErrorf(ErrUnitVersionMismatch, "err.backup.unit_version_mismatch", stackName, PinsVersion(def), PinsVersion(man.Data.ImagePins))
}
return man.Data.ImagePins, nil
}
// RestoreFromRecoveryUnit recreates an app from its on-drive recovery unit.
@@ -323,60 +358,35 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
res.ManifestVolumes, res.ManifestDBs = len(manifest.VolumeDumps), len(manifest.DBDumps)
composeDir := UnitComposeDir(unitDir)
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
// reconcileRestoreSecrets), then the fail-closed gate.
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
fullEnv, missing, err := reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
if err != nil {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
return res, err
// v0.275.0 (R-696, `07` §6.6) — A RESTORE NEVER STARTS DATA WITH A DEFINITION IT DOES NOT BELONG TO.
// The unit keeps the definition of its data (data_versions.go); this refuses, before anything is
// touched, a unit where the two disagree. Measured before the fix (A1, 9202): the unit's PostgreSQL 18
// definition over a 16 datadir tar — the tar replaced the live volume, 18 refused it, the app stayed down.
dataPins, verr := unitVersionCheck(stackName, manifest, composeDir)
if verr != nil {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: the unit's definition %v, its data %v (mixed=%v) — a restore never starts data with a definition it does not belong to; nothing was touched",
stackName, ParseComposeImages(filepath.Join(composeDir, "docker-compose.yml")), manifest.Data.ImagePins, manifest.Data.Mixed)
return res, verr
}
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
// never generated: the fail-closed gate above already refused if one was missing, and the
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
//
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
// generator run means the secret was empty at capture AND absent from the guest.
//
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
// regenerated value leaves the app unable to authenticate against its own restored rows while the
// dump replay, which uses the container's local trust socket, still reports success. The old wording
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
// DB/root-password fields and is now scoped to what is actually true.
if len(missing) > 0 {
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
for _, dk := range manifest.DataKeyEnvVars {
dataKeySet[dk] = true
}
var generated, unresolved []string
for _, name := range missing {
if !dataKeySet[name] && m.generateSecret != nil {
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
fullEnv[name] = v
generated = append(generated, name)
continue
}
if dataPins == nil {
res.VersionsUnknown = true
m.logger.Printf("[WARN] [backup] Restore %s from %s: the unit does not record which versions wrote its data (written before v0.275.0, or an unstamped file) — restoring its definition as captured, as before", stackName, unitDir)
} else {
res.DataPins = dataPins
res.DataAt, _ = manifest.Data.DataTime()
if info, ok := m.stackProvider.GetStackRecoveryInfo(stackName); ok {
if live := definitionPins(info); len(live) > 0 && !samePins(live, dataPins) {
res.VersionChanged = true
m.logger.Printf("[INFO] [backup] Restore %s: the data belongs to %v (written %s); the app ran %v — it comes back at its data's version, and the update climbs from there",
stackName, dataPins, manifest.Data.At, live)
}
unresolved = append(unresolved, name)
}
if len(generated) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
stackName, generated)
}
if len(unresolved) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
stackName, len(unresolved), unresolved)
}
}
fullEnv, missing, err := m.unitRestoreEnv(stackName, composeDir, manifest)
if err != nil {
return res, err
}
// R-102: the unit DIRECTORY is logged. Which copy a restore read from is now a real question with
// two answers, and "an absent log line is not evidence" — the drill reads this line to prove the
// secondary mirror, not the primary unit, was the source. It is a path, never a secret.
@@ -472,3 +482,66 @@ func (m *Manager) RestoreFromRecoveryUnitAtWith(stackName, unitDir string, opt U
m.clearUpdateHoldAfterRestore(stackName)
return res, nil
}
// unitRestoreEnv rebuilds the env a restore starts the unit's definition with: the unit's plain config,
// its portable secrets (D5), the guest's for the withheld class, the fail-closed data-key gate, and a
// generated replacement for a missing RESETTABLE secret (O4). ONE implementation for the unit restore and,
// since v0.275.0, the off-site restore that brings an app back at its snapshot's version. Values are
// never logged.
func (m *Manager) unitRestoreEnv(stackName, composeDir string, manifest *RecoveryManifest) (fullEnv map[string]string, missing []string, err error) {
nonSecretEnv, unitSecrets := readUnitEnv(filepath.Join(composeDir, "app.yaml"), manifest.PortableSecretEnvVars)
// D5: the unit carries the portable class, so this is the leg that no longer needs the guest. The
// guest is still consulted for the WITHHELD class (internet-reachable admin logins) and as the
// fallback for a schema-1 unit — it returns an empty map when the guest is gone, which is the whole
// point: a Tier-1/2 restore must survive that. Precedence is unit-over-guest (see
// reconcileRestoreSecrets), then the fail-closed gate.
guestSecrets := m.stackProvider.RecoverStackSecrets(stackName, manifest.SecretEnvVars)
fullEnv, missing, err = reconcileRestoreSecrets(nonSecretEnv, unitSecrets, guestSecrets, manifest.SecretEnvVars, manifest.DataKeyEnvVars)
if err != nil {
m.logger.Printf("[ERROR] [backup] Restore REFUSED for %s: %v", stackName, err)
return nil, missing, err
}
// O4: a missing RESETTABLE secret used to redeploy blank (compose "Defaulting to a blank
// string" → exit 1). Generate a replacement via the deploy flow's generator instead —
// RecreateStackFromUnit persists fullEnv through SaveAppConfig, so the new value lands
// encrypted in the guest app.yaml and round-trips on the next backup/restore. Data-keys are
// never generated: the fail-closed gate above already refused if one was missing, and the
// generator itself refuses data-key fields (defense-in-depth). Values are never logged.
//
// D5 shrinks this path to the rare case: the portable class now comes from the unit, so a
// generator run means the secret was empty at capture AND absent from the guest.
//
// It does NOT claim the reset is harmless. R-127: for a DB password it is not — a restored data
// directory keeps the OLD role hash (POSTGRES_PASSWORD is ignored once PGDATA is non-empty), so a
// regenerated value leaves the app unable to authenticate against its own restored rows while the
// dump replay, which uses the container's local trust socket, still reports success. The old wording
// here asserted "stored data is unaffected" for every non-data-key secret; that is false for the 18
// DB/root-password fields and is now scoped to what is actually true.
if len(missing) > 0 {
dataKeySet := make(map[string]bool, len(manifest.DataKeyEnvVars))
for _, dk := range manifest.DataKeyEnvVars {
dataKeySet[dk] = true
}
var generated, unresolved []string
for _, name := range missing {
if !dataKeySet[name] && m.generateSecret != nil {
if v, ok := m.generateSecret(stackName, name); ok && v != "" {
fullEnv[name] = v
generated = append(generated, name)
continue
}
}
unresolved = append(unresolved, name)
}
if len(generated) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: generated replacement for %v — the credential was reset (old value unrecoverable); no data-encrypting key was involved, but a regenerated DATABASE password will not match the restored data directory's stored hash (R-127) — check the app can reach its data",
stackName, generated)
}
if len(unresolved) > 0 {
m.logger.Printf("[WARN] [backup] Restore %s: %d resettable secret(s) unrecoverable and have no generator %v — proceeding, but the app may fail to start until the credential is set manually",
stackName, len(unresolved), unresolved)
}
}
return fullEnv, missing, nil
}
@@ -24,7 +24,7 @@ func (h *admissionHarness) digestOf(kind string) *RunSummary {
doneAdm := h.m.beginAdmissionRun()
doneSum := h.m.beginRunSummary(kind, "run-test")
h.m.runVolumeDumps()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
h.m.emitRunSummary()
doneSum()
doneAdm()
@@ -153,7 +153,7 @@ func TestRunSummary_RefreshSweepHasNoRunID(t *testing.T) {
defer h.m.beginAdmissionRun()()
defer h.m.beginRunSummary(runKindRefresh, "")()
defer h.m.emitRunSummary()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
}()
if got == nil {
t.Fatal("the periodic sweep emitted no digest — with the per-app event now record-only, a " +
@@ -148,7 +148,7 @@ func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) {
if len(h.volDumped) != 1 || h.volDumped[0] != "free" {
t.Errorf("positive control: the unheld app must still be dumped, got %v", h.volDumped)
}
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
for _, n := range h.prov.infoHits {
if n == "held" {
t.Error("the capture must not rewrite a HELD app's restore point")
@@ -178,7 +178,7 @@ func TestSlice4_NightlyLegsLeaveAnAppMidUpdateAlone(t *testing.T) {
h.m.settings = slice4Settings(t)
h.m.SetUpdatingCheck(func(name string) bool { return name == "updating" })
h.m.runVolumeDumps()
h.m.captureAllRecoveryUnits()
h.m.captureAllRecoveryUnits(false)
var mirrored []string
h.m.perAppTier2 = func(name string) error { mirrored = append(mirrored, name); return nil }
h.m.RunAllTier2()
+16 -4
View File
@@ -43,6 +43,9 @@ type Tier2RestorePoint struct {
PackagePreserved bool
// CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded.
CopyLastSuccess string
// DataDate (v0.275.0, R-696) — the mirrored unit's DATA time (unitNewestArtifact on the mirror): when
// the data the copy holds was written, which a mirror run copies but never makes newer. "" = unknown.
DataDate string
}
// restorePointFromCoverage is the pure half of the predicate.
@@ -54,6 +57,7 @@ func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint {
CopyDateProven: cov.CopyLastSuccess != "",
PackagePreserved: preserved,
CopyLastSuccess: cov.CopyLastSuccess,
DataDate: cov.UnitDataDate,
}
}
@@ -93,6 +97,12 @@ func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
if err != nil {
return time.Time{}, false
}
// v0.275.0 (R-696): a mirror run copies the unit's data; it never makes the data newer. When the
// mirror's data time is known and older than the copy, the data time is the copy's age — a mirror
// taken right after an update, of a unit whose dump is from before it, is as old as that dump.
if d, derr := time.Parse(time.RFC3339, p.DataDate); derr == nil && d.Before(t) {
return d, true
}
return t, true
}
@@ -209,8 +219,9 @@ var updateOffsiteCheckTimeout = 15 * time.Second
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
type UpdateTierPoint struct {
Tier int
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime (capped by the
// mirror's data time), Tier 1 the unit's DATA time (ListRestorePoints → unitNewestArtifact), Tier 3
// the newest snapshot, capped by the data time the box recorded when it pushed it (v0.275.0, R-696).
At time.Time
}
@@ -281,7 +292,7 @@ func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier in
return UpdateTierPoint{}, false
}
if at, ok := got[stackName]; ok && !at.IsZero() {
return UpdateTierPoint{Tier: tier, At: at}, true
return UpdateTierPoint{Tier: tier, At: m.offsiteDataTime(stackName, at)}, true
}
}
return UpdateTierPoint{}, false
@@ -391,11 +402,12 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
if db.StackName != stackName {
continue
}
res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug())
res := m.dumpOneOrDefault(ctx, db, AppDBDumpPath(nsRoot, stackName))
if res.Error != nil {
return util.MsgError("err.backup.adatbazis_mentes_sikertelen", db.ContainerName, res.Error)
}
dumped++
m.stampDataFile(stackName, RecoveryUnitPath(nsRoot, stackName), "db-dumps/"+filepath.Base(res.FilePath))
m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size))
}
+6 -1
View File
@@ -2454,5 +2454,10 @@
"err.kept.not_listed": "This is not kept data, so nothing was touched.",
"err.kept.occupied": "The app's folder already holds data, so the kept files were not put back. Nothing changed.",
"api.kept.busy": "A backup or a restore is running now. Try again when it has finished.",
"page.title.kept_data": "Kept data"
"page.title.kept_data": "Kept data",
"note.restore.older_version": "%s is back from the backup of %s, at version %s. An update is available: the box brings it up to date one step at a time.",
"note.restore.older_version_no_step": "%s is back from the backup of %s, at version %s. No tested update step leads on from this version, so the box does not update it by itself.",
"err.backup.unit_versions_mixed": "%s: the data in this backup was written by different versions, so the restore did not start. The app is untouched; the copy on the second drive or off-site can restore it.",
"err.backup.unit_version_mismatch": "%s: this backup's definition (%s) does not belong to the version that wrote its data (%s), so the restore did not start. The app is untouched.",
"deploy.login_from_backup": "Restored from a backup: log in with the password that was valid when the backup was taken. The value stored here is not it, so it is not shown."
}
+6 -1
View File
@@ -2442,5 +2442,10 @@
"err.kept.not_listed": "Ez nem megőrzött adat, ezért nem nyúltunk hozzá.",
"err.kept.occupied": "Az alkalmazás mappájában már vannak adatok, ezért a megőrzött fájlokat nem tettük vissza. Nem változott semmi.",
"api.kept.busy": "Most egy mentés vagy visszaállítás fut. Ha befejeződött, próbáld újra.",
"page.title.kept_data": "Megőrzött adatok"
"page.title.kept_data": "Megőrzött adatok",
"note.restore.older_version": "A(z) %s visszaállt a(z) %s-i mentésből, a(z) %s verzióra. Elérhető frissítés: a doboz lépésenként hozza naprakészre.",
"note.restore.older_version_no_step": "A(z) %s visszaállt a(z) %s-i mentésből, a(z) %s verzióra. Ettől a verziótól nem vezet kipróbált frissítési lépés, ezért a doboz magától nem frissíti.",
"err.backup.unit_versions_mixed": "A(z) %s mentésében az adatokat különböző verziók írták, ezért a visszaállítás nem indult el. Az alkalmazás érintetlen; a második meghajtón vagy a távoli helyen lévő másolatból visszaállítható.",
"err.backup.unit_version_mismatch": "A(z) %s mentésében a beállítás (%s) nem ahhoz a verzióhoz tartozik, amelyik az adatokat írta (%s), ezért a visszaállítás nem indult el. Az alkalmazás érintetlen.",
"deploy.login_from_backup": "Mentésből töltötted vissza: a belépéshez a mentés idején érvényes jelszavad kell. Az itt tárolt érték nem az, ezért nem mutatjuk."
}
+13 -3
View File
@@ -145,7 +145,17 @@ func RenderCloudflared(d CloudflaredData) (map[string]FileSpec, error) {
// RenderFileBrowserCompose returns FileBrowser's docker-compose.yml for the given domain and storage
// volume-mount lines. Ported verbatim from internal/web/handlers.go (the single source of truth now
// lives here so the pinned image can't diverge between bring-up and the web storage-sync path).
func RenderFileBrowserCompose(domain string, storageMounts []string) string {
func RenderFileBrowserCompose(domain string, storageMounts []string, groupAdd ...int) string {
// R-691 (v0.275.0): supplementary groups — the owning groups of kept folders the view must READ (a
// 0770 folder another user owns, e.g. nextcloud's www-data). Their binds are `:ro`. Absent → the
// compose is byte-for-byte what it was.
groupSection := ""
if len(groupAdd) > 0 {
groupSection = "\n # Kept data's owning groups (auto-generated): the read-only view reads a folder another user owns.\n group_add:"
for _, g := range groupAdd {
groupSection += fmt.Sprintf("\n - \"%d\"", g)
}
}
storageSection := ""
if len(storageMounts) > 0 {
storageSection = "\n # Storage paths (auto-generated by felhom-controller)\n" +
@@ -165,7 +175,7 @@ services:
# setgid), letting the content apps (group 1000) write into them. The gtstef/filebrowser image is a
# single Go binary (entrypoint ./filebrowser) and does NOT honor a UMASK env (verified: -e UMASK=002
# leaves PID1 at 0022), so we wrap the entrypoint to set the process umask before exec.
entrypoint: ["sh", "-c", "umask 002; exec /home/filebrowser/filebrowser"]
entrypoint: ["sh", "-c", "umask 002; exec /home/filebrowser/filebrowser"]%s
environment:
- TZ=Europe/Budapest
- FILEBROWSER_CONFIG=/home/filebrowser/config.yaml
@@ -198,7 +208,7 @@ volumes:
networks:
traefik-public:
external: true
`, domain, FileBrowserImage, storageSection, domain)
`, domain, FileBrowserImage, groupSection, storageSection, domain)
}
// RenderControllerRoute returns a traefik file-provider dynamic config routing the controller's own
+31
View File
@@ -183,6 +183,12 @@ type Settings struct {
// Per-app backup preferences
AppBackup map[string]AppBackupPrefs `json:"app_backup,omitempty"`
// OffsiteDataAt (v0.275.0, R-696) — per app, the DATA time of the recovery unit this box last pushed
// off-site, and when. An off-site snapshot's own time is when it was TAKEN; a run whose dump leg failed
// still pushes the older dumps, so the snapshot time can overstate its data. The update's precondition
// caps the off-site copy's age with this record.
OffsiteDataAt map[string]OffsiteDataRecord `json:"offsite_data_at,omitempty"`
// Customer-configurable backup-window start "HH:MM" (v0.168.0). "" = use controller.yaml
// db_dump_schedule (then the "02:30" default). Every nightly leg derives from this at fixed
// offsets; overrides yaml when a valid value is present (mirrors PasswordHash precedence).
@@ -1202,6 +1208,31 @@ func (s *Settings) SetClaimConsumedGeneration(gen int) error {
return s.save()
}
// OffsiteDataRecord is one app's entry in Settings.OffsiteDataAt. RFC3339 UTC both.
type OffsiteDataRecord struct {
PushedAt string `json:"pushed_at"`
DataAt string `json:"data_at"`
}
// GetOffsiteDataAt returns the app's record, false when none was written.
func (s *Settings) GetOffsiteDataAt(app string) (OffsiteDataRecord, bool) {
s.mu.RLock()
defer s.mu.RUnlock()
r, ok := s.OffsiteDataAt[app]
return r, ok
}
// SetOffsiteDataAt records the app's pushed data time and persists to disk.
func (s *Settings) SetOffsiteDataAt(app string, r OffsiteDataRecord) error {
s.mu.Lock()
defer s.mu.Unlock()
if s.OffsiteDataAt == nil {
s.OffsiteDataAt = make(map[string]OffsiteDataRecord)
}
s.OffsiteDataAt[app] = r
return s.save()
}
// GetDBValidations returns a copy of the cached DB validations.
func (s *Settings) GetDBValidations() map[string]DBValidationCache {
s.mu.RLock()
+39
View File
@@ -169,6 +169,11 @@ type AppConfig struct {
// ConversionCopy (v0.273.0, `09` §6.4 part 10) is the OLD datadir's copy kept after a successful
// PostgreSQL major conversion, until a backup of the converted app is proven (ReleaseConversionCopies).
ConversionCopy *ConversionCopy `yaml:"conversion_copy,omitempty" json:"conversion_copy,omitempty"`
// RestoredLogins (v0.275.0, R-694) are the `type: password` fields whose stored value was GENERATED by a
// restore (the unit never carries an admin login, D5, and the guest had none — a load of kept data, a
// removed app, a rebuilt guest) while the app's own login came back with its data. The page then shows
// no value for them and says the old password is the one that works (restoredLoginFields).
RestoredLogins []string `yaml:"restored_logins,omitempty" json:"restored_logins,omitempty"`
}
// InstalledImage is one compose service's observed image. See AppConfig.InstalledImages.
@@ -684,6 +689,9 @@ func (m *Manager) PersistUnitRedeployConfig(name string, env map[string]string)
stackDir := filepath.Dir(stack.ComposePath)
meta := LoadMetadata(stackDir)
// R-694: which admin logins did the restore have to GENERATE? Exactly the `type: password` fields the
// guest held no value for before this write (the unit never carries one) — read BEFORE it is replaced.
prior := LoadAppConfigDecrypted(stackDir, m.encKey)
cfg := &AppConfig{
Deployed: true,
DeployedAt: time.Now().UTC().Format(time.RFC3339),
@@ -694,6 +702,10 @@ func (m *Manager) PersistUnitRedeployConfig(name string, env map[string]string)
cfg.LockedFields = append(cfg.LockedFields, f.EnvVar)
}
}
cfg.RestoredLogins = restoredLoginFields(name, meta, prior, env)
if len(cfg.RestoredLogins) > 0 {
m.logger.Printf("[INFO] [stacks] %s: the restore generated %v — the app's own login came back with its data; the page will not show the new value as the password", name, cfg.RestoredLogins)
}
if err := SaveAppConfig(stackDir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
return fmt.Errorf("saving app config: %w", err)
}
@@ -1332,3 +1344,30 @@ func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLim
}
return nil, warning
}
// loginAppliedEveryStart is the register of `type: password` fields whose app APPLIES the env value at
// EVERY start, so a value generated at a restore IS the login afterwards (R-694, measured 2026-09-26 from
// each image's entrypoint at the catalog's tag, `audits/version-travel-2026-09-26/D4/`): code-server's
// s6 run script passes $PASSWORD to `code-server --auth password` on every start and stores none. The six
// other apps with such a field (crafty-controller, gokapi, grafana, kimai, nextcloud, paperless-ngx) use it
// only at first initialisation — the restored data's login wins. Code, not a catalog flag, like
// nonPortableSecrets: it decides what a household is told about how to get into its own app.
var loginAppliedEveryStart = map[string]map[string]bool{
"code-server": {"PASSWORD": true},
}
// restoredLoginFields names the `type: password` fields a restore GENERATED — no value in the guest's
// app.yaml before, a value now — for an app whose login lives in its data. Pinned by TestR694_*.
func restoredLoginFields(app string, meta Metadata, prior *AppConfig, env map[string]string) []string {
var out []string
for _, f := range meta.DeployFields {
if f.Type != "password" || env[f.EnvVar] == "" || loginAppliedEveryStart[app][f.EnvVar] {
continue
}
if prior != nil && prior.Env[f.EnvVar] != "" {
continue // the guest kept the household's own value — not generated
}
out = append(out, f.EnvVar)
}
return out
}
+6
View File
@@ -304,6 +304,12 @@ func (m *Manager) ListKept(drives []string) []KeptItem {
continue
}
dir := filepath.Join(d, KeptDirName, a.Name(), s.Name())
// R-695 (v0.275.0): an EMPTY dated folder is not kept data — it is what Docker leaves when a
// file-browser bind outlives a Delete (it recreates the missing source, empty, as root).
// Listed, it was bound again, and the bind recreated it: a loop. Never listed, never bound.
if !dirHasEntries(dir) {
continue
}
it := KeptItem{App: a.Name(), DisplayName: a.Name(), Path: dir, Drive: d, Kind: KeptKindDated,
Marker: readKeptMarker(dir), SizeBytes: sizeFn(dir)}
if st, ok := m.GetStack(a.Name()); ok && st.Meta.DisplayName != "" {
+24
View File
@@ -295,3 +295,27 @@ func TestKept_OwnerIsNeverTheFileBrowser(t *testing.T) {
t.Fatalf("the leftovers must be named by the app that binds them through HDD_PATH, never the file browser: %v", got)
}
}
// R-695 (v0.275.0) — an EMPTY dated kept folder (what Docker recreates when a file-browser bind outlives
// a Delete) is never listed, so it is never bound and cannot recreate itself. A dated folder that holds
// something is still listed.
//
// COMPANION RED-PROOF (REPORT.md): drop the dirHasEntries check in ListKept's dated loop — the empty
// folder is then listed and this fails.
func TestR695_AnEmptyDatedKeptFolderIsNeverListed(t *testing.T) {
m, drive := keptManager(t)
m.mu.Lock()
s := m.stacks["cloudapp"]
s.Deployed = true
s.AppConfig = &AppConfig{Deployed: true, Env: map[string]string{"HDD_PATH": drive}}
m.mu.Unlock()
empty := filepath.Join(drive, KeptDirName, "nextcloud", "2026-09-25_141014")
must(t, os.MkdirAll(empty, 0o755))
full := filepath.Join(drive, KeptDirName, "nextcloud", "2026-09-24_101010")
must(t, os.MkdirAll(full, 0o755))
must(t, os.WriteFile(filepath.Join(full, "f"), []byte("kept"), 0o644))
items := m.ListKept([]string{drive})
if len(items) != 1 || items[0].Path != full {
t.Fatalf("listing = %+v — want only the folder that holds something", items)
}
}
+24
View File
@@ -234,3 +234,27 @@ func loadMetadataFile(path string) (Metadata, error) {
}
return LoadProbeMetadata(tmp), nil // R-670
}
// RestoredVersionPosition says where an app stands after a restore brought it back at an older version
// (v0.275.0, `07` §6.6, the page's first sentence): behind — its pin is not the catalog's; climbable — a
// tested ladder step leads on from its pin, so the automatic leg climbs it (legCandidate refuses an app
// older than the ladder, LegSkipOlderThanLadder, and a template with no ladder, LegSkipNoTestRecord).
// Digests are ignored on both sides: this is about versions, not about which bytes a tag names today.
func (m *Manager) RestoredVersionPosition(name string) (behind, climbable bool) {
st, ok := m.GetStack(name)
if !ok || st.AppConfig == nil || len(st.AppConfig.PinnedImages) == 0 || len(st.CatalogImages) == 0 {
return false, false
}
strip := func(in map[string]string) map[string]string {
out := make(map[string]string, len(in))
for k, v := range in {
out[k] = StripDigest(v)
}
return out
}
if sameRefs(strip(st.AppConfig.PinnedImages), strip(st.CatalogImages)) {
return false, false
}
tpl := filepath.Dir(m.CatalogTemplatePath(name, "docker-compose.yml"))
return true, ladderStepsLeft(tpl, st.AppConfig.PinnedImages) > 0
}
+31
View File
@@ -214,3 +214,34 @@ func mustWriteMk(t *testing.T, p, body string) {
}
mustWrite(t, p, body)
}
// TestA3_RestoredVersionPosition — after a restore brought an app back at an older pin: behind, and
// climbable only when a tested ladder step leads on from that pin (the automatic leg's own rule:
// LegSkipOlderThanLadder). A pin at the catalog's version is not behind; digests do not count.
func TestA3_RestoredVersionPosition(t *testing.T) {
m, _, _, _, _ := ladderManager(t, true)
set := func(pin string) {
m.mu.Lock()
st := m.stacks["nextcloud"]
st.CatalogImages = map[string]string{"web": ladderC}
if st.AppConfig == nil {
st.AppConfig = &AppConfig{}
}
st.AppConfig.PinnedImages = map[string]string{"web": pin}
m.mu.Unlock()
}
for _, c := range []struct {
pin string
behind, climbable bool
}{
{ladderA, true, true}, // on the ladder: the leg climbs it
{"nextcloud:20.0.0-apache", true, false}, // older than the ladder: the leg will not
{ladderC, false, false}, // at the catalog's version
{ladderC + "@sha256:0123", false, false}, // a digest is not a version
} {
set(c.pin)
if b, cl := m.RestoredVersionPosition("nextcloud"); b != c.behind || cl != c.climbable {
t.Errorf("pin %s: behind=%v climbable=%v, want %v %v", c.pin, b, cl, c.behind, c.climbable)
}
}
}
+68 -1
View File
@@ -79,6 +79,60 @@ type ConversionCopy struct {
At string `yaml:"at" json:"at"` // RFC3339 — a backup proven after this releases the copy
From int `yaml:"from" json:"from"`
To int `yaml:"to" json:"to"`
// Service (v0.275.0) is the converted compose service — the release looks for a dump written by ITS
// engine at major To. "" on a record written before v0.275.0: then every Postgres-family image in the
// dump's recorded set must be at To.
Service string `yaml:"service,omitempty" json:"service,omitempty"`
}
// DataDumpStamp is one database dump in the app's own recovery unit, with what the backup side recorded
// when it was WRITTEN (v0.275.0, R-696): its time and the running images (service -> ref@digest).
type DataDumpStamp struct {
File string
At time.Time
Images map[string]string
}
// DumpStampSource is the backup side's record of the app's own unit's database dumps (v0.275.0). An
// OPTIONAL extension of UpdateGuards: guards without it never release a conversion copy (fail closed —
// a copy outliving its backup costs disk, never data).
type DumpStampSource interface {
DumpStamps(name string) []DataDumpStamp
}
// convertedDumpAt is the release's second condition (A4): a database dump written AFTER the conversion
// whose recorded engine is the NEW major. The first condition — a copy of the app proven after the
// conversion on any tier — says the DATA is newer; this one says it was written by the converted engine,
// which a unit's refresh time or a snapshot's time cannot say (R-696: the demo-hp release cited a unit
// re-captured over a PostgreSQL 16 dump).
func convertedDumpAt(stamps []DataDumpStamp, cc *ConversionCopy, after time.Time) (DataDumpStamp, bool) {
for _, st := range stamps {
if !st.At.After(after) || len(st.Images) == 0 {
continue
}
if cc.Service != "" {
if ref, ok := st.Images[cc.Service]; ok {
if mj, ok := postgresMajor(ref); ok && mj == cc.To {
return st, true
}
}
continue
}
seen, all := 0, true
for _, ref := range st.Images {
if !isPostgresImage(ref) {
continue
}
seen++
if mj, ok := postgresMajor(ref); !ok || mj != cc.To {
all = false
}
}
if seen > 0 && all {
return st, true
}
}
return DataDumpStamp{}, false
}
// conversionDumpMargin is A5's margin on the dump's bound (the DB volume's own size).
@@ -576,12 +630,25 @@ func (m *Manager) ReleaseConversionCopies(ctx context.Context) []string {
if !ok {
continue
}
// v0.275.0 (A4): AND a dump the converted engine wrote. Without the stamps (older guards, or a unit
// whose data is unstamped) the copy is KEPT — logged, retried at the next pass.
src, hasStamps := g.(DumpStampSource)
var dump DataDumpStamp
if hasStamps {
dump, ok = convertedDumpAt(src.DumpStamps(st.Name), cc, at)
}
if !hasStamps || !ok {
if m.isDebug() {
m.logger.Printf("[DEBUG] [stacks] %s: the pre-conversion copy %s is KEPT — a copy proven at %s exists, but no database dump written after %s by PostgreSQL %d is recorded yet", st.Name, cc.Copy, rp.ProvenAt.UTC().Format(time.RFC3339), cc.At, cc.To)
}
continue
}
if err := m.copier().Remove(cc.Copy); err != nil {
m.logger.Printf("[WARN] [stacks] %s: could not remove the pre-conversion copy %s: %v — kept, tried again later", st.Name, cc.Copy, err)
continue
}
m.recordConversionCopy(st.Name, filepath.Dir(st.ComposePath), nil)
m.logger.Printf("[INFO] [stacks] %s: REMOVED the pre-conversion datadir copy %s (PostgreSQL %d) — the converted app has a backup proven on %d: %s at %s", st.Name, cc.Copy, cc.From, cc.To, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
m.logger.Printf("[INFO] [stacks] %s: REMOVED the pre-conversion datadir copy %s (PostgreSQL %d) — the converted app has a backup proven on %d: %s at %s, its dump %s written %s by %v", st.Name, cc.Copy, cc.From, cc.To, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), dump.File, dump.At.UTC().Format(time.RFC3339), dump.Images)
released = append(released, st.Name)
}
return released
@@ -423,6 +423,8 @@ func TestConvert_ReleaseAfterAProvenBackup(t *testing.T) {
}
g.mu.Lock()
g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(time.Hour)}}
// v0.275.0 (A4): and the dump in that backup was written by the converted engine.
g.stamps = []DataDumpStamp{{File: "db-dumps/nextcloud-postgres.sql", At: slice4T0.Add(time.Hour), Images: map[string]string{"db": "postgres:18-alpine@sha256:18"}}}
g.mu.Unlock()
if got := m.ReleaseConversionCopies(context.Background()); len(got) != 1 || fc.nCopies() != 0 {
t.Fatalf("released %v after a proven backup; copies=%d", got, fc.nCopies())
@@ -505,3 +507,68 @@ func TestConvert_MarkDoesNotChangeOldLadderPrints(t *testing.T) {
t.Fatalf("an entry without the mark prints it: %s", b)
}
}
// A5 red-proof 2 — the release refuses a PRE-CONVERSION dump (R-696). The measured demo-hp night: a copy
// "proven" after the conversion (the unit's refresh time) while its dump was written by PostgreSQL 16
// before the conversion; v0.274.0 removed the 16 datadir copy on that. Now the copy stays until a dump
// written after the conversion BY THE NEW ENGINE is recorded — and each half of that is needed.
func TestA5_TheReleaseRefusesAPreConversionDump(t *testing.T) {
m, dir, g, _, fc, _ := convManager(t, goodMark)
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("setup: %q", st.UpdatePhase)
}
m.mu.Lock()
m.stacks["nextcloud"].AppConfig = LoadAppConfig(dir)
m.mu.Unlock()
cc := LoadAppConfig(dir).ConversionCopy
if cc == nil || cc.Service != "db" {
t.Fatalf("conversion_copy = %+v, want one naming the converted service", cc)
}
at, _ := time.Parse(time.RFC3339, cc.At)
g.mu.Lock()
g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: at.Add(10 * time.Minute)}} // "proven" after it
g.mu.Unlock()
for _, c := range []struct {
why string
stamp DataDumpStamp
}{
{"a dump written BEFORE the conversion, by 16", DataDumpStamp{File: "db-dumps/nextcloud-postgres.sql", At: at.Add(-5 * time.Minute), Images: map[string]string{"db": "postgres:16-alpine@sha256:16"}}},
{"a dump after the conversion, but recorded as 16", DataDumpStamp{File: "db-dumps/nextcloud-postgres.sql", At: at.Add(5 * time.Minute), Images: map[string]string{"db": "postgres:16-alpine@sha256:16"}}},
{"a dump after the conversion with no recorded images", DataDumpStamp{File: "db-dumps/nextcloud-postgres.sql", At: at.Add(5 * time.Minute)}},
} {
g.mu.Lock()
g.stamps = []DataDumpStamp{c.stamp}
g.mu.Unlock()
if got := m.ReleaseConversionCopies(context.Background()); len(got) != 0 || fc.nCopies() != 1 {
t.Fatalf("%s: released %v; copies=%d — the 16 datadir copy went on a backup that holds no 18 dump", c.why, got, fc.nCopies())
}
}
g.mu.Lock()
g.stamps = []DataDumpStamp{{File: "db-dumps/nextcloud-postgres.sql", At: at.Add(5 * time.Minute), Images: map[string]string{"db": "postgres:18-alpine@sha256:18"}}}
g.mu.Unlock()
if got := m.ReleaseConversionCopies(context.Background()); len(got) != 1 || fc.nCopies() != 0 {
t.Fatalf("released %v with an 18 dump after the conversion; copies=%d", got, fc.nCopies())
}
}
// A record written before v0.275.0 carries no service: every Postgres-family image in the dump's
// recorded set must then be at the new major.
func TestA4_AnOldRecordWithoutAServiceChecksEveryPostgresImage(t *testing.T) {
cc := &ConversionCopy{From: 16, To: 18}
after := slice4T0
st := func(imgs map[string]string) []DataDumpStamp {
return []DataDumpStamp{{File: "db-dumps/x-postgres.sql", At: after.Add(time.Minute), Images: imgs}}
}
if _, ok := convertedDumpAt(st(map[string]string{"app": "x/app:1", "db": "postgres:18-alpine"}), cc, after); !ok {
t.Fatal("an 18 dump was not accepted")
}
if _, ok := convertedDumpAt(st(map[string]string{"app": "x/app:1", "db": "postgres:16-alpine"}), cc, after); ok {
t.Fatal("a 16 dump was accepted")
}
if _, ok := convertedDumpAt(st(map[string]string{"app": "x/app:1"}), cc, after); ok {
t.Fatal("a dump with no Postgres image recorded was accepted")
}
}
@@ -0,0 +1,57 @@
package stacks
import (
"io"
"log"
"os"
"path/filepath"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
)
// R-694 (v0.275.0) — a restore with no guest app.yaml (a load of kept data, a removed app, a rebuilt
// guest) GENERATES the withheld admin login (D5: the unit never carries it), while the app's own login
// comes back with its data. For six of the seven catalog apps with such a field the old password is the
// one that works (measured from each entrypoint, `audits/version-travel-2026-09-26/D4/`); the page used
// to show the new value as "the first password set at install".
//
// Through the PRODUCTION write (PersistUnitRedeployConfig), not the helper alone.
//
// COMPANION RED-PROOF (REPORT.md): make restoredLoginFields return nil — the nextcloud case then records
// nothing and this fails at "restored_logins".
func TestR694_ARestoreThatGeneratesTheLoginRecordsIt(t *testing.T) {
for _, c := range []struct {
name string
app, env string
guestHad bool
wantNoted bool
}{
{"nextcloud, guest app.yaml gone", "nextcloud", "NEXTCLOUD_ADMIN_PASSWORD", false, true},
{"nextcloud, guest kept the household's value", "nextcloud", "NEXTCLOUD_ADMIN_PASSWORD", true, false},
{"code-server applies the env at every start", "code-server", "PASSWORD", false, false},
} {
t.Run(c.name, func(t *testing.T) {
dir := t.TempDir()
cfg := &config.Config{}
cfg.Paths.StacksDir = filepath.Join(dir, "stacks")
cfg.Stacks.ComposeCommand = "docker compose"
app := filepath.Join(cfg.Paths.StacksDir, c.app)
must(t, os.MkdirAll(app, 0o755))
must(t, os.WriteFile(filepath.Join(app, "docker-compose.yml"), []byte("services:\n web:\n image: x/y:1\n"), 0o644))
must(t, os.WriteFile(filepath.Join(app, ".felhom.yml"), []byte("display_name: X\ndeploy_fields:\n - env_var: "+c.env+"\n label: Admin\n type: password\n"), 0o644))
if c.guestHad {
must(t, os.WriteFile(filepath.Join(app, "app.yaml"), []byte("deployed: true\nenv:\n "+c.env+": households-own\n"), 0o600))
}
m, err := NewManager(cfg, log.New(io.Discard, "", 0))
must(t, err)
must(t, m.ScanStacks())
must(t, m.PersistUnitRedeployConfig(c.app, map[string]string{c.env: "generated-at-restore"}))
got := LoadAppConfig(app)
noted := got != nil && len(got.RestoredLogins) == 1 && got.RestoredLogins[0] == c.env
if noted != c.wantNoted {
t.Fatalf("restored_logins = %v, want noted=%v", got.RestoredLogins, c.wantNoted)
}
})
}
}
+1 -1
View File
@@ -1011,7 +1011,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
}
m.removeUndoCopies(name, rest)
if keep != nil {
m.recordConversionCopy(name, dir, &ConversionCopy{Volume: keep.Volume, Copy: keep.Copy, At: m.now().UTC().Format(time.RFC3339), From: entry.Convert.From, To: entry.Convert.To})
m.recordConversionCopy(name, dir, &ConversionCopy{Volume: keep.Volume, Copy: keep.Copy, At: m.now().UTC().Format(time.RFC3339), From: entry.Convert.From, To: entry.Convert.To, Service: entry.Convert.Service})
m.logger.Printf("[INFO] [stacks] update %s: KEEPING the pre-conversion datadir copy %s (PostgreSQL %d) until a backup of the converted app is proven", name, keep.Copy, entry.Convert.From)
}
_ = os.RemoveAll(filepath.Join(dir, preUpdateConvertDir))
@@ -37,6 +37,14 @@ type fakeGuards struct {
pinAtDump string
stackDir string
undoState string // what the last hold said a failed undo left (v0.263.0)
// stamps are the app's own unit's recorded database dumps (v0.275.0, DumpStampSource).
stamps []DataDumpStamp
}
func (f *fakeGuards) DumpStamps(string) []DataDumpStamp {
f.mu.Lock()
defer f.mu.Unlock()
return append([]DataDumpStamp(nil), f.stamps...)
}
func (f *fakeGuards) note(c string) { f.mu.Lock(); f.calls = append(f.calls, c); f.mu.Unlock() }
@@ -0,0 +1,48 @@
package web
import (
"strings"
"testing"
"time"
)
// TestA3_RestoredVersionSentence — the FIRST sentence after a restore of an older version (v0.275.0,
// `07` §6.6, D4 option A), both languages, and its absence when nothing changed. The brief's wording is
// the contract: which backup, which version, and what happens next.
func TestA3_RestoredVersionSentence(t *testing.T) {
s := noteServer(t)
at := time.Date(2026, 9, 26, 7, 29, 45, 0, time.UTC)
pins := []string{"docmost/docmost:0.96.0@sha256:b5", "postgres:16-alpine", "redis:7-alpine"}
s.versionPosition = func(string) (bool, bool) { return true, true }
got := s.restoredVersionPrefix("docmost", true, pins, at)
for _, want := range []string{"A(z) docmost visszaállt a(z) 2026-09-26 ", "-i mentésből", "docmost:0.96.0, postgres:16-alpine, redis:7-alpine verzióra", "lépésenként hozza naprakészre"} {
if !strings.Contains(got, want) {
t.Errorf("hu: missing %q in %q", want, got)
}
}
if !strings.HasSuffix(got, " ") {
t.Errorf("the prefix must end with a space before the outcome sentence: %q", got)
}
if err := s.settings.SetLanguage("en"); err != nil {
t.Fatal(err)
}
got = s.restoredVersionPrefix("docmost", true, pins, at)
if !strings.Contains(got, "docmost is back from the backup of 2026-09-26 ") || !strings.Contains(got, "one step at a time") {
t.Errorf("en: %q", got)
}
// No tested step leads on from the restored version: say the box will not update it by itself.
s.versionPosition = func(string) (bool, bool) { return true, false }
if got := s.restoredVersionPrefix("docmost", true, pins, at); !strings.Contains(got, "does not update it by itself") {
t.Errorf("no ladder step: %q", got)
}
// Nothing changed, or the app is not behind: no sentence.
if got := s.restoredVersionPrefix("docmost", false, pins, at); got != "" {
t.Errorf("unchanged version: %q", got)
}
s.versionPosition = func(string) (bool, bool) { return false, false }
if got := s.restoredVersionPrefix("docmost", true, pins, at); got != "" {
t.Errorf("not behind: %q", got)
}
}
+87 -14
View File
@@ -487,7 +487,17 @@ func (s *Server) deployHandler(w http.ResponseWriter, r *http.Request, name stri
data["AutoFieldValues"] = autoFieldValues
// For deployed apps, pass stored field values (decrypted) so fields show current values
if alreadyDeployed && decryptedEnv != nil {
// R-694 (v0.275.0): a login a restore GENERATED is not the app's login (its login came back with
// the data) — its value is never rendered, and the field says to use the old password.
restored := map[string]bool{}
if appCfg != nil {
for _, n := range appCfg.RestoredLogins {
restored[n] = true
delete(decryptedEnv, n)
}
}
data["DeployedFieldValues"] = decryptedEnv
data["RestoredLogins"] = restored
}
// R-351 SCENARIO A — an app being reinstalled so its data can come back should not ask the
// customer to remember what their own backup already recorded. The address and the data folder
@@ -1723,7 +1733,7 @@ func (s *Server) backupRestoreHandler(w http.ResponseWriter, r *http.Request) {
// The customer reads it as "my data is back". The snapshot id is dropped from the sentence
// deliberately: it identified WHICH backup ran and told the customer nothing about what came out
// of it, which is the question the sentence exists to answer.
s.backupMgr.EndRestoreOp(true, s.unitRestoreOutcomeMsg(stackName, res))
s.backupMgr.EndRestoreOp(true, s.restoredVersionPrefix(stackName, res.VersionChanged, res.DataPins, res.DataAt)+s.unitRestoreOutcomeMsg(stackName, res))
}()
http.Redirect(w, r, "/backups/restore?"+flashQuery("flash", "flash.restore.started"), http.StatusFound)
}
@@ -1780,6 +1790,39 @@ func (s *Server) unitRestoreOutcomeMsg(app string, res backup.UnitRestoreResult)
return s.note(unitRestoreSettingsOnlyKey, app)
}
// restoredVersionPrefix is the FIRST sentence after a restore that brought an app back at an older
// version (v0.275.0, `07` §6.6 "Which version a restore brings back", D4 option A): which backup, which
// version, and what happens next — the box climbs it one tested step at a time, or, when no tested step
// leads on from that version, that the box will not update it by itself (a person's press would jump to
// the catalog's current definition, which is not a tested step). "" when the version did not change or
// the app is not behind. Pinned by TestA3_RestoredVersionSentence.
func (s *Server) restoredVersionPrefix(app string, changed bool, pins []string, at time.Time) string {
if !changed || len(pins) == 0 {
return ""
}
pos := s.versionPosition
if pos == nil {
if s.stackMgr == nil {
return ""
}
pos = s.stackMgr.RestoredVersionPosition
}
behind, climbable := pos(app)
if !behind {
return ""
}
key := restoredOlderVersionKey
if !climbable {
key = restoredOlderVersionNoStepKey
}
return s.note(key, app, at.In(getTimezone()).Format("2006-01-02 15:04"), backup.PinsVersion(pins)) + " "
}
const (
restoredOlderVersionKey = "note.restore.older_version"
restoredOlderVersionNoStepKey = "note.restore.older_version_no_step"
)
// R-353 customer-facing strings. Named constants, not inlined, because each is asserted verbatim by
// r353_unit_outcome_test.go — a silent edit to any of them is how an honest message drifts back into a
// comforting one, which is the exact history of the sentence they replace.
@@ -2082,7 +2125,7 @@ func (s *Server) backupTier2UnitRestoreHandler(w http.ResponseWriter, r *http.Re
s.logger.Printf("[INFO] [web] Tier-2 whole restore completed (async): stack=%s in %s (files restored %d, replaced %d, kept-newer %d, unchanged %d; volumes %d/%d, dbs %d/%d)",
stackName, time.Since(start), res.Files.Restored, res.Files.Replaced, res.Files.KeptNewer, res.Files.Unchanged,
res.Unit.VolumesReplayed, res.Unit.ManifestVolumes, res.Unit.DBsReplayed, res.Unit.ManifestDBs)
s.backupMgr.EndRestoreOp(true, s.unitRestoreOutcomeMsg(stackName, res.Unit)+" "+files)
s.backupMgr.EndRestoreOp(true, s.restoredVersionPrefix(stackName, res.Unit.VersionChanged, res.Unit.DataPins, res.Unit.DataAt)+s.unitRestoreOutcomeMsg(stackName, res.Unit)+" "+files)
}()
http.Redirect(w, r, "/backups/apps?"+flashQuery("flash", "flash.restore.full_started"), http.StatusFound)
return
@@ -2120,7 +2163,7 @@ func (s *Server) backupTier2UnitRestoreHandler(w http.ResponseWriter, r *http.Re
// second thing to keep honest. What IS added is which copy it came from and how old that copy
// is: this action overwrote the customer's live data, and the sentence they are left with has
// to say what it overwrote it with (Scenario E).
msg := s.unitRestoreOutcomeMsg(stackName, res)
msg := s.restoredVersionPrefix(stackName, res.VersionChanged, res.DataPins, res.DataAt) + s.unitRestoreOutcomeMsg(stackName, res)
if src := s.tier2UnitSourceMsg(cov); src != "" {
msg += " " + src
}
@@ -3350,10 +3393,43 @@ func skipFileBrowserPath(path string, isMount func(string) bool) bool {
}
func (s *Server) syncFileBrowserMounts(resetDBOnChange bool) {
// R-695 (v0.275.0): SINGLE-FLIGHT. A request is covered by any sync that STARTS reading the state
// after the request was made: a caller that waited behind a running sync returns as soon as one
// such sync has finished, instead of each queued caller restarting the file browser again. Measured
// 2026-09-25 on 9202: two kept-data Deletes in the same second each ran a sync; one restarted the
// file browser with a bind list read before the second Delete, Docker recreated the deleted folder
// empty, and the next sync listed and bound it again.
s.fbReqMu.Lock()
s.fbReqGen++
mine := s.fbReqGen
s.fbReqMu.Unlock()
// Prevent concurrent syncs — multiple callers can race on the same files (H5 fix).
s.fileBrowserMu.Lock()
defer s.fileBrowserMu.Unlock()
s.fbReqMu.Lock()
if !resetDBOnChange && s.fbDoneGen >= mine {
s.fbReqMu.Unlock()
if s.cfg != nil && s.isDebug() {
s.logger.Printf("[DEBUG] [web] FileBrowser sync request %d already covered by a sync that read the state after it", mine)
}
return
}
covers := s.fbReqGen // every request made before THIS read of the state is covered by this sync
s.fbReqMu.Unlock()
defer func() {
s.fbReqMu.Lock()
if covers > s.fbDoneGen {
s.fbDoneGen = covers
}
s.fbReqMu.Unlock()
}()
if s.syncFileBrowserHook != nil {
s.syncFileBrowserHook()
return
}
stackDir := "/opt/docker/stacks/filebrowser"
composePath := stackDir + "/docker-compose.yml"
@@ -3407,9 +3483,13 @@ func (s *Server) syncFileBrowserMounts(resetDBOnChange bool) {
// `09` §3 decision 36: the read-only „Megőrzött adatok" source — one `:ro` bind per kept item, and
// the source only when there is at least one (a source with no mount is a broken sidebar entry, R-67).
keptLabel := ""
if kb := s.keptFileBrowserBinds(); len(kb) > 0 {
storageMounts = append(storageMounts, kb...)
keptLabel = s.msgLang(s.boxLang(), "kept.fb_source")
var keptGroups []int
if s.stackMgr != nil {
if items := s.stackMgr.ListKept(s.keptDrives()); len(items) > 0 {
storageMounts = append(storageMounts, keptBindLines(items)...)
keptLabel = s.msgLang(s.boxLang(), "kept.fb_source")
keptGroups = keptReadGroups(items, statOwner)
}
}
configPath := stackDir + "/config.yaml"
@@ -3436,7 +3516,7 @@ func (s *Server) syncFileBrowserMounts(resetDBOnChange bool) {
}
// Generate and write compose (includes config.yaml mount)
compose := generateFileBrowserCompose(domain, storageMounts)
compose := infra.RenderFileBrowserCompose(domain, storageMounts, keptGroups...)
if err := os.WriteFile(composePath, []byte(compose), 0644); err != nil {
s.logger.Printf("[ERROR] [web] Failed to write FileBrowser compose: %v", err)
return
@@ -3574,13 +3654,6 @@ func fbNeedsRecreate(oldConfig, newConfig, oldCompose, newCompose []byte) bool {
return !bytes.Equal(oldConfig, newConfig) || !bytes.Equal(oldCompose, newCompose)
}
// generateFileBrowserCompose returns a FileBrowser docker-compose.yml string with the given domain
// and storage volume-mount lines. Delegates to internal/infra (the single source of truth — so the
// pinned image and the base-infra bring-up path can never diverge).
func generateFileBrowserCompose(domain string, storageMounts []string) string {
return infra.RenderFileBrowserCompose(domain, storageMounts)
}
// generateFileBrowserConfig returns a FileBrowser Quantum config.yaml with a separate source per
// registered storage path. Delegates to internal/infra (single source of truth).
func generateFileBrowserConfig(paths []settings.StoragePath, importSource bool) string {
@@ -175,6 +175,13 @@ func i18nCasesA() []i18nCase {
{"deploy_deployed_stopped", "deploy", func() map[string]interface{} {
return i18nDeployData(true, stacks.StateStopped, 2)
}},
// v0.275.0 (R-694): a login the restore generated — no value, the "use your old password" hint.
{"deploy_deployed_restored_login", "deploy", func() map[string]interface{} {
d := i18nDeployData(true, stacks.StateRunning, 1)
d["DeployedFieldValues"] = map[string]string{"SUBDOMAIN": "paste"}
d["RestoredLogins"] = map[string]bool{"ADMIN_PASSWORD": true}
return d
}},
{"settings_system_full", "settings_system", func() map[string]interface{} {
d := i18nLayoutData("settings", "Rendszer beállítások")
d["CustomerID"], d["CustomerDomain"] = "test-customer", "example.hu"
+8
View File
@@ -270,6 +270,14 @@ func (s *Server) languageSwitchHandler(w http.ResponseWriter, r *http.Request) {
}
s.logger.Printf("[INFO] [web] language switch: household language set to %s", lang)
s.reportTriggerNow() // the hub learns the language on the next report, in seconds rather than minutes
// R-691 (v0.275.0): the file browser's „Megőrzött adatok" / "Kept data" source is named in the box's
// language when the sync writes its config — so the sync runs now, not at the next unrelated change.
// A no-op when the box keeps no data (the source is absent) and when nothing changed.
if s.langSwitchSync != nil {
s.langSwitchSync()
} else {
go s.SyncFileBrowserMounts()
}
http.Redirect(w, r, stripLangQuery(redirectBackTo(r, "/launcher")), http.StatusFound)
}
+40 -9
View File
@@ -3,8 +3,11 @@ package web
import (
"net/http"
"net/url"
"os"
"path/filepath"
"sort"
"strings"
"syscall"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
@@ -190,15 +193,6 @@ func (s *Server) keptLoadHandler(w http.ResponseWriter, r *http.Request) {
http.Redirect(w, r, "/backups/restore?"+flashQuery("flash", "flash.restore.started"), http.StatusFound)
}
// keptFileBrowserBinds are the read-only binds of the „Megőrzött adatok" source: one per listed item,
// `:ro`, under /srv/<source dir> — never a live app's folder (ListKept never lists one).
func (s *Server) keptFileBrowserBinds() []string {
if s.stackMgr == nil {
return nil
}
return keptBindLines(s.stackMgr.ListKept(s.keptDrives()))
}
// keptBindLines renders the compose bind lines — each READ-ONLY. Pinned by TestKept_FileBrowserBindsAreReadOnly.
func keptBindLines(items []stacks.KeptItem) []string {
var out []string
@@ -207,3 +201,40 @@ func keptBindLines(items []stacks.KeptItem) []string {
}
return out
}
// keptReadGroups is the R-691 rule — decided by CC unattended 2026-09-26, operator may reverse (`07` §6.5):
// the read-only view reads a kept folder another user owns by joining that folder's OWNING GROUP, never by
// changing the household's files or their permissions (nextcloud checks its data folder's mode after a
// Load). A group is added only when the folder is group-READABLE and the group is neither root's (0 — it
// would reach every root-group file in the view's other mounts) nor the view's own (1000). The kept binds
// are `:ro`, so the added group cannot write kept data. Sorted, unique. Pinned by TestR691_KeptReadGroups.
func keptReadGroups(items []stacks.KeptItem, owner func(path string) (gid int, mode os.FileMode, ok bool)) []int {
seen := map[int]bool{}
var out []int
for _, it := range items {
gid, mode, ok := owner(it.Path)
if !ok || gid == 0 || gid == fileBrowserUID || mode&0o040 == 0 || seen[gid] {
continue
}
seen[gid] = true
out = append(out, gid)
}
sort.Ints(out)
return out
}
// fileBrowserUID is the uid:gid the file-browser image runs as (gtstef/filebrowser: `filebrowser`, 1000).
const fileBrowserUID = 1000
// statOwner is keptReadGroups' production owner reader.
func statOwner(path string) (int, os.FileMode, bool) {
fi, err := os.Stat(path)
if err != nil {
return 0, 0, false
}
st, ok := fi.Sys().(*syscall.Stat_t)
if !ok {
return 0, 0, false
}
return int(st.Gid), fi.Mode().Perm(), true
}
+1 -1
View File
@@ -471,7 +471,7 @@ func (s *Server) offboxReconstituteHandler(w http.ResponseWriter, r *http.Reques
}
s.logger.Printf("[INFO] [web] off-box reconstitute %s completed (async): files=%d dbs=%d snapshot=%s",
app, res.FilesPlaced, res.DBsReplayed, res.SnapshotID)
s.backupMgr.EndRestoreOp(true, reconstituteOutcomeMsg(app, res, s.boxLang()))
s.backupMgr.EndRestoreOp(true, s.restoredVersionPrefix(app, res.VersionChanged, res.DataPins, res.DataAt)+reconstituteOutcomeMsg(app, res, s.boxLang()))
}()
offboxRedirectTo(w, r, restoreWizardPath(app), "flash.offbox.full_restore_started", false)
}
@@ -0,0 +1,70 @@
package web
import (
"net/http"
"net/http/httptest"
"net/url"
"os"
"strings"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/infra"
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
)
// R-691 (v0.275.0) — the read-only „Megőrzött adatok" view could not open nextcloud's kept folder
// (`www-data` 33, mode 0770; the view runs as 1000). The rule: join the folder's OWNING GROUP when it
// is group-readable — never root's, never the view's own — and change nothing on the household's files.
//
// COMPANION RED-PROOF (REPORT.md): return nil from keptReadGroups — the compose then carries no
// group_add and this fails at "group_add missing".
func TestR691_KeptReadGroups(t *testing.T) {
owners := map[string]struct {
gid int
mode os.FileMode
}{
"/d/kept/nextcloud/2026-09-25_141014": {33, 0o770}, // the measured case → 33
"/d/appdata/nextcloud": {33, 0o770}, // the same group once
"/d/kept/romm/x": {1000, 0o755}, // the view's own group: nothing to add
"/d/kept/secret/x": {0, 0o750}, // root's group: NEVER
"/d/kept/private/x": {999, 0o700}, // not group-readable: adding it would not help
"/d/kept/jellyfin/x": {911, 0o750}, // another readable group → 911
}
var items []stacks.KeptItem
for p := range owners {
items = append(items, stacks.KeptItem{Path: p})
}
got := keptReadGroups(items, func(p string) (int, os.FileMode, bool) {
o, ok := owners[p]
return o.gid, o.mode, ok
})
if len(got) != 2 || got[0] != 33 || got[1] != 911 {
t.Fatalf("groups = %v, want [33 911]", got)
}
compose := infra.RenderFileBrowserCompose("example.hu", keptBindLines([]stacks.KeptItem{{Path: "/d/kept/nextcloud/2026-09-25_141014", App: "nextcloud"}}), got...)
if !strings.Contains(compose, "group_add:\n - \"33\"\n - \"911\"") {
t.Fatalf("group_add missing or malformed:\n%s", compose)
}
if !strings.Contains(compose, ":/srv/"+infra.FileBrowserKeptMount+"/") || !strings.Contains(compose, ":ro") {
t.Fatalf("the kept bind must stay read-only:\n%s", compose)
}
// No kept groups → the compose is exactly what it was before v0.275.0.
if strings.Contains(infra.RenderFileBrowserCompose("example.hu", nil), "group_add") {
t.Fatal("a box with no kept data got a group_add")
}
}
// The source's name follows the box's language: switching the language triggers the file-browser sync
// that writes it (before, it waited for the next unrelated change — seen on 9202 2026-09-25).
func TestR691_ALanguageSwitchResyncsTheFileBrowser(t *testing.T) {
s := noteServer(t)
called := 0
s.langSwitchSync = func() { called++ }
req := httptest.NewRequest(http.MethodPost, "/settings/language", strings.NewReader(url.Values{"lang": {"en"}}.Encode()))
req.Header.Set("Content-Type", "application/x-www-form-urlencoded")
rec := httptest.NewRecorder()
s.languageSwitchHandler(rec, req)
if rec.Code != http.StatusFound || called != 1 {
t.Fatalf("status %d, file-browser syncs %d — want a redirect and one sync", rec.Code, called)
}
}
@@ -0,0 +1,34 @@
package web
import (
"strings"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
)
// R-694 (v0.275.0) — a deployed app whose admin login was GENERATED by a restore: the page shows no value
// for it and says the old password is the one that works; an app without that record renders as before.
// Rendered through the production template path (renderI18nCase), both languages.
func TestR694_TheDeployPageDoesNotOfferAGeneratedLogin(t *testing.T) {
s := i18nTestServer(t)
base := func() map[string]interface{} { return i18nDeployData(true, stacks.StateRunning, 1) }
restored := func() map[string]interface{} {
d := base()
d["DeployedFieldValues"] = map[string]string{"SUBDOMAIN": "paste"} // the handler removed the value
d["RestoredLogins"] = map[string]bool{"ADMIN_PASSWORD": true}
return d
}
hu := renderI18nCase(t, s, "hu", i18nCase{"r694", "deploy", restored})
if !strings.Contains(hu, "Mentésből töltötted vissza") || strings.Contains(hu, "rejtett érték") || strings.Contains(hu, "Telepítéskor beállított kezdeti jelszó") {
t.Fatalf("hu page does not say to use the old password, or still shows a value")
}
en := renderI18nCase(t, s, "en", i18nCase{"r694", "deploy", restored})
if !strings.Contains(en, "Restored from a backup: log in with the password that was valid when the backup was taken") {
t.Fatal("en page lacks the sentence")
}
plain := renderI18nCase(t, s, "hu", i18nCase{"r694-plain", "deploy", base})
if strings.Contains(plain, "Mentésből töltötted vissza") || !strings.Contains(plain, "Telepítéskor beállított kezdeti jelszó") {
t.Fatal("an app without restored_logins changed")
}
}
@@ -0,0 +1,53 @@
package web
import (
"io"
"log"
"sync"
"sync/atomic"
"testing"
"time"
)
// R-695 (v0.275.0) — the file-browser sync is single-flight: callers that queued behind a running sync
// are covered by ONE sync that read the state after they asked, not by one restart each; a caller whose
// request came after a sync began reading is never dropped.
//
// COMPANION RED-PROOF (REPORT.md): remove the `fbDoneGen >= mine` early return — five queued callers
// then run five syncs.
func TestR695_QueuedSyncsCoalesceIntoOne(t *testing.T) {
s := &Server{logger: log.New(io.Discard, "", 0)}
var runs int32
release := make(chan struct{})
first := true
var fmu sync.Mutex
s.syncFileBrowserHook = func() {
atomic.AddInt32(&runs, 1)
fmu.Lock()
f := first
first = false
fmu.Unlock()
if f {
<-release // the first sync is slow: everything below queues behind it
}
}
var wg sync.WaitGroup
wg.Add(1)
go func() { defer wg.Done(); s.SyncFileBrowserMounts() }()
time.Sleep(50 * time.Millisecond) // the first sync holds the lock
for i := 0; i < 5; i++ {
wg.Add(1)
go func() { defer wg.Done(); s.SyncFileBrowserMounts() }()
}
time.Sleep(50 * time.Millisecond) // all five are queued
close(release)
wg.Wait()
if got := atomic.LoadInt32(&runs); got != 2 {
t.Fatalf("%d syncs ran for 6 requests (1 running + 5 queued) — want 2: the running one and ONE that re-reads for the queue", got)
}
// A request after everything finished is never dropped.
s.SyncFileBrowserMounts()
if got := atomic.LoadInt32(&runs); got != 3 {
t.Fatalf("a new request after the queue drained ran %d syncs in total, want 3", got)
}
}
+13
View File
@@ -51,6 +51,10 @@ type Server struct {
tmplByLang map[string]*template.Template
i18n *i18n.Bundle
// versionPosition (v0.275.0) — where a just-restored app stands against the catalog; nil → the stack
// manager's RestoredVersionPosition. A seam so the restore sentence is testable without a catalog.
versionPosition func(app string) (behind, climbable bool)
sessions map[string]*session
sessionsMu sync.RWMutex
loginAttempts map[string]*loginAttempt
@@ -73,6 +77,15 @@ type Server struct {
// Guard for FileBrowser sync — prevents concurrent file writes (H5 fix)
fileBrowserMu sync.Mutex
// fbReqMu guards the file-browser sync's single-flight counters (R-695, v0.275.0): fbReqGen counts
// requests, fbDoneGen is the newest request a finished sync is known to have covered.
fbReqMu sync.Mutex
fbReqGen uint64
fbDoneGen uint64
// syncFileBrowserHook replaces the sync's body in tests (the single-flight is the unit under test).
syncFileBrowserHook func()
// langSwitchSync replaces the language switch's file-browser sync in tests (R-691).
langSwitchSync func()
// Shared agent local-API client (built once, reused). cfg.LocalAPI is static per process (a
// config-apply triggers a graceful self-restart), so the client is memoized via agentCliOnce —
@@ -577,7 +577,7 @@
{{end}}
</div>
{{if $.AlreadyDeployed}}
<span class="form-hint">{{T "deploy.telepiteskor_beallitott_kezdeti_jelszo_h"}}</span>
<span class="form-hint">{{if and $.RestoredLogins (index $.RestoredLogins .EnvVar)}}{{T "deploy.login_from_backup"}}{{else}}{{T "deploy.telepiteskor_beallitott_kezdeti_jelszo_h"}}{{end}}</span>
{{else}}
<div class="input-with-button" style="margin-top:.25rem">
<input type="password" id="field-confirm-{{.EnvVar}}"
File diff suppressed because it is too large Load Diff