From 820e8efde15d35244c944f0662ec18185efaffa3 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Sun, 27 Sep 2026 17:59:04 +0200 Subject: [PATCH] v0.276.0: a restore and a drive move keep the app's records (R-697, R-700) A drive move persisted through the restore's fresh app.yaml write and dropped the pin: the syncer then copied the catalog verbatim and the next start jumped the app past its ladder (R-700). persistDriveFlip now changes HDD_PATH and nothing else. The restore's write carries the life records (conversion copies, desired_state, update history) from the app.yaml it replaces, and a second conversion no longer overwrites the first kept copy's record (R-697). Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- CHANGELOG.md | 20 +++ controller/README.md | 4 +- controller/internal/stacks/deploy.go | 11 ++ controller/internal/stacks/life_records.go | 41 +++++ controller/internal/stacks/migrate.go | 47 ++++-- controller/internal/stacks/pgconvert.go | 109 ++++++++++---- .../stacks/r700_records_carried_test.go | 141 ++++++++++++++++++ 7 files changed, 331 insertions(+), 42 deletions(-) create mode 100644 controller/internal/stacks/life_records.go create mode 100644 controller/internal/stacks/r700_records_carried_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index 5346f13..c074a12 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,23 @@ +## v0.276.0 — a restore and a drive move keep the app's records (R-697, R-700) (2026-09-27) + +**MinAgent: 0.131.0** (unchanged). Needs hub v0.123.0 (unchanged). New strings: none. Evidence: +`felhom.eu/documentation/audits/records-carried-2026-09-27/`. + +- **R-700 (new, found reading the code for R-697): a drive move unpinned the app.** `doFlipRedeploy` persisted + through the restore's fresh `app.yaml` write, which drops `pinned_images`, `desired_state`, `installed_images`, the + update records and the kept conversion copies. Unpinned, the catalog syncer copies the catalog's compose verbatim + and the next start takes the newest version — past the ladder, and for a PostgreSQL app past its conversion step. + Now `persistDriveFlip` changes `HDD_PATH` and nothing else (load-then-save); the up-and-report tail is + `upFromAppConfig`, shared with `RedeployFromEnv`. +- **R-697: a restore dropped the kept pre-conversion copy's record, so the copy was never released.** The restore's + write (`PersistUnitRedeployConfig`) now carries the app's life records from the `app.yaml` it replaces + (`carryLifeRecords`): `conversion_copy`, `earlier_conversion_copies`, `desired_state` (dropped, a dead app after a + restore read as "unknown intent" and was never alarmed), `failed_update_step`, `last_update_undone`, + `last_auto_update`. Not carried: the pin (the restore pins to the unit), `installed_images` (an observation of what + ran before). And a second conversion after a restore to the old major no longer overwrites the first copy's record: + it moves to `earlier_conversion_copies`, released by the same rule. +- Tests: `internal/stacks/r700_records_carried_test.go` (4). Red-proofs RP1–RP4, each seen failing. + ## v0.275.0 — a backup's data and its version travel together (R-696, `07` §6.6, D4 option A); R-695, R-691, R-694, R-699 (2026-09-26) **MinAgent: 0.131.0** (unchanged). Needs hub v0.123.0 (unchanged). New strings: yes (hu + en, 5 keys). Evidence: diff --git a/controller/README.md b/controller/README.md index 2360cee..797a54a 100644 --- a/controller/README.md +++ b/controller/README.md @@ -665,9 +665,11 @@ entrypoint's empty databases dropped, `CREATE ROLE` skipped for roles that exist verzióját váltaná, de nincs róla próba. Nem változott semmi." — a PostgreSQL major move WITHOUT the mark is refused by the preflight and by the job. After success the old datadir's copy is kept (`app.yaml` `conversion_copy`) until a backup is proven after the conversion; the hourly `conversion-copy-release` job then -removes it, logged by name. PostgreSQL 18 mounts its volume at `/var/lib/postgresql` — the step's definition +removes it, logged by name. A restore keeps that record (v0.276.0, R-697 — the copy is released by the same rule), and a later conversion never overwrites an older kept copy's record (`earlier_conversion_copies`, released the same way). PostgreSQL 18 mounts its volume at `/var/lib/postgresql` — the step's definition carries that. Code: `internal/stacks/pgconvert.go`. Proof: `felhom.eu/documentation/audits/night-2026-09-26/`. +**A restore and a drive move keep the app's records (v0.276.0, R-697, R-700).** A restore writes a fresh `app.yaml` from the unit but keeps `desired_state`, the kept conversion copies and the update history (`failed_update_step`, `last_update_undone`, `last_auto_update`); the pin comes from the unit. A drive move (`doFlipRedeploy`) changes `HDD_PATH` and nothing else (`persistDriveFlip`) — before v0.276.0 it dropped the pin, and the syncer then gave the app the catalog's newest version at its next start. + **Held apps say so (v0.265.0, R-625).** While a hold stands, the update badge reads „Megállítva — visszaállítás szükséges" / "Stopped — restore needed" (`tag-error`, title = the hold sentence's first sentence, in the reader's language) and no Update button is rendered; the API still answers 409 `held`. A diff --git a/controller/internal/stacks/deploy.go b/controller/internal/stacks/deploy.go index 36005db..82e055d 100644 --- a/controller/internal/stacks/deploy.go +++ b/controller/internal/stacks/deploy.go @@ -169,6 +169,11 @@ type AppConfig struct { // ConversionCopy (v0.273.0, `09` §6.4 part 10) is the OLD datadir's copy kept after a successful // PostgreSQL major conversion, until a backup of the converted app is proven (ReleaseConversionCopies). ConversionCopy *ConversionCopy `yaml:"conversion_copy,omitempty" json:"conversion_copy,omitempty"` + // EarlierConversionCopies (v0.276.0, R-697) are older kept copies a later conversion superseded — a + // restore to the old major, then the ladder converting again. Released by the same rule as + // ConversionCopy; without this list the newer record overwrote the older one and its volume was orphaned + // (TestR697_ASecondConversionDoesNotOrphanTheFirstCopy). + EarlierConversionCopies []ConversionCopy `yaml:"earlier_conversion_copies,omitempty" json:"earlier_conversion_copies,omitempty"` // RestoredLogins (v0.275.0, R-694) are the `type: password` fields whose stored value was GENERATED by a // restore (the unit never carries an admin login, D5, and the guest had none — a load of kept data, a // removed app, a rebuilt guest) while the app's own login came back with its data. The page then shows @@ -659,6 +664,11 @@ func (m *Manager) RedeployFromEnv(name string, env map[string]string) error { if err := m.PersistUnitRedeployConfig(name, env); err != nil { return err } + return m.upFromAppConfig(name) +} + +// upFromAppConfig is RedeployFromEnv's up-and-report tail: `compose up -d` from the stored app.yaml. +func (m *Manager) upFromAppConfig(name string) error { stack, ok := m.GetStack(name) if !ok { return fmt.Errorf("stack %q not found", name) @@ -703,6 +713,7 @@ func (m *Manager) PersistUnitRedeployConfig(name string, env map[string]string) } } cfg.RestoredLogins = restoredLoginFields(name, meta, prior, env) + carryLifeRecords(m.logger, name, LoadAppConfig(stackDir), cfg) if len(cfg.RestoredLogins) > 0 { m.logger.Printf("[INFO] [stacks] %s: the restore generated %v — the app's own login came back with its data; the page will not show the new value as the password", name, cfg.RestoredLogins) } diff --git a/controller/internal/stacks/life_records.go b/controller/internal/stacks/life_records.go new file mode 100644 index 0000000..575315e --- /dev/null +++ b/controller/internal/stacks/life_records.go @@ -0,0 +1,41 @@ +package stacks + +import "log" + +// carryLifeRecords copies, from the app.yaml a restore is about to replace (prior, as on disk), the records +// that describe the app's LIFE on this box rather than its definition (R-697, v0.276.0). The restore's +// write is a fresh AppConfig by design — the env, the locked fields and the pin come from the unit — and it +// used to drop these with it: +// +// - conversion_copy + earlier_conversion_copies: a kept pre-conversion datadir copy whose record is dropped +// is never released (R-697, measured on 9202). A restore does not remove the volume, so it must not +// forget it either. +// - desired_state: the household's intent. Dropped, a dead app after a restore was read as "unknown +// intent" and never alarmed (R-166's absent case). +// - failed_update_step, last_update_undone, last_auto_update: the ladder's history on THIS box; the +// automatic leg must not re-press a step that already failed here because a restore happened. +// +// NOT carried: pinned_images (the restore pins to the unit's definition right after — carrying the old pin +// would freeze a failed SetPin onto the wrong version), installed_images (an observation of what ran +// before the restore, possibly another version), restored_logins (computed per restore). +// Pinned by TestR697_ARestoreKeepsTheConversionCopyRecordSoTheCopyIsReleased. +func carryLifeRecords(logger *log.Logger, name string, prior, cfg *AppConfig) { + if prior == nil { + return + } + cfg.ConversionCopy = prior.ConversionCopy + cfg.EarlierConversionCopies = prior.EarlierConversionCopies + if cfg.DesiredState == "" { + cfg.DesiredState = prior.DesiredState + } + cfg.FailedStep = prior.FailedStep + cfg.LastUpdateUndone = prior.LastUpdateUndone + cfg.LastAutoUpdate = prior.LastAutoUpdate + if n := len(prior.EarlierConversionCopies); prior.ConversionCopy != nil || n > 0 { + cur := "" + if prior.ConversionCopy != nil { + cur = prior.ConversionCopy.Copy + } + logger.Printf("[INFO] [stacks] %s: the restore keeps the record of the kept pre-conversion copy %q (+%d earlier) — it is released when a backup written by the converted engine is proven", name, cur, n) + } +} diff --git a/controller/internal/stacks/migrate.go b/controller/internal/stacks/migrate.go index 2d584bd..1573520 100644 --- a/controller/internal/stacks/migrate.go +++ b/controller/internal/stacks/migrate.go @@ -742,11 +742,13 @@ func (m *Manager) doFlipRedeploy(name, target string) error { if m.testSeams != nil && m.testSeams.flipRedeploy != nil { return m.testSeams.flipRedeploy(name, target) } - cfg := m.LoadAppConfigByName(name) - if cfg == nil { - return fmt.Errorf("app config not found") + // R-700 (v0.276.0): the move changes WHERE the data lives and nothing else — persistDriveFlip keeps + // the pin and every record. This used to be RedeployFromEnv, whose fresh app.yaml dropped the pin: the + // syncer then copied the catalog verbatim and the next start jumped the app past its ladder. + if err := m.persistDriveFlip(name, target); err != nil { + return err } - if err := m.RedeployFromEnv(name, flipEnv(cfg.Env, target)); err != nil { + if err := m.upFromAppConfig(name); err != nil { return err } if !m.waitHealthy(name) { @@ -755,14 +757,37 @@ func (m *Manager) doFlipRedeploy(name, target string) error { return nil } -// flipEnv returns a copy of env with HDD_PATH set to target. -func flipEnv(env map[string]string, target string) map[string]string { - out := make(map[string]string, len(env)+1) - for k, v := range env { - out[k] = v +// persistDriveFlip points app.yaml's HDD_PATH at target and changes NOTHING else (R-700, v0.276.0): +// load-then-save, the SaveAppConfig rule — the pin, the intent, the installed images, the update records +// and the kept conversion copies all stay. Starts nothing. +func (m *Manager) persistDriveFlip(name, target string) error { + stack, ok := m.GetStack(name) + if !ok { + return fmt.Errorf("stack %q not found", name) } - out["HDD_PATH"] = target - return out + dir := filepath.Dir(stack.ComposePath) + cfg := LoadAppConfig(dir) + if cfg == nil { + return fmt.Errorf("app config not found") + } + if cfg.Env == nil { + cfg.Env = map[string]string{} + } + from := cfg.Env["HDD_PATH"] + cfg.Env["HDD_PATH"] = target + cfg.Deployed = true + meta := LoadMetadata(dir) + if err := SaveAppConfig(dir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil { + return fmt.Errorf("saving app config: %w", err) + } + m.mu.Lock() + if s, ok := m.stacks[name]; ok { + s.Deployed = true + s.AppConfig = cfg + } + m.mu.Unlock() + m.logger.Printf("[INFO] [stacks] %s: data moved %s -> %s — app.yaml keeps its pin (%d service(s)) and records", name, from, target, len(cfg.PinnedImages)) + return nil } // waitHealthy polls until the stack is up (running/unhealthy) or times out. diff --git a/controller/internal/stacks/pgconvert.go b/controller/internal/stacks/pgconvert.go index 99e272b..55fa6f5 100644 --- a/controller/internal/stacks/pgconvert.go +++ b/controller/internal/stacks/pgconvert.go @@ -590,11 +590,19 @@ func diffSnapshots(before, after []string) string { // ── keeping, then releasing, the old datadir's copy (B5) ──────────────────────────────────────── func (m *Manager) recordConversionCopy(name, dir string, cc *ConversionCopy) { + set := func(cfg *AppConfig) { + // R-697 (v0.276.0): a newer conversion never overwrites an older kept copy's record — the older one + // moves to the earlier list and is released by the same rule. + if cc != nil && cfg.ConversionCopy != nil && cfg.ConversionCopy.Copy != cc.Copy && !hasCopy(cfg.EarlierConversionCopies, cfg.ConversionCopy.Copy) { + cfg.EarlierConversionCopies = append(cfg.EarlierConversionCopies, *cfg.ConversionCopy) + } + cfg.ConversionCopy = cc + } cfg := LoadAppConfig(dir) if cfg == nil || (cc == nil && cfg.ConversionCopy == nil) { return } - cfg.ConversionCopy = cc + set(cfg) meta := LoadMetadata(dir) if err := SaveAppConfig(dir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil { m.logger.Printf("[ERROR] [stacks] update %s: recording conversion_copy failed: %v", name, err) @@ -602,11 +610,38 @@ func (m *Manager) recordConversionCopy(name, dir string, cc *ConversionCopy) { } m.mu.Lock() if st, ok := m.stacks[name]; ok && st.AppConfig != nil { - st.AppConfig.ConversionCopy = cc + st.AppConfig.ConversionCopy = cfg.ConversionCopy + st.AppConfig.EarlierConversionCopies = cfg.EarlierConversionCopies } m.mu.Unlock() } +func hasCopy(list []ConversionCopy, copyVol string) bool { + for _, c := range list { + if c.Copy == copyVol { + return true + } + } + return false +} + +// forgetEarlierConversionCopy drops one released copy from the earlier list. +func (m *Manager) forgetEarlierConversionCopy(name, dir, copyVol string) { + m.mutateAppConfig(name, dir, "earlier_conversion_copies", func(cfg *AppConfig) bool { + var keep []ConversionCopy + for _, c := range cfg.EarlierConversionCopies { + if c.Copy != copyVol { + keep = append(keep, c) + } + } + if len(keep) == len(cfg.EarlierConversionCopies) { + return false + } + cfg.EarlierConversionCopies = keep + return true + }) +} + // ReleaseConversionCopies removes each kept pre-conversion datadir copy whose app now has a backup // PROVEN after the conversion (any tier — the backup side returns only proven copies). Returns the // names released. Run periodically from main.go (TestConvert_ReleaseIsWiredAtStartup). @@ -617,43 +652,57 @@ func (m *Manager) ReleaseConversionCopies(ctx context.Context) []string { } var released []string for _, st := range m.GetStacks() { - if st.AppConfig == nil || st.AppConfig.ConversionCopy == nil { + if st.AppConfig == nil { continue } - cc := st.AppConfig.ConversionCopy - at, err := time.Parse(time.RFC3339, cc.At) - if err != nil { - m.logger.Printf("[WARN] [stacks] %s: conversion_copy has an unreadable time %q — the copy %s is kept", st.Name, cc.At, cc.Copy) - continue - } - rp, ok, _ := g.RestorePoints(ctx, st.Name, func(p UpdateRestorePoint) bool { return p.ProvenAt.After(at) }) - if !ok { - continue - } - // v0.275.0 (A4): AND a dump the converted engine wrote. Without the stamps (older guards, or a unit - // whose data is unstamped) the copy is KEPT — logged, retried at the next pass. - src, hasStamps := g.(DumpStampSource) - var dump DataDumpStamp - if hasStamps { - dump, ok = convertedDumpAt(src.DumpStamps(st.Name), cc, at) - } - if !hasStamps || !ok { - if m.isDebug() { - m.logger.Printf("[DEBUG] [stacks] %s: the pre-conversion copy %s is KEPT — a copy proven at %s exists, but no database dump written after %s by PostgreSQL %d is recorded yet", st.Name, cc.Copy, rp.ProvenAt.UTC().Format(time.RFC3339), cc.At, cc.To) + // v0.276.0 (R-697): the current record and every earlier one a later conversion superseded. + for _, e := range st.AppConfig.EarlierConversionCopies { + e := e + if m.releaseOneConversionCopy(ctx, g, st, &e) { + m.forgetEarlierConversionCopy(st.Name, filepath.Dir(st.ComposePath), e.Copy) + released = append(released, st.Name) } - continue } - if err := m.copier().Remove(cc.Copy); err != nil { - m.logger.Printf("[WARN] [stacks] %s: could not remove the pre-conversion copy %s: %v — kept, tried again later", st.Name, cc.Copy, err) - continue + if cc := st.AppConfig.ConversionCopy; cc != nil && m.releaseOneConversionCopy(ctx, g, st, cc) { + m.recordConversionCopy(st.Name, filepath.Dir(st.ComposePath), nil) + released = append(released, st.Name) } - m.recordConversionCopy(st.Name, filepath.Dir(st.ComposePath), nil) - m.logger.Printf("[INFO] [stacks] %s: REMOVED the pre-conversion datadir copy %s (PostgreSQL %d) — the converted app has a backup proven on %d: %s at %s, its dump %s written %s by %v", st.Name, cc.Copy, cc.From, cc.To, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), dump.File, dump.At.UTC().Format(time.RFC3339), dump.Images) - released = append(released, st.Name) } return released } +// releaseOneConversionCopy removes one kept copy when its release conditions hold; true = removed. +func (m *Manager) releaseOneConversionCopy(ctx context.Context, g UpdateGuards, st Stack, cc *ConversionCopy) bool { + at, err := time.Parse(time.RFC3339, cc.At) + if err != nil { + m.logger.Printf("[WARN] [stacks] %s: conversion_copy has an unreadable time %q — the copy %s is kept", st.Name, cc.At, cc.Copy) + return false + } + rp, ok, _ := g.RestorePoints(ctx, st.Name, func(p UpdateRestorePoint) bool { return p.ProvenAt.After(at) }) + if !ok { + return false + } + // v0.275.0 (A4): AND a dump the converted engine wrote. Without the stamps (older guards, or a unit + // whose data is unstamped) the copy is KEPT — logged, retried at the next pass. + src, hasStamps := g.(DumpStampSource) + var dump DataDumpStamp + if hasStamps { + dump, ok = convertedDumpAt(src.DumpStamps(st.Name), cc, at) + } + if !hasStamps || !ok { + if m.isDebug() { + m.logger.Printf("[DEBUG] [stacks] %s: the pre-conversion copy %s is KEPT — a copy proven at %s exists, but no database dump written after %s by PostgreSQL %d is recorded yet", st.Name, cc.Copy, rp.ProvenAt.UTC().Format(time.RFC3339), cc.At, cc.To) + } + return false + } + if err := m.copier().Remove(cc.Copy); err != nil { + m.logger.Printf("[WARN] [stacks] %s: could not remove the pre-conversion copy %s: %v — kept, tried again later", st.Name, cc.Copy, err) + return false + } + m.logger.Printf("[INFO] [stacks] %s: REMOVED the pre-conversion datadir copy %s (PostgreSQL %d) — the converted app has a backup proven on %d: %s at %s, its dump %s written %s by %v", st.Name, cc.Copy, cc.From, cc.To, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), dump.File, dump.At.UTC().Format(time.RFC3339), dump.Images) + return true +} + // ── production boundary ───────────────────────────────────────────────────────────────────────── type dockerPGConverter struct{ m *Manager } diff --git a/controller/internal/stacks/r700_records_carried_test.go b/controller/internal/stacks/r700_records_carried_test.go new file mode 100644 index 0000000..a1631a5 --- /dev/null +++ b/controller/internal/stacks/r700_records_carried_test.go @@ -0,0 +1,141 @@ +package stacks + +import ( + "context" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// R-697 + R-700 (v0.276.0) — the two app.yaml rewrites that are NOT a new install keep the records that +// describe the app's LIFE, not its definition. +// +// Before: PersistUnitRedeployConfig built a fresh AppConfig (Deployed, DeployedAt, Env, LockedFields), and +// a drive move (doFlipRedeploy) went through it too. So a restore dropped `conversion_copy` (the kept +// PostgreSQL datadir copy was never released — R-697, measured on 9202) and `desired_state`; a drive move +// ALSO dropped `pinned_images` — the syncer then copies the catalog verbatim and the next start jumps the +// app past its ladder (R-700). + +// R-697: the whole consequence — convert, restore, back up on the new engine: the old datadir copy GOES. +// COMPANION RED-PROOF (REPORT.md): at v0.275.0 the restore drops the record, the release never sees it, +// and this fails at "copies=1". +func TestR697_ARestoreKeepsTheConversionCopyRecordSoTheCopyIsReleased(t *testing.T) { + m, dir, g, _, fc, _ := convManager(t, goodMark) + if err := m.StartGuardedUpdate("nextcloud"); err != nil { + t.Fatal(err) + } + if st := waitUpdateDone(t, m, "nextcloud"); st.UpdatePhase != UpdatePhaseDone { + t.Fatalf("setup: %q", st.UpdatePhase) + } + cc := LoadAppConfig(dir).ConversionCopy + if cc == nil || fc.nCopies() != 1 { + t.Fatalf("setup: conversion_copy %+v, copies %d", cc, fc.nCopies()) + } + must(t, os.WriteFile(filepath.Join(dir, "app.yaml"), []byte(strings.Replace(mustRead(t, filepath.Join(dir, "app.yaml")), "deployed: true", "deployed: true\ndesired_state: running", 1)), 0o600)) + + // The restore's app.yaml write (the production one both restore paths use). + must(t, m.PersistUnitRedeployConfig("nextcloud", map[string]string{"HDD_PATH": "/mnt/drv"})) + + got := LoadAppConfig(dir) + if got.ConversionCopy == nil || got.ConversionCopy.Copy != cc.Copy { + t.Fatalf("after the restore conversion_copy = %+v, want %s kept", got.ConversionCopy, cc.Copy) + } + if got.DesiredState != DesiredStateRunning { + t.Fatalf("after the restore desired_state = %q, want the household's %q kept", got.DesiredState, DesiredStateRunning) + } + at, _ := time.Parse(time.RFC3339, cc.At) + g.mu.Lock() + g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: at.Add(10 * time.Minute)}} + g.stamps = []DataDumpStamp{{File: "db-dumps/nextcloud-postgres.sql", At: at.Add(5 * time.Minute), Images: map[string]string{"db": "postgres:18-alpine@sha256:18"}}} + g.mu.Unlock() + if rel := m.ReleaseConversionCopies(context.Background()); len(rel) != 1 || fc.nCopies() != 0 { + t.Fatalf("released %v; copies=%d — the pre-conversion copy is orphaned by the restore", rel, fc.nCopies()) + } +} + +// R-697's second half: after a restore to the OLD major the ladder converts again. The newer copy's record +// must not overwrite the older one — both are released by the same rule. +func TestR697_ASecondConversionDoesNotOrphanTheFirstCopy(t *testing.T) { + m, dir, g, _, fc, _ := convManager(t, goodMark) + t0 := slice4T0 + first := &ConversionCopy{Volume: convVol, Copy: convVol + ".pre-update-A", At: t0.Format(time.RFC3339), From: 16, To: 18, Service: "db"} + second := &ConversionCopy{Volume: convVol, Copy: convVol + ".pre-update-B", At: t0.Add(time.Hour).Format(time.RFC3339), From: 16, To: 18, Service: "db"} + fc.mu.Lock() + fc.copies[first.Copy], fc.copies[second.Copy] = "16-A", "16-B" + fc.mu.Unlock() + m.recordConversionCopy("nextcloud", dir, first) + m.recordConversionCopy("nextcloud", dir, second) + m.recordConversionCopy("nextcloud", dir, second) // idempotent: the same copy is never listed twice + + got := LoadAppConfig(dir) + if got.ConversionCopy == nil || got.ConversionCopy.Copy != second.Copy || len(got.EarlierConversionCopies) != 1 || got.EarlierConversionCopies[0].Copy != first.Copy { + t.Fatalf("records: current %+v, earlier %+v — want B current and A kept as earlier", got.ConversionCopy, got.EarlierConversionCopies) + } + g.mu.Lock() + g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: t0.Add(3 * time.Hour)}} + g.stamps = []DataDumpStamp{{File: "db-dumps/nextcloud-postgres.sql", At: t0.Add(2 * time.Hour), Images: map[string]string{"db": "postgres:18-alpine@sha256:18"}}} + g.mu.Unlock() + m.mu.Lock() + m.stacks["nextcloud"].AppConfig = LoadAppConfig(dir) + m.mu.Unlock() + m.ReleaseConversionCopies(context.Background()) + if fc.nCopies() != 0 { + t.Fatalf("copies left %d — a copy was orphaned", fc.nCopies()) + } + if got := LoadAppConfig(dir); got.ConversionCopy != nil || len(got.EarlierConversionCopies) != 0 { + t.Fatalf("records left: %+v, %+v", got.ConversionCopy, got.EarlierConversionCopies) + } +} + +// R-700: a drive move changes ONE thing — where the data lives. Every other record stays, above all the +// pin: an unpinned app takes the catalog's version on its next start (sync.renderSource's table). +// COMPANION RED-PROOF (REPORT.md): at v0.275.0 doFlipRedeploy persisted through PersistUnitRedeployConfig +// and this fails at "pinned_images". +func TestR700_ADriveMoveKeepsThePinAndTheRecords(t *testing.T) { + m, dir := newPinManager(t, "services:\n web:\n image: x/web:1\n", "services:\n web:\n image: x/web:3\n", + "deployed: true\ndeployed_at: \"2026-01-02T03:04:05Z\"\ndesired_state: stopped\nenv:\n HDD_PATH: /mnt/a\n SUBDOMAIN: web\n"+ + "pinned_images:\n web: x/web:1\ninstalled_images:\n web:\n ref: x/web:1\n at: \"2026-01-02T03:04:05Z\"\n"+ + "conversion_copy:\n volume: v\n copy: v.pre\n at: \"2026-01-02T03:04:05Z\"\n from: 16\n to: 18\n"+ + "failed_update_step:\n to:\n web: x/web:2\n ladder: L\n at: \"2026-01-02T03:04:05Z\"\n outcome: undone\n") + must(t, m.persistDriveFlip("nextcloud", "/mnt/b")) + + got := LoadAppConfig(dir) + if got.Env["HDD_PATH"] != "/mnt/b" || got.Env["SUBDOMAIN"] != "web" { + t.Fatalf("env = %v, want HDD_PATH moved and the rest kept", got.Env) + } + if got.PinnedImages["web"] != "x/web:1" { + t.Fatalf("pinned_images = %v — the moved app is unpinned and will jump to the catalog's version", got.PinnedImages) + } + if got.DesiredState != DesiredStateStopped || got.ConversionCopy == nil || got.FailedStep == nil || got.InstalledImages["web"].Ref != "x/web:1" || got.DeployedAt != "2026-01-02T03:04:05Z" { + t.Fatalf("records dropped by the move: desired=%q conversion=%+v failed=%+v installed=%v deployed_at=%q", + got.DesiredState, got.ConversionCopy, got.FailedStep, got.InstalledImages, got.DeployedAt) + } + if st, _ := m.GetStack("nextcloud"); st.AppConfig == nil || st.AppConfig.Env["HDD_PATH"] != "/mnt/b" || len(st.AppConfig.PinnedImages) == 0 { + t.Fatalf("in-memory record not updated: %+v", st.AppConfig) + } +} + +// The wiring: the drive move persists through persistDriveFlip, never the restore's fresh write. +func TestR700_TheDriveMoveUsesTheKeepingWrite(t *testing.T) { + src := mustRead(t, "migrate.go") + i := strings.Index(src, "func (m *Manager) doFlipRedeploy(") + if i < 0 { + t.Fatal("doFlipRedeploy not found") + } + body := src[i:] + body = body[:strings.Index(body, "\n}\n")] + if !strings.Contains(body, "m.persistDriveFlip(") || strings.Contains(body, "RedeployFromEnv(") || strings.Contains(body, "PersistUnitRedeployConfig(") { + t.Fatalf("doFlipRedeploy must persist through persistDriveFlip only:\n%s", body) + } +} + +func mustRead(t *testing.T, p string) string { + t.Helper() + b, err := os.ReadFile(p) + if err != nil { + t.Fatal(err) + } + return string(b) +}