controller v0.240.0: seven defects from the any-tier proof and the first nightly rotation
gates / gates (push) Successful in 13s

R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record,
so the second-drive restore is no longer refused over an intact mirror.
R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps).
R-485: the backup card sizes the recovery unit and the mirror(s).
R-480: a held update's sentence leaves the card once the hold is lifted.
R-477: the update's off-site lookup is one snapshots call, no stats.
R-478: a copy older than this install's deploy does not count.
R-474: "delete backups" deletes the unit, the mirror(s) and the prefs.

Tests and red-proofs per row; evidence in felhom.eu
documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
This commit is contained in:
2026-09-13 19:26:50 +02:00
parent 0e3d831030
commit bdcbd50b42
20 changed files with 673 additions and 82 deletions
+15 -11
View File
@@ -598,9 +598,15 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
return resp, nil
}
// GetStackBackupData returns information about backup data for a stack.
// drivePath is the app's home drive (HDD or system data path).
func (m *Manager) GetStackBackupData(name string, drivePath string) (*BackupDataResponse, error) {
// GetStackBackupData returns information about backup data for a stack — what the remove dialog
// sizes "delete backups" from. drivePath is the app's namespace root; mirrorDirs are the app's
// Tier-2 mirror directories (backup.Manager.Tier2MirrorDirsForApp — stacks cannot import backup).
//
// R-485 (v0.240.0): until v0.239.0 this read `<ns>/backups/primary/<app>/db-dumps` and a pre-v2
// `<ns>/backups/secondary/<app>/rsync` path — both dead for an app whose backups are the recovery
// unit and a v2 mirror — and answered `has_backups:false` over 484 MB (measured on demo-hp
// 2026-09-13). It now sizes the whole recovery unit and every mirror.
func (m *Manager) GetStackBackupData(name string, drivePath string, mirrorDirs []string) (*BackupDataResponse, error) {
_, ok := m.GetStack(name)
if !ok {
return nil, fmt.Errorf("stack %q not found", name)
@@ -617,14 +623,12 @@ func (m *Manager) GetStackBackupData(name string, drivePath string) (*BackupData
return resp, nil
}
// Check DB dump directory. drivePath is the felhom-data namespace ROOT (Model A: the in-guest
// drive mount itself), so backups/ sits directly under it: <nsRoot>/backups/primary/<stack>/db-dumps
dbDumpPath := filepath.Join(drivePath, "backups", "primary", name, "db-dumps")
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(dbDumpPath))
// Check cross-drive rsync directory: <nsRoot>/backups/secondary/<stack>/rsync
rsyncPath := filepath.Join(drivePath, "backups", "secondary", name, "rsync")
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(rsyncPath))
// The recovery unit — definition, db-dumps and volume-dumps together. drivePath is the felhom-data
// namespace ROOT (Model A: the in-guest drive mount itself), so backups/ sits directly under it.
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(appbackup.RecoveryUnitPath(drivePath, name)))
for _, d := range mirrorDirs {
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(d))
}
if m.isDebug() {
for _, p := range resp.BackupPaths {
+3
View File
@@ -149,6 +149,9 @@ type Stack struct {
UpdatePhase string `json:"update_phase,omitempty"`
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
UpdateError string `json:"update_error,omitempty"`
// updateHeld (R-480, v0.240.0) — the last update ended HELD, so UpdateError is the hold's own
// sentence („… leállítva marad"). It is shown only while that hold is in force; fillHoldReason.
updateHeld bool
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
@@ -0,0 +1,93 @@
package stacks
import (
"context"
"errors"
"testing"
"time"
)
// R-480 — the failed update's sentence goes when the hold it announced is lifted, or the app is gone.
func heldFailedUpdate(t *testing.T) (*Manager, *fakeGuards) {
t.Helper()
m, _, g, _ := newSlice4Manager(t)
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdatePhase != UpdatePhaseFailed || st.UpdateError != "HELD-SENTENCE" {
t.Fatalf("fixture: a held failure must show the hold sentence while held; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
return m, g
}
func TestR480_TheFailureSentenceGoesWhenItsHoldIsLifted(t *testing.T) {
m, g := heldFailedUpdate(t)
g.mu.Lock()
g.held, g.holdWhy = false, "" // a successful restore cleared the hold
g.mu.Unlock()
if st, _ := m.GetStack("nextcloud"); st.UpdateError != "" || st.UpdatePhase != "" {
t.Errorf("GetStack after the hold is lifted: phase=%q err=%q — the card would say a running app is stopped", st.UpdatePhase, st.UpdateError)
}
for _, st := range m.GetStacks() {
if st.Name == "nextcloud" && st.UpdateError != "" {
t.Errorf("GetStacks after the hold is lifted still carries %q", st.UpdateError)
}
}
}
func TestR480_ARemovedAppShowsNoUpdateOutcome(t *testing.T) {
m, _ := heldFailedUpdate(t)
m.mu.Lock()
m.stacks["nextcloud"].Deployed = false
m.mu.Unlock()
if st, _ := m.GetStack("nextcloud"); st.UpdateError != "" {
t.Errorf("a removed app must not carry the held update's sentence, got %q", st.UpdateError)
}
}
func TestR480_AFailureThatHeldNothingKeepsItsSentence(t *testing.T) {
m, _, _, c := newSlice4Manager(t)
c.fail["pull"] = errors.New("manifest unknown")
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
st := waitUpdateDone(t, m, "nextcloud")
if st.UpdateError != MsgUpdatePullFailed || st.UpdatePhase != UpdatePhaseFailed {
t.Errorf("a pull failure held nothing and its sentence is still true; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
}
// R-478 — a copy older than THIS install's deploy does not carry the update.
func runWithDeployedAt(t *testing.T, deployedAgo time.Duration) *fakeGuards {
t.Helper()
m, _, g, _ := newSlice4Manager(t)
m.mu.Lock()
if m.stacks["nextcloud"].AppConfig == nil {
m.stacks["nextcloud"].AppConfig = &AppConfig{}
}
m.stacks["nextcloud"].AppConfig.DeployedAt = slice4T0.Add(-deployedAgo).Format(time.RFC3339)
m.mu.Unlock()
g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-2 * time.Hour)}}
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-time.Minute)}}
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
t.Fatal(err)
}
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdatePhase != UpdatePhaseDone {
t.Fatalf("phase=%q err=%q", st.UpdatePhase, st.UpdateError)
}
return g
}
func TestR478_ACopyOlderThanThisInstallDoesNotCount(t *testing.T) {
if g := runWithDeployedAt(t, 10*time.Minute); !hasGuardCall(g, "BackupNow") {
t.Errorf("a 2-hour-old unit under a 10-minute-old install belongs to the previous install — back up first; calls=%v", g.callList())
}
// Control: the same unit under a 3-hour-old install is this install's own copy.
if g := runWithDeployedAt(t, 3*time.Hour); hasGuardCall(g, "BackupNow") {
t.Errorf("control: a copy newer than the deploy must carry the update with no backup; calls=%v", g.callList())
}
}
@@ -0,0 +1,55 @@
package stacks
import (
"os"
"path/filepath"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
)
// R-485 — the backup card sizes the recovery unit and the mirror, not two dead paths.
//
// COMPANION RED-PROOF (REPORT.md): put the old `db-dumps` + `secondary/<app>/rsync` paths back —
// this fails with has_backups=false over a unit that exists.
func TestR485_BackupCardSeesTheUnitAndTheMirror(t *testing.T) {
m, _ := newPinManager(t, pinTplOld, pinTplNew, "deployed: true\nenv: {}\n")
ns := t.TempDir()
unit := appbackup.RecoveryUnitPath(ns, "nextcloud")
if err := os.MkdirAll(filepath.Join(unit, "volume-dumps"), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(unit, "volume-dumps", "x.tar"), make([]byte, 4096), 0o644); err != nil {
t.Fatal(err)
}
mirror := filepath.Join(t.TempDir(), "backups", "secondary", "nextcloud")
if err := os.MkdirAll(mirror, 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(mirror, "manifest.json"), []byte("{}"), 0o644); err != nil {
t.Fatal(err)
}
resp, err := m.GetStackBackupData("nextcloud", ns, []string{mirror})
if err != nil {
t.Fatal(err)
}
if !resp.HasBackups {
t.Fatal("a recovery unit on disk IS a backup — has_backups must be true")
}
var seen []string
for _, p := range resp.BackupPaths {
if p.Exists {
seen = append(seen, p.Path)
}
}
if len(seen) != 2 || seen[0] != unit || seen[1] != mirror {
t.Errorf("want the unit and the mirror, got %v", seen)
}
if resp.BackupPaths[0].SizeBytes < 4096 {
t.Errorf("the unit's size must include its volume dumps, got %d", resp.BackupPaths[0].SizeBytes)
}
// No unit, no mirror: nothing to delete, honestly.
if resp, _ := m.GetStackBackupData("nextcloud", t.TempDir(), nil); resp.HasBackups {
t.Error("nothing on disk must be has_backups=false")
}
}
+65 -8
View File
@@ -148,6 +148,50 @@ func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePo
}
}
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
fresh := freshRestorePoint(now, maxAge)
return func(p UpdateRestorePoint) bool {
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
return false
}
return fresh(p)
}
}
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
func (m *Manager) currentDeployTime(name string) time.Time {
st, ok := m.GetStack(name)
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
return time.Time{}
}
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
if err != nil {
return time.Time{}
}
return t
}
func (m *Manager) markUpdateHeld(name string) {
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.updateHeld = true
}
m.mu.Unlock()
}
func fmtDeployTime(t time.Time) string {
if t.IsZero() {
return "unknown"
}
return t.UTC().Format(time.RFC3339)
}
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
if len(pts) == 0 {
return "none"
@@ -189,11 +233,22 @@ func (m *Manager) guards() UpdateGuards {
}
func fillHoldReason(g UpdateGuards, st *Stack) {
if g == nil || st == nil || !st.Deployed {
if st == nil {
return
}
if held, why := g.HoldFor(st.Name); held {
st.HoldReason = why
held := false
if g != nil && st.Deployed {
if h, why := g.HoldFor(st.Name); h {
st.HoldReason, held = why, true
}
}
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
}
}
@@ -329,7 +384,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
m.mu.Unlock()
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
}
s.Updating, s.UpdateError = true, ""
s.Updating, s.UpdateError, s.updateHeld = true, "", false
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
m.mu.Unlock()
@@ -445,11 +500,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
maxAge := m.backupMaxAge()
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
deployedAt := m.currentDeployTime(name)
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
if ok {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
} else {
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
fail(MsgUpdateJournalFailed, "journal write failed")
return
@@ -459,7 +515,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
return
}
now := m.now()
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
if !ok {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
return
@@ -577,6 +633,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
m.markUpdateHeld(name)
}
_ = m.RefreshStatus()
m.clearJournal(name)
@@ -814,7 +871,7 @@ func (m *Manager) RecoverUpdates() []string {
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Updating, s.UpdateError = true, ""
s.Updating, s.UpdateError, s.updateHeld = true, "", false
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
}
m.updateResume = append(m.updateResume, name)