controller v0.240.0: seven defects from the any-tier proof and the first nightly rotation
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record, so the second-drive restore is no longer refused over an intact mirror. R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps). R-485: the backup card sizes the recovery unit and the mirror(s). R-480: a held update's sentence leaves the card once the hold is lifted. R-477: the update's off-site lookup is one snapshots call, no stats. R-478: a copy older than this install's deploy does not count. R-474: "delete backups" deletes the unit, the mirror(s) and the prefs. Tests and red-proofs per row; evidence in felhom.eu documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
This commit is contained in:
@@ -598,9 +598,15 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// GetStackBackupData returns information about backup data for a stack.
|
||||
// drivePath is the app's home drive (HDD or system data path).
|
||||
func (m *Manager) GetStackBackupData(name string, drivePath string) (*BackupDataResponse, error) {
|
||||
// GetStackBackupData returns information about backup data for a stack — what the remove dialog
|
||||
// sizes "delete backups" from. drivePath is the app's namespace root; mirrorDirs are the app's
|
||||
// Tier-2 mirror directories (backup.Manager.Tier2MirrorDirsForApp — stacks cannot import backup).
|
||||
//
|
||||
// R-485 (v0.240.0): until v0.239.0 this read `<ns>/backups/primary/<app>/db-dumps` and a pre-v2
|
||||
// `<ns>/backups/secondary/<app>/rsync` path — both dead for an app whose backups are the recovery
|
||||
// unit and a v2 mirror — and answered `has_backups:false` over 484 MB (measured on demo-hp
|
||||
// 2026-09-13). It now sizes the whole recovery unit and every mirror.
|
||||
func (m *Manager) GetStackBackupData(name string, drivePath string, mirrorDirs []string) (*BackupDataResponse, error) {
|
||||
_, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("stack %q not found", name)
|
||||
@@ -617,14 +623,12 @@ func (m *Manager) GetStackBackupData(name string, drivePath string) (*BackupData
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// Check DB dump directory. drivePath is the felhom-data namespace ROOT (Model A: the in-guest
|
||||
// drive mount itself), so backups/ sits directly under it: <nsRoot>/backups/primary/<stack>/db-dumps
|
||||
dbDumpPath := filepath.Join(drivePath, "backups", "primary", name, "db-dumps")
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(dbDumpPath))
|
||||
|
||||
// Check cross-drive rsync directory: <nsRoot>/backups/secondary/<stack>/rsync
|
||||
rsyncPath := filepath.Join(drivePath, "backups", "secondary", name, "rsync")
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(rsyncPath))
|
||||
// The recovery unit — definition, db-dumps and volume-dumps together. drivePath is the felhom-data
|
||||
// namespace ROOT (Model A: the in-guest drive mount itself), so backups/ sits directly under it.
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(appbackup.RecoveryUnitPath(drivePath, name)))
|
||||
for _, d := range mirrorDirs {
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(d))
|
||||
}
|
||||
|
||||
if m.isDebug() {
|
||||
for _, p := range resp.BackupPaths {
|
||||
|
||||
@@ -149,6 +149,9 @@ type Stack struct {
|
||||
UpdatePhase string `json:"update_phase,omitempty"`
|
||||
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
|
||||
UpdateError string `json:"update_error,omitempty"`
|
||||
// updateHeld (R-480, v0.240.0) — the last update ended HELD, so UpdateError is the hold's own
|
||||
// sentence („… leállítva marad"). It is shown only while that hold is in force; fillHoldReason.
|
||||
updateHeld bool
|
||||
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
|
||||
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
|
||||
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-480 — the failed update's sentence goes when the hold it announced is lifted, or the app is gone.
|
||||
|
||||
func heldFailedUpdate(t *testing.T) (*Manager, *fakeGuards) {
|
||||
t.Helper()
|
||||
m, _, g, _ := newSlice4Manager(t)
|
||||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdatePhase != UpdatePhaseFailed || st.UpdateError != "HELD-SENTENCE" {
|
||||
t.Fatalf("fixture: a held failure must show the hold sentence while held; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
return m, g
|
||||
}
|
||||
|
||||
func TestR480_TheFailureSentenceGoesWhenItsHoldIsLifted(t *testing.T) {
|
||||
m, g := heldFailedUpdate(t)
|
||||
g.mu.Lock()
|
||||
g.held, g.holdWhy = false, "" // a successful restore cleared the hold
|
||||
g.mu.Unlock()
|
||||
if st, _ := m.GetStack("nextcloud"); st.UpdateError != "" || st.UpdatePhase != "" {
|
||||
t.Errorf("GetStack after the hold is lifted: phase=%q err=%q — the card would say a running app is stopped", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
for _, st := range m.GetStacks() {
|
||||
if st.Name == "nextcloud" && st.UpdateError != "" {
|
||||
t.Errorf("GetStacks after the hold is lifted still carries %q", st.UpdateError)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestR480_ARemovedAppShowsNoUpdateOutcome(t *testing.T) {
|
||||
m, _ := heldFailedUpdate(t)
|
||||
m.mu.Lock()
|
||||
m.stacks["nextcloud"].Deployed = false
|
||||
m.mu.Unlock()
|
||||
if st, _ := m.GetStack("nextcloud"); st.UpdateError != "" {
|
||||
t.Errorf("a removed app must not carry the held update's sentence, got %q", st.UpdateError)
|
||||
}
|
||||
}
|
||||
|
||||
func TestR480_AFailureThatHeldNothingKeepsItsSentence(t *testing.T) {
|
||||
m, _, _, c := newSlice4Manager(t)
|
||||
c.fail["pull"] = errors.New("manifest unknown")
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdateError != MsgUpdatePullFailed || st.UpdatePhase != UpdatePhaseFailed {
|
||||
t.Errorf("a pull failure held nothing and its sentence is still true; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
}
|
||||
|
||||
// R-478 — a copy older than THIS install's deploy does not carry the update.
|
||||
|
||||
func runWithDeployedAt(t *testing.T, deployedAgo time.Duration) *fakeGuards {
|
||||
t.Helper()
|
||||
m, _, g, _ := newSlice4Manager(t)
|
||||
m.mu.Lock()
|
||||
if m.stacks["nextcloud"].AppConfig == nil {
|
||||
m.stacks["nextcloud"].AppConfig = &AppConfig{}
|
||||
}
|
||||
m.stacks["nextcloud"].AppConfig.DeployedAt = slice4T0.Add(-deployedAgo).Format(time.RFC3339)
|
||||
m.mu.Unlock()
|
||||
g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-2 * time.Hour)}}
|
||||
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-time.Minute)}}
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdatePhase != UpdatePhaseDone {
|
||||
t.Fatalf("phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
func TestR478_ACopyOlderThanThisInstallDoesNotCount(t *testing.T) {
|
||||
if g := runWithDeployedAt(t, 10*time.Minute); !hasGuardCall(g, "BackupNow") {
|
||||
t.Errorf("a 2-hour-old unit under a 10-minute-old install belongs to the previous install — back up first; calls=%v", g.callList())
|
||||
}
|
||||
// Control: the same unit under a 3-hour-old install is this install's own copy.
|
||||
if g := runWithDeployedAt(t, 3*time.Hour); hasGuardCall(g, "BackupNow") {
|
||||
t.Errorf("control: a copy newer than the deploy must carry the update with no backup; calls=%v", g.callList())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// R-485 — the backup card sizes the recovery unit and the mirror, not two dead paths.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): put the old `db-dumps` + `secondary/<app>/rsync` paths back —
|
||||
// this fails with has_backups=false over a unit that exists.
|
||||
func TestR485_BackupCardSeesTheUnitAndTheMirror(t *testing.T) {
|
||||
m, _ := newPinManager(t, pinTplOld, pinTplNew, "deployed: true\nenv: {}\n")
|
||||
ns := t.TempDir()
|
||||
unit := appbackup.RecoveryUnitPath(ns, "nextcloud")
|
||||
if err := os.MkdirAll(filepath.Join(unit, "volume-dumps"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(unit, "volume-dumps", "x.tar"), make([]byte, 4096), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mirror := filepath.Join(t.TempDir(), "backups", "secondary", "nextcloud")
|
||||
if err := os.MkdirAll(mirror, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(mirror, "manifest.json"), []byte("{}"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
resp, err := m.GetStackBackupData("nextcloud", ns, []string{mirror})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !resp.HasBackups {
|
||||
t.Fatal("a recovery unit on disk IS a backup — has_backups must be true")
|
||||
}
|
||||
var seen []string
|
||||
for _, p := range resp.BackupPaths {
|
||||
if p.Exists {
|
||||
seen = append(seen, p.Path)
|
||||
}
|
||||
}
|
||||
if len(seen) != 2 || seen[0] != unit || seen[1] != mirror {
|
||||
t.Errorf("want the unit and the mirror, got %v", seen)
|
||||
}
|
||||
if resp.BackupPaths[0].SizeBytes < 4096 {
|
||||
t.Errorf("the unit's size must include its volume dumps, got %d", resp.BackupPaths[0].SizeBytes)
|
||||
}
|
||||
// No unit, no mirror: nothing to delete, honestly.
|
||||
if resp, _ := m.GetStackBackupData("nextcloud", t.TempDir(), nil); resp.HasBackups {
|
||||
t.Error("nothing on disk must be has_backups=false")
|
||||
}
|
||||
}
|
||||
@@ -148,6 +148,50 @@ func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePo
|
||||
}
|
||||
}
|
||||
|
||||
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
|
||||
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
|
||||
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
|
||||
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
|
||||
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
|
||||
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
|
||||
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
|
||||
fresh := freshRestorePoint(now, maxAge)
|
||||
return func(p UpdateRestorePoint) bool {
|
||||
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
|
||||
return false
|
||||
}
|
||||
return fresh(p)
|
||||
}
|
||||
}
|
||||
|
||||
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
|
||||
func (m *Manager) currentDeployTime(name string) time.Time {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
|
||||
return time.Time{}
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
|
||||
if err != nil {
|
||||
return time.Time{}
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
func (m *Manager) markUpdateHeld(name string) {
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.updateHeld = true
|
||||
}
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
func fmtDeployTime(t time.Time) string {
|
||||
if t.IsZero() {
|
||||
return "unknown"
|
||||
}
|
||||
return t.UTC().Format(time.RFC3339)
|
||||
}
|
||||
|
||||
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
|
||||
if len(pts) == 0 {
|
||||
return "none"
|
||||
@@ -189,11 +233,22 @@ func (m *Manager) guards() UpdateGuards {
|
||||
}
|
||||
|
||||
func fillHoldReason(g UpdateGuards, st *Stack) {
|
||||
if g == nil || st == nil || !st.Deployed {
|
||||
if st == nil {
|
||||
return
|
||||
}
|
||||
if held, why := g.HoldFor(st.Name); held {
|
||||
st.HoldReason = why
|
||||
held := false
|
||||
if g != nil && st.Deployed {
|
||||
if h, why := g.HoldFor(st.Name); h {
|
||||
st.HoldReason, held = why, true
|
||||
}
|
||||
}
|
||||
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
|
||||
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
|
||||
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
|
||||
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
|
||||
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
|
||||
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
|
||||
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
|
||||
}
|
||||
}
|
||||
|
||||
@@ -329,7 +384,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
|
||||
m.mu.Unlock()
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
|
||||
}
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
||||
m.mu.Unlock()
|
||||
|
||||
@@ -445,11 +500,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
|
||||
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
|
||||
maxAge := m.backupMaxAge()
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
|
||||
deployedAt := m.currentDeployTime(name)
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
|
||||
if ok {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
@@ -459,7 +515,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
return
|
||||
}
|
||||
now := m.now()
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
|
||||
if !ok {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
||||
return
|
||||
@@ -577,6 +633,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
m.markUpdateHeld(name)
|
||||
}
|
||||
_ = m.RefreshStatus()
|
||||
m.clearJournal(name)
|
||||
@@ -814,7 +871,7 @@ func (m *Manager) RecoverUpdates() []string {
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
|
||||
}
|
||||
m.updateResume = append(m.updateResume, name)
|
||||
|
||||
Reference in New Issue
Block a user