controller v0.240.0: seven defects from the any-tier proof and the first nightly rotation
gates / gates (push) Successful in 13s

R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record,
so the second-drive restore is no longer refused over an intact mirror.
R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps).
R-485: the backup card sizes the recovery unit and the mirror(s).
R-480: a held update's sentence leaves the card once the hold is lifted.
R-477: the update's off-site lookup is one snapshots call, no stats.
R-478: a copy older than this install's deploy does not count.
R-474: "delete backups" deletes the unit, the mirror(s) and the prefs.

Tests and red-proofs per row; evidence in felhom.eu
documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
This commit is contained in:
2026-09-13 19:26:50 +02:00
parent 0e3d831030
commit bdcbd50b42
20 changed files with 673 additions and 82 deletions
+65 -8
View File
@@ -148,6 +148,50 @@ func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePo
}
}
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
fresh := freshRestorePoint(now, maxAge)
return func(p UpdateRestorePoint) bool {
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
return false
}
return fresh(p)
}
}
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
func (m *Manager) currentDeployTime(name string) time.Time {
st, ok := m.GetStack(name)
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
return time.Time{}
}
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
if err != nil {
return time.Time{}
}
return t
}
func (m *Manager) markUpdateHeld(name string) {
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.updateHeld = true
}
m.mu.Unlock()
}
func fmtDeployTime(t time.Time) string {
if t.IsZero() {
return "unknown"
}
return t.UTC().Format(time.RFC3339)
}
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
if len(pts) == 0 {
return "none"
@@ -189,11 +233,22 @@ func (m *Manager) guards() UpdateGuards {
}
func fillHoldReason(g UpdateGuards, st *Stack) {
if g == nil || st == nil || !st.Deployed {
if st == nil {
return
}
if held, why := g.HoldFor(st.Name); held {
st.HoldReason = why
held := false
if g != nil && st.Deployed {
if h, why := g.HoldFor(st.Name); h {
st.HoldReason, held = why, true
}
}
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
}
}
@@ -329,7 +384,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
m.mu.Unlock()
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
}
s.Updating, s.UpdateError = true, ""
s.Updating, s.UpdateError, s.updateHeld = true, "", false
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
m.mu.Unlock()
@@ -445,11 +500,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
maxAge := m.backupMaxAge()
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
deployedAt := m.currentDeployTime(name)
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
if ok {
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
} else {
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
fail(MsgUpdateJournalFailed, "journal write failed")
return
@@ -459,7 +515,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
return
}
now := m.now()
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
if !ok {
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
return
@@ -577,6 +633,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
} else if _, why := g.HoldFor(name); why != "" {
msg = why
m.markUpdateHeld(name)
}
_ = m.RefreshStatus()
m.clearJournal(name)
@@ -814,7 +871,7 @@ func (m *Manager) RecoverUpdates() []string {
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
m.mu.Lock()
if s, ok := m.stacks[name]; ok {
s.Updating, s.UpdateError = true, ""
s.Updating, s.UpdateError, s.updateHeld = true, "", false
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
}
m.updateResume = append(m.updateResume, name)