controller v0.240.0: seven defects from the any-tier proof and the first nightly rotation
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record, so the second-drive restore is no longer refused over an intact mirror. R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps). R-485: the backup card sizes the recovery unit and the mirror(s). R-480: a held update's sentence leaves the card once the hold is lifted. R-477: the update's off-site lookup is one snapshots call, no stats. R-478: a copy older than this install's deploy does not count. R-474: "delete backups" deletes the unit, the mirror(s) and the prefs. Tests and red-proofs per row; evidence in felhom.eu documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
This commit is contained in:
@@ -148,6 +148,50 @@ func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePo
|
||||
}
|
||||
}
|
||||
|
||||
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
|
||||
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
|
||||
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
|
||||
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
|
||||
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
|
||||
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
|
||||
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
|
||||
fresh := freshRestorePoint(now, maxAge)
|
||||
return func(p UpdateRestorePoint) bool {
|
||||
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
|
||||
return false
|
||||
}
|
||||
return fresh(p)
|
||||
}
|
||||
}
|
||||
|
||||
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
|
||||
func (m *Manager) currentDeployTime(name string) time.Time {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
|
||||
return time.Time{}
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
|
||||
if err != nil {
|
||||
return time.Time{}
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
func (m *Manager) markUpdateHeld(name string) {
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.updateHeld = true
|
||||
}
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
func fmtDeployTime(t time.Time) string {
|
||||
if t.IsZero() {
|
||||
return "unknown"
|
||||
}
|
||||
return t.UTC().Format(time.RFC3339)
|
||||
}
|
||||
|
||||
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
|
||||
if len(pts) == 0 {
|
||||
return "none"
|
||||
@@ -189,11 +233,22 @@ func (m *Manager) guards() UpdateGuards {
|
||||
}
|
||||
|
||||
func fillHoldReason(g UpdateGuards, st *Stack) {
|
||||
if g == nil || st == nil || !st.Deployed {
|
||||
if st == nil {
|
||||
return
|
||||
}
|
||||
if held, why := g.HoldFor(st.Name); held {
|
||||
st.HoldReason = why
|
||||
held := false
|
||||
if g != nil && st.Deployed {
|
||||
if h, why := g.HoldFor(st.Name); h {
|
||||
st.HoldReason, held = why, true
|
||||
}
|
||||
}
|
||||
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
|
||||
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
|
||||
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
|
||||
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
|
||||
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
|
||||
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
|
||||
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
|
||||
}
|
||||
}
|
||||
|
||||
@@ -329,7 +384,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
|
||||
m.mu.Unlock()
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
|
||||
}
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
||||
m.mu.Unlock()
|
||||
|
||||
@@ -445,11 +500,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
|
||||
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
|
||||
maxAge := m.backupMaxAge()
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
|
||||
deployedAt := m.currentDeployTime(name)
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
|
||||
if ok {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
@@ -459,7 +515,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
return
|
||||
}
|
||||
now := m.now()
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
|
||||
if !ok {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
||||
return
|
||||
@@ -577,6 +633,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
m.markUpdateHeld(name)
|
||||
}
|
||||
_ = m.RefreshStatus()
|
||||
m.clearJournal(name)
|
||||
@@ -814,7 +871,7 @@ func (m *Manager) RecoverUpdates() []string {
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
|
||||
}
|
||||
m.updateResume = append(m.updateResume, name)
|
||||
|
||||
Reference in New Issue
Block a user