controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s
gates / gates (push) Successful in 14s
Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1 (own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable counts as absent with a WARN) and leans on the first FRESH copy; the backup_max_age rule applies to whichever tier is chosen. No copy anywhere: back up first. Refused only when nothing exists and no backup can be taken. RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured unit proven current. The hold names the tier (második meghajtó / saját meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text. A successful off-site restore now lifts an update hold. The backups page still uses Tier2UnitRestorePoint unchanged. Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
@@ -6,6 +6,7 @@ import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||||
@@ -82,7 +83,7 @@ const (
|
||||
MsgUpdateAlreadyFmt = "A(z) %s frissítése már folyamatban van."
|
||||
MsgUpdateBusy = "A frissítés most nem indítható: mentés/visszaállítás folyamatban. Próbáld újra, ha befejeződött."
|
||||
MsgUpdateMigrating = "A frissítés most nem indítható: adatáthelyezés folyamatban."
|
||||
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani. Kapcsold be a 2. mentést az alkalmazás mentési beállításainál a Mentések oldalon, és várd meg az első sikeres másolatot — utána a frissítés elindítható."
|
||||
MsgUpdateNoBackupFmt = "A(z) %s nem frissíthető, mert nincs olyan biztonsági mentése, amelyből vissza lehetne állítani, és most új mentés sem készíthető róla. Ellenőrizd a Mentések oldalon, hogy az alkalmazás meghajtója elérhető-e — utána a frissítés elindítható."
|
||||
MsgUpdateDiskFmt = "Nincs elég szabad hely a frissítéshez: %.1f GB szabad, az új verzió letöltéséhez legalább %.0f GB szükséges."
|
||||
MsgUpdateBackupFailFmt = "A frissítés nem indult el, mert a frissítés előtti biztonsági mentés nem sikerült: %v. Az alkalmazás változatlanul fut tovább."
|
||||
MsgUpdateBackupNoUnit = "A frissítés nem indult el: a frissítés előtti mentés lefutott, de nem jött létre friss, visszaállítható másolat. Az alkalmazás változatlanul fut tovább."
|
||||
@@ -107,11 +108,55 @@ const updateSettleWindow = 60 * time.Second
|
||||
// updatePollEvery is how often the health wait re-reads the stack.
|
||||
const updatePollEvery = 5 * time.Second
|
||||
|
||||
// UpdateRestorePoint is the precondition answer, reduced to what the update needs.
|
||||
// Backup tiers (R-475), mirroring backup.UpdateTier* — stacks cannot import backup, so
|
||||
// TestR475_TierConstantsAgree (cmd/controller) pins the two sets equal.
|
||||
const (
|
||||
UpdateTierLocal = 1 // the app's own recovery unit
|
||||
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
|
||||
UpdateTierOffsite = 3 // off-site
|
||||
)
|
||||
|
||||
// UpdateRestorePoint is one proven, restorable copy the update may lean on. The backup side returns
|
||||
// only proven, restorable copies (never an attempt clock, never an unopenable unit), so there is no
|
||||
// "maybe" field here to forget to check.
|
||||
type UpdateRestorePoint struct {
|
||||
Restorable bool // an openable recovery unit exists in the Tier-2 copy
|
||||
Proven bool // a copy actually succeeded (never an attempt clock)
|
||||
ProvenAt time.Time // when the data in that copy was last proven copied
|
||||
Tier int // UpdateTierSecondDrive / UpdateTierLocal / UpdateTierOffsite
|
||||
ProvenAt time.Time // when the data in that copy was last proven written
|
||||
}
|
||||
|
||||
func updateTierName(tier int) string {
|
||||
switch tier {
|
||||
case UpdateTierSecondDrive:
|
||||
return "Tier 2 (second drive)"
|
||||
case UpdateTierLocal:
|
||||
return "Tier 1 (own recovery unit)"
|
||||
case UpdateTierOffsite:
|
||||
return "Tier 3 (off-site)"
|
||||
}
|
||||
return fmt.Sprintf("tier %d", tier)
|
||||
}
|
||||
|
||||
// freshRestorePoint is THE age rule (backup_max_age), applied to whichever tier is being considered
|
||||
// — R-475 Scenario M: a stale copy on ANY tier is stale.
|
||||
//
|
||||
// COMPANION RED-PROOF M (REPORT.md): check the age only for Tier 2 (let any other tier through
|
||||
// whatever its age). TestR475_M_TheAgeRuleAppliesToTheChosenTier then fails: a 30-hour-old copy of
|
||||
// the app's own unit carries the update with no backup first.
|
||||
func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePoint) bool {
|
||||
return func(p UpdateRestorePoint) bool {
|
||||
return !p.ProvenAt.IsZero() && now.Sub(p.ProvenAt) <= maxAge
|
||||
}
|
||||
}
|
||||
|
||||
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
|
||||
if len(pts) == 0 {
|
||||
return "none"
|
||||
}
|
||||
parts := make([]string, 0, len(pts))
|
||||
for _, p := range pts {
|
||||
parts = append(parts, fmt.Sprintf("%s at %s (%s old)", updateTierName(p.Tier), p.ProvenAt.UTC().Format(time.RFC3339), now.Sub(p.ProvenAt).Round(time.Minute)))
|
||||
}
|
||||
return strings.Join(parts, "; ")
|
||||
}
|
||||
|
||||
// UpdateGuards is everything the update needs from the backup side. The stacks package cannot import
|
||||
@@ -119,10 +164,14 @@ type UpdateRestorePoint struct {
|
||||
type UpdateGuards interface {
|
||||
HoldFor(name string) (bool, string)
|
||||
Busy(name string) (bool, string)
|
||||
RestorePoint(name string) (UpdateRestorePoint, error)
|
||||
// RestorePoints walks the tiers in preference order (2, 1, 3) and returns the first copy accept
|
||||
// admits (nil = any), whether one was found, and every copy looked at (R-475).
|
||||
RestorePoints(ctx context.Context, name string, accept func(UpdateRestorePoint) bool) (UpdateRestorePoint, bool, []UpdateRestorePoint)
|
||||
// CanBackUp reports whether "back up first" can run for this app now (R-475 Scenario L).
|
||||
CanBackUp(name string) (bool, string)
|
||||
BackupNow(ctx context.Context, name string) error
|
||||
SafetyDump(ctx context.Context, name string) ([]string, error)
|
||||
HoldAfterFailedUpdate(name string, at, provenCopyAt time.Time) error
|
||||
HoldAfterFailedUpdate(name string, at time.Time, rp UpdateRestorePoint) error
|
||||
}
|
||||
|
||||
// SetUpdateGuards wires the backup side. INIT-ONLY. Unwired, every update is refused (fail closed):
|
||||
@@ -199,10 +248,19 @@ func (m *Manager) UpdatePreflight(name string) *UpdateRefusal {
|
||||
if m.IsMigrating() {
|
||||
return m.refuseUpdate(name, "migrating", MsgUpdateMigrating, "a data migration is running")
|
||||
}
|
||||
rp, err := g.RestorePoint(name)
|
||||
if err != nil || !rp.Restorable || !rp.Proven {
|
||||
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
|
||||
fmt.Sprintf("no restorable proven Tier-2 unit (restorable=%v proven=%v err=%v)", rp.Restorable, rp.Proven, err))
|
||||
// R-475: any tier counts, and an app with no copy at all is backed up first by the job. So the only
|
||||
// refusal left here is Scenario L — no copy on any tier AND no way to make one now. (With a copy
|
||||
// but no way to back up, the job still applies the age rule and refuses then if the copy is stale.)
|
||||
// Tier 3 is looked at only on this branch, so an ordinary update never waits on the network here.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): drop the `!found` condition. TestR475_L then fails — an app
|
||||
// with a copy but no way to back up is refused too.
|
||||
if canBackUp, why := g.CanBackUp(name); !canBackUp {
|
||||
if _, found, seen := g.RestorePoints(context.Background(), name, nil); !found {
|
||||
return m.refuseUpdate(name, "no_backup", fmt.Sprintf(MsgUpdateNoBackupFmt, name),
|
||||
fmt.Sprintf("no copy on any tier (found: %s) and no backup can be taken now: %s", describeRestorePoints(m.now(), seen), why))
|
||||
}
|
||||
m.logger.Printf("[WARN] [stacks] update %s: no backup can be taken now (%s) — an existing copy must carry the update", name, why)
|
||||
}
|
||||
if ref := m.updateMemoryRefusal(name, st); ref != nil {
|
||||
return ref
|
||||
@@ -383,15 +441,15 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
fail(MsgUpdateNoGuards, "no UpdateGuards wired")
|
||||
return
|
||||
}
|
||||
rp, err := g.RestorePoint(name)
|
||||
if err != nil || !rp.Restorable || !rp.Proven {
|
||||
fail(fmt.Sprintf(MsgUpdateNoBackupFmt, name), fmt.Sprintf("precondition vanished: restorable=%v proven=%v err=%v", rp.Restorable, rp.Proven, err))
|
||||
return
|
||||
}
|
||||
|
||||
// R-475: the precondition is a copy on ANY tier, chosen in the order 2, 1, 3, and the age rule
|
||||
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
|
||||
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
|
||||
maxAge := m.backupMaxAge()
|
||||
if age := start.Sub(rp.ProvenAt); age > maxAge {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: the proven copy is %s old (limit %s) — backing up first", name, age.Round(time.Minute), maxAge)
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
|
||||
if ok {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
@@ -400,15 +458,16 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
fail(fmt.Sprintf(MsgUpdateBackupFailFmt, err), "pre-update backup: "+err.Error())
|
||||
return
|
||||
}
|
||||
rp, err = g.RestorePoint(name)
|
||||
if err != nil || !rp.Restorable || !rp.Proven || m.now().Sub(rp.ProvenAt) > maxAge {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup: restorable=%v proven=%v at=%s err=%v", rp.Restorable, rp.Proven, rp.ProvenAt.Format(time.RFC3339), err))
|
||||
now := m.now()
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
|
||||
if !ok {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
||||
return
|
||||
}
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met — proven copy from %s (%s old, limit %s)", name, rp.ProvenAt.UTC().Format(time.RFC3339), age.Round(time.Minute), maxAge)
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met after the backup — %s copy from %s", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339))
|
||||
}
|
||||
entry.ProvenCopyAt = rp.ProvenAt.UTC().Format(time.RFC3339)
|
||||
entry.ProvenTier = rp.Tier
|
||||
|
||||
// SAFETY DUMP BEFORE THE PIN MOVES — "a minute ago", before any migration can have run.
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseSafetyDump) {
|
||||
@@ -472,20 +531,20 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
}
|
||||
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseStarting) {
|
||||
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "journal write failed before up")
|
||||
m.failAndHold(ctx, name, dir, env, rp, "journal write failed before up")
|
||||
return
|
||||
}
|
||||
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
// Containers may already have been recreated on the new image — something may have run.
|
||||
m.failAndHold(ctx, name, dir, env, rp.ProvenAt, "compose up failed: "+err.Error())
|
||||
m.failAndHold(ctx, name, dir, env, rp, "compose up failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
m.verifyAndConclude(ctx, name, dir, env, rp.ProvenAt, start, &entry)
|
||||
m.verifyAndConclude(ctx, name, dir, env, rp, start, &entry)
|
||||
}
|
||||
|
||||
// verifyAndConclude is the TRUTH half (R-443): success is declared only after the app's health is
|
||||
// known, and a failure holds the app.
|
||||
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, provenAt, start time.Time, entry *updateJournalEntry) {
|
||||
func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, start time.Time, entry *updateJournalEntry) {
|
||||
if !m.enterUpdatePhase(name, entry, UpdatePhaseVerifying) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: could not journal the verifying phase — verifying anyway", name)
|
||||
}
|
||||
@@ -493,7 +552,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
|
||||
waitStart := m.now()
|
||||
healthy, detail := m.updateHealth(ctx, name, timeout)
|
||||
if !healthy {
|
||||
m.failAndHold(ctx, name, dir, env, provenAt, "not healthy: "+detail)
|
||||
m.failAndHold(ctx, name, dir, env, rp, "not healthy: "+detail)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [stacks] update %s: healthy after %s (%s)", name, m.now().Sub(waitStart).Round(time.Second), detail)
|
||||
@@ -506,7 +565,7 @@ func (m *Manager) verifyAndConclude(ctx context.Context, name, dir string, env [
|
||||
}
|
||||
|
||||
// failAndHold is Scenario F: stop the app, record the hold, tell the customer the route back.
|
||||
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, provenAt time.Time, why string) {
|
||||
func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []string, rp UpdateRestorePoint, why string) {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s FAILED after the new version was started: %s — stopping and HOLDING the app; the pin stays on the new version (its migration may have run)", name, why)
|
||||
if _, err := m.updateCompose(dir, env, "down"); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: stopping the failed app also failed: %v", name, err)
|
||||
@@ -514,7 +573,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
|
||||
msg := MsgUpdateHoldUnsaved
|
||||
if g := m.guards(); g == nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: no UpdateGuards — the hold CANNOT be recorded", name)
|
||||
} else if err := g.HoldAfterFailedUpdate(name, m.now(), provenAt); err != nil {
|
||||
} else if err := g.HoldAfterFailedUpdate(name, m.now(), rp); err != nil {
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
@@ -619,6 +678,9 @@ type updateJournalEntry struct {
|
||||
PrevCompose string `json:"prev_compose,omitempty"`
|
||||
PrevApplied string `json:"prev_applied,omitempty"`
|
||||
ProvenCopyAt string `json:"proven_copy_at,omitempty"`
|
||||
// ProvenTier (R-475) — which tier ProvenCopyAt belongs to, so a resumed update that fails names
|
||||
// the right copy. 0 in a journal written by v0.238.1 or older.
|
||||
ProvenTier int `json:"proven_tier,omitempty"`
|
||||
}
|
||||
|
||||
type updateJournal struct {
|
||||
@@ -787,16 +849,17 @@ func (m *Manager) ResumeInterruptedUpdates(ctx context.Context) int {
|
||||
continue
|
||||
}
|
||||
provenAt, _ := time.Parse(time.RFC3339, e.ProvenCopyAt)
|
||||
rp := UpdateRestorePoint{Tier: e.ProvenTier, ProvenAt: provenAt}
|
||||
dir := filepath.Dir(st.ComposePath)
|
||||
go func(name, dir string, e updateJournalEntry, provenAt time.Time) {
|
||||
go func(name, dir string, e updateJournalEntry, rp UpdateRestorePoint) {
|
||||
env := m.stackEnv(dir)
|
||||
m.logger.Printf("[INFO] [stacks] update %s: resuming after a controller restart — `up -d` then the health wait", name)
|
||||
if _, err := m.updateCompose(dir, env, "up", "-d", "--remove-orphans"); err != nil {
|
||||
m.failAndHold(ctx, name, dir, env, provenAt, "resumed compose up failed: "+err.Error())
|
||||
m.failAndHold(ctx, name, dir, env, rp, "resumed compose up failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
m.verifyAndConclude(ctx, name, dir, env, provenAt, e.StartedAt, &e)
|
||||
}(name, dir, e, provenAt)
|
||||
m.verifyAndConclude(ctx, name, dir, env, rp, e.StartedAt, &e)
|
||||
}(name, dir, e, rp)
|
||||
}
|
||||
return len(names)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user