controller v0.269.0: whole restore from the second drive; crash loops stopped; exact image digests; steps judged by their own .felhom.yml (decisions 26-28, R-661 R-666 R-667 R-668 R-664 R-665 R-662, 09 6.4 part 6)
gates / gates (push) Successful in 27s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-24 12:18:39 +02:00
parent 7c3b3a9694
commit 3c6b49b31c
141 changed files with 3401 additions and 237 deletions
+69 -2
View File
@@ -620,7 +620,9 @@ func (m *Manager) UpdateHeldStacks() map[string]bool {
}
var out map[string]bool
for _, h := range m.settings.ListRestoreHolds() {
if h.Reason != settings.HoldReasonUpdateFailed {
// v0.269.0 (decision 28): an app the box stopped for a crash loop / OOM storm is stopped BY THE
// PRODUCT and has its own event (app_stopped_unhealthy) — the same class as an update hold.
if h.Reason != settings.HoldReasonUpdateFailed && h.Reason != settings.HoldReasonUnhealthyStop {
continue
}
if out == nil {
@@ -659,8 +661,16 @@ func (m *Manager) WholeOnTier(stackName string, tier int) bool {
switch tier {
case UpdateTierOffsite:
return true
case UpdateTierLocal, UpdateTierSecondDrive:
case UpdateTierLocal:
return !m.HasDriveFileLegs(stackName)
case UpdateTierSecondDrive:
if !m.HasDriveFileLegs(stackName) {
return true
}
// v0.269.0 (decision 26): a file app is whole on the second drive when the mirror holds BOTH an
// openable unit and its file legs — RestoreTier2Whole brings back both.
cov, err := m.Tier2RestoreCoverage(stackName)
return err == nil && cov.CanRestoreUnit() && cov.CanRestore()
}
return false
}
@@ -737,3 +747,60 @@ func (m *Manager) UpdateHold(stackName string) (settings.RestoreHold, bool) {
}
return h, true
}
// ── Decision 28 (v0.269.0): the box stops an app in a crash loop or an out-of-memory storm ─────────
// UnhealthyRepeatWindow is how soon a second stop counts as a repeat: the sentence then says support is
// informed.
const UnhealthyRepeatWindow = 24 * time.Hour
// HoldUnhealthy records that the box stopped `stack` (kind "crash_loop" or "oom_storm") and returns the
// trip number: 1, or 2 when the previous stop was less than 24 h ago. Same store as every hold, so no
// start path — the boot sweep, the drive gate, the nightly legs — revives it silently.
func (m *Manager) HoldUnhealthy(stack, kind string, at time.Time) (int, error) {
if m == nil || m.settings == nil {
return 0, fmt.Errorf("no settings wired — the unhealthy stop of %s cannot be recorded", stack)
}
trip := 1
if last, ok := m.settings.LastUnhealthyStop(stack); ok && at.Sub(last) < UnhealthyRepeatWindow {
trip = 2
}
h := settings.RestoreHold{Stack: stack, At: at.UTC().Format(time.RFC3339), Reason: settings.HoldReasonUnhealthyStop,
UnhealthyKind: kind, Trip: trip}
if err := m.settings.SetRestoreHold(h); err != nil {
return trip, fmt.Errorf("persisting the unhealthy stop of %s: %w", stack, err)
}
if err := m.settings.RecordUnhealthyStop(stack, at); err != nil {
m.logger.Printf("[WARN] [backup] %s: recording the stop time failed: %v", stack, err)
}
m.logger.Printf("[WARN] [backup] %s is STOPPED by the box: %s (trip %d within %s) — Start gives it one more try (decision 28)", stack, kind, trip, UnhealthyRepeatWindow)
return trip, nil
}
// LiftUnhealthyStop is the Start button's half: an unhealthy-stop hold is lifted, any other kind stays.
func (m *Manager) LiftUnhealthyStop(stack string) bool {
if m == nil || m.settings == nil {
return false
}
ok, err := m.settings.ClearUnhealthyStopHold(stack)
if err != nil {
m.logger.Printf("[ERROR] [backup] lifting the unhealthy stop of %s failed: %v", stack, err)
return false
}
return ok
}
// HoldKind names the kind of hold in force ("" when none): update_failed, unhealthy_stop, or "restore".
func (m *Manager) HoldKind(stack string) string {
if m == nil || m.settings == nil {
return ""
}
h, ok := m.settings.GetRestoreHold(stack)
if !ok {
return ""
}
if h.Reason == "" {
return "restore"
}
return h.Reason
}