1453cfc69b
gates / gates (push) Failing after 50s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
155 lines
6.1 KiB
Go
155 lines
6.1 KiB
Go
package backup
|
|
|
|
import (
|
|
"encoding/json"
|
|
"os"
|
|
"sort"
|
|
"time"
|
|
)
|
|
|
|
// R-550 (operator ruling "fix", 2026-09-17) — the restore record survives a restart.
|
|
//
|
|
// A DESIGN REVERSED, AND RECORDED AS SUCH. opstatus.go was deliberately in-memory ("same precedent as
|
|
// notification cooldowns"). Chaos night round 10 measured the cost: a restore accepted at 23:26:08Z,
|
|
// the box hard-reset four seconds later, and afterwards the status answered the Go zero value — the
|
|
// household had pressed a button, been told it started, and could never learn whether it finished.
|
|
// The operator reversed the choice for the RESTORE RECORD ONLY; notification cooldowns stay in memory.
|
|
//
|
|
// Shape: one JSON file in the controller's state directory (DataDir, beside settings.json), written
|
|
// atomically (atomicWrite: tmp + rename) at BOTH ends of an op. At startup a record still marked
|
|
// running is, by construction, a restore nothing is running any more: it becomes a terminal failure
|
|
// (Interrupted, RestoreInterruptedKey) and a per-app notice that stays until that app's next
|
|
// restore. LoadRestoreRecord reports the conversion ONCE, so the caller raises restore_interrupted once.
|
|
//
|
|
// No path set (tests that build a bare Manager, a box with backup disabled) = the old in-memory
|
|
// behaviour, silently: persistence is a property of the wired controller, not of every Manager.
|
|
|
|
// RestoreInterruptedKey names what the household reads when the box stopped mid-restore. It is
|
|
// SAVED in the restore record and read on the page afterwards, so it is rendered in the box's
|
|
// language at the moment it is written (release C, R-557).
|
|
const RestoreInterruptedKey = "note.restore.interrupted"
|
|
|
|
// restoreRecordFile is the on-disk shape.
|
|
type restoreRecordFile struct {
|
|
Running bool `json:"running"`
|
|
Op string `json:"op,omitempty"`
|
|
Stack string `json:"stack,omitempty"`
|
|
StartedAt time.Time `json:"started_at,omitempty"`
|
|
Last *RestoreOpResult `json:"last,omitempty"`
|
|
Interrupted map[string]RestoreOpResult `json:"interrupted,omitempty"`
|
|
}
|
|
|
|
// SetRestoreRecordPath wires persistence. Call before LoadRestoreRecord and before serving requests.
|
|
func (m *Manager) SetRestoreRecordPath(path string) {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
m.opRecordPath = path
|
|
}
|
|
|
|
// LoadRestoreRecord restores the record at startup. It returns the restore that was interrupted by the
|
|
// stop — non-nil exactly once per interruption — so the caller can log it and raise restore_interrupted.
|
|
func (m *Manager) LoadRestoreRecord() *RestoreOpResult {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
if m.opRecordPath == "" {
|
|
return nil
|
|
}
|
|
b, err := os.ReadFile(m.opRecordPath)
|
|
if err != nil {
|
|
if !os.IsNotExist(err) && m.logger != nil {
|
|
m.logger.Printf("[WARN] [backup] restore record unreadable (%v) — starting with no restore history", err)
|
|
}
|
|
return nil
|
|
}
|
|
var rec restoreRecordFile
|
|
if err := json.Unmarshal(b, &rec); err != nil {
|
|
if m.logger != nil {
|
|
m.logger.Printf("[WARN] [backup] restore record corrupt (%v) — starting with no restore history", err)
|
|
}
|
|
return nil
|
|
}
|
|
m.opLast = rec.Last
|
|
m.opInterrupted = rec.Interrupted
|
|
if !rec.Running || rec.Stack == "" {
|
|
if m.logger != nil && len(m.opInterrupted) > 0 {
|
|
m.logger.Printf("[INFO] [backup] restore record loaded: %d interrupted restore notice(s) still shown", len(m.opInterrupted))
|
|
}
|
|
return nil
|
|
}
|
|
res := RestoreOpResult{
|
|
Op: rec.Op, Stack: rec.Stack, OK: false,
|
|
Message: m.note(RestoreInterruptedKey), FinishedAt: time.Now(), Interrupted: true,
|
|
}
|
|
m.opLast = &res
|
|
if m.opInterrupted == nil {
|
|
m.opInterrupted = map[string]RestoreOpResult{}
|
|
}
|
|
m.opInterrupted[rec.Stack] = res
|
|
m.opRunning = false
|
|
if m.logger != nil {
|
|
m.logger.Printf("[WARN] [backup] restore of %s (%s, started %s) was INTERRUPTED by a controller stop — recorded as failed; the household is told to run it again",
|
|
rec.Stack, rec.Op, rec.StartedAt.Format(time.RFC3339))
|
|
}
|
|
m.persistRestoreRecordLocked()
|
|
out := res
|
|
return &out
|
|
}
|
|
|
|
// InterruptedRestore reports the standing interrupted-restore notice for one app, if any.
|
|
func (m *Manager) InterruptedRestore(stack string) (RestoreOpResult, bool) {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
r, ok := m.opInterrupted[stack]
|
|
return r, ok
|
|
}
|
|
|
|
// InterruptedRestores lists every standing notice, sorted by app, for the restore page.
|
|
func (m *Manager) InterruptedRestores() []RestoreOpResult {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
out := make([]RestoreOpResult, 0, len(m.opInterrupted))
|
|
for _, r := range m.opInterrupted {
|
|
out = append(out, r)
|
|
}
|
|
sort.Slice(out, func(i, j int) bool { return out[i].Stack < out[j].Stack })
|
|
return out
|
|
}
|
|
|
|
// persistRestoreRecordLocked writes the current op-status. Caller holds m.mu. A failed write is logged
|
|
// and the in-memory status carries on — the page still works for this process's lifetime.
|
|
func (m *Manager) persistRestoreRecordLocked() {
|
|
if m.opRecordPath == "" {
|
|
return
|
|
}
|
|
rec := restoreRecordFile{
|
|
Running: m.opRunning, Op: m.opName, Stack: m.opStack, StartedAt: m.opStartedAt,
|
|
Last: m.opLast, Interrupted: m.opInterrupted,
|
|
}
|
|
b, err := json.Marshal(rec)
|
|
if err == nil {
|
|
err = atomicWrite(m.opRecordPath, b, 0o600)
|
|
}
|
|
if err != nil && m.logger != nil {
|
|
m.logger.Printf("[WARN] [backup] could not persist the restore record to %s: %v (kept in memory)", m.opRecordPath, err)
|
|
}
|
|
}
|
|
|
|
// ClearInterruptedRestore drops an app's standing interrupted-restore notice and persists the record
|
|
// (R-552). Called when the app is REMOVED: the notice asks the household to run the restore again,
|
|
// and for an app that no longer exists that is a card about nothing, shown for ever — before this,
|
|
// only a new restore of the same app (BeginRestoreOp) ever cleared it. Reports whether a notice was
|
|
// there. Pinned by TestClearInterruptedRestore_DropsTheNoticeAndPersists.
|
|
func (m *Manager) ClearInterruptedRestore(stack string) bool {
|
|
if m == nil || stack == "" {
|
|
return false
|
|
}
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
if _, ok := m.opInterrupted[stack]; !ok {
|
|
return false
|
|
}
|
|
delete(m.opInterrupted, stack)
|
|
m.persistRestoreRecordLocked()
|
|
return true
|
|
}
|