package backup import ( "encoding/json" "os" "sort" "time" ) // R-550 (operator ruling "fix", 2026-09-17) — the restore record survives a restart. // // A DESIGN REVERSED, AND RECORDED AS SUCH. opstatus.go was deliberately in-memory ("same precedent as // notification cooldowns"). Chaos night round 10 measured the cost: a restore accepted at 23:26:08Z, // the box hard-reset four seconds later, and afterwards the status answered the Go zero value — the // household had pressed a button, been told it started, and could never learn whether it finished. // The operator reversed the choice for the RESTORE RECORD ONLY; notification cooldowns stay in memory. // // Shape: one JSON file in the controller's state directory (DataDir, beside settings.json), written // atomically (atomicWrite: tmp + rename) at BOTH ends of an op. At startup a record still marked // running is, by construction, a restore nothing is running any more: it becomes a terminal failure // (Interrupted, RestoreInterruptedMessage) and a per-app notice that stays until that app's next // restore. LoadRestoreRecord reports the conversion ONCE, so the caller raises restore_interrupted once. // // No path set (tests that build a bare Manager, a box with backup disabled) = the old in-memory // behaviour, silently: persistence is a property of the wired controller, not of every Manager. // RestoreInterruptedMessage is what the household reads when the box stopped mid-restore. const RestoreInterruptedMessage = "A visszaállítás megszakadt (a doboz újraindult) — indítsd el újra." // restoreRecordFile is the on-disk shape. type restoreRecordFile struct { Running bool `json:"running"` Op string `json:"op,omitempty"` Stack string `json:"stack,omitempty"` StartedAt time.Time `json:"started_at,omitempty"` Last *RestoreOpResult `json:"last,omitempty"` Interrupted map[string]RestoreOpResult `json:"interrupted,omitempty"` } // SetRestoreRecordPath wires persistence. Call before LoadRestoreRecord and before serving requests. func (m *Manager) SetRestoreRecordPath(path string) { m.mu.Lock() defer m.mu.Unlock() m.opRecordPath = path } // LoadRestoreRecord restores the record at startup. It returns the restore that was interrupted by the // stop — non-nil exactly once per interruption — so the caller can log it and raise restore_interrupted. func (m *Manager) LoadRestoreRecord() *RestoreOpResult { m.mu.Lock() defer m.mu.Unlock() if m.opRecordPath == "" { return nil } b, err := os.ReadFile(m.opRecordPath) if err != nil { if !os.IsNotExist(err) && m.logger != nil { m.logger.Printf("[WARN] [backup] restore record unreadable (%v) — starting with no restore history", err) } return nil } var rec restoreRecordFile if err := json.Unmarshal(b, &rec); err != nil { if m.logger != nil { m.logger.Printf("[WARN] [backup] restore record corrupt (%v) — starting with no restore history", err) } return nil } m.opLast = rec.Last m.opInterrupted = rec.Interrupted if !rec.Running || rec.Stack == "" { if m.logger != nil && len(m.opInterrupted) > 0 { m.logger.Printf("[INFO] [backup] restore record loaded: %d interrupted restore notice(s) still shown", len(m.opInterrupted)) } return nil } res := RestoreOpResult{ Op: rec.Op, Stack: rec.Stack, OK: false, Message: RestoreInterruptedMessage, FinishedAt: time.Now(), Interrupted: true, } m.opLast = &res if m.opInterrupted == nil { m.opInterrupted = map[string]RestoreOpResult{} } m.opInterrupted[rec.Stack] = res m.opRunning = false if m.logger != nil { m.logger.Printf("[WARN] [backup] restore of %s (%s, started %s) was INTERRUPTED by a controller stop — recorded as failed; the household is told to run it again", rec.Stack, rec.Op, rec.StartedAt.Format(time.RFC3339)) } m.persistRestoreRecordLocked() out := res return &out } // InterruptedRestore reports the standing interrupted-restore notice for one app, if any. func (m *Manager) InterruptedRestore(stack string) (RestoreOpResult, bool) { m.mu.Lock() defer m.mu.Unlock() r, ok := m.opInterrupted[stack] return r, ok } // InterruptedRestores lists every standing notice, sorted by app, for the restore page. func (m *Manager) InterruptedRestores() []RestoreOpResult { m.mu.Lock() defer m.mu.Unlock() out := make([]RestoreOpResult, 0, len(m.opInterrupted)) for _, r := range m.opInterrupted { out = append(out, r) } sort.Slice(out, func(i, j int) bool { return out[i].Stack < out[j].Stack }) return out } // persistRestoreRecordLocked writes the current op-status. Caller holds m.mu. A failed write is logged // and the in-memory status carries on — the page still works for this process's lifetime. func (m *Manager) persistRestoreRecordLocked() { if m.opRecordPath == "" { return } rec := restoreRecordFile{ Running: m.opRunning, Op: m.opName, Stack: m.opStack, StartedAt: m.opStartedAt, Last: m.opLast, Interrupted: m.opInterrupted, } b, err := json.Marshal(rec) if err == nil { err = atomicWrite(m.opRecordPath, b, 0o600) } if err != nil && m.logger != nil { m.logger.Printf("[WARN] [backup] could not persist the restore record to %s: %v (kept in memory)", m.opRecordPath, err) } }