v0.246.0: an interrupted restore is told; the recovery-code reminder waits until the box can take it
gates / gates (push) Successful in 15s
gates / gates (push) Successful in 15s
MinAgent: 0.131.0 (unchanged). Requires hub v0.117.0 for restore_interrupted. R-550 (operator ruling: fix). A design reversed and recorded: the restore op-status was in memory by choice. Now restore-status.json in DataDir, written atomically at both ends of an op. At startup a record still marked running becomes a failed, interrupted result kept per app until that app's next restore, shown on /backups/restore and the off-site wizard, and raised once as restore_interrupted. Cooldowns stay in memory. R-546. The R-543 reminder bar consults the agent's own preflight ok (every blocking item, not a copy of pbs_storage_id), cached 60 s, probed only while paused. /backup/escrow shows a waiting card that polls and reloads instead of red crosses and English diagnostics. POST /api/escrow/start refuses 409 before staging or starting - the direct path chaos night used. Unknown readiness keeps the bar. Red-proofs (each seen failing): restore record across restart; main() calls both startup functions; startup helper with loading skipped; restore page card; bar held back; waiting card; start refusal. go build/vet/test ./... green, 28 packages; controller_gates --fast all OK. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -264,6 +264,9 @@ type Manager struct {
|
||||
opStack string
|
||||
opStartedAt time.Time
|
||||
opLast *RestoreOpResult
|
||||
// R-550: persistence of the above (restore_record.go) and the per-app interrupted notices.
|
||||
opRecordPath string
|
||||
opInterrupted map[string]RestoreOpResult
|
||||
|
||||
// Cached status for page rendering (refreshed periodically)
|
||||
cachedStatus *FullBackupStatus
|
||||
|
||||
@@ -6,8 +6,8 @@ import "time"
|
||||
// backups page can show a progress banner (running → success/failure) instead of blocking the HTTP
|
||||
// request until the restore completes. It is display-only and mutex-guarded on the Manager's `mu`;
|
||||
// it does NOT gate concurrency (that stays the restore functions' internal single-flight acquire).
|
||||
// In-memory only — lost on a controller restart (same precedent as notification cooldowns); a page
|
||||
// load mid-op after a restart simply shows no banner.
|
||||
// PERSISTED since v0.246.0 (R-550, operator ruling 2026-09-17 — a reversal of the original in-memory
|
||||
// choice, for the restore record only; notification cooldowns stay in memory). See restore_record.go.
|
||||
|
||||
// RestoreOpResult is the terminal record of the most recent restore op.
|
||||
type RestoreOpResult struct {
|
||||
@@ -16,6 +16,9 @@ type RestoreOpResult struct {
|
||||
OK bool `json:"ok"`
|
||||
Message string `json:"message"`
|
||||
FinishedAt time.Time `json:"finished_at"`
|
||||
// Interrupted marks a restore that was still in flight when the controller stopped (R-550): found
|
||||
// at the next start, recorded as a failure with RestoreInterruptedMessage.
|
||||
Interrupted bool `json:"interrupted,omitempty"`
|
||||
}
|
||||
|
||||
// RestoreResultWindow bounds how long a finished restore still counts as "what just happened".
|
||||
@@ -54,6 +57,9 @@ func (m *Manager) BeginRestoreOp(op, stack string) {
|
||||
m.opName = op
|
||||
m.opStack = stack
|
||||
m.opStartedAt = time.Now()
|
||||
// A new restore of this app supersedes its interrupted notice (R-550).
|
||||
delete(m.opInterrupted, stack)
|
||||
m.persistRestoreRecordLocked()
|
||||
}
|
||||
|
||||
// EndRestoreOp records the terminal result (called from the goroutine on completion, success or
|
||||
@@ -69,6 +75,7 @@ func (m *Manager) EndRestoreOp(ok bool, message string) {
|
||||
FinishedAt: time.Now(),
|
||||
}
|
||||
m.opRunning = false
|
||||
m.persistRestoreRecordLocked()
|
||||
}
|
||||
|
||||
// RestoreStatus returns a deep copy of the current restore op-status for the page/API.
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-550 (operator ruling "fix", 2026-09-17) — the restore record survives a restart.
|
||||
//
|
||||
// A DESIGN REVERSED, AND RECORDED AS SUCH. opstatus.go was deliberately in-memory ("same precedent as
|
||||
// notification cooldowns"). Chaos night round 10 measured the cost: a restore accepted at 23:26:08Z,
|
||||
// the box hard-reset four seconds later, and afterwards the status answered the Go zero value — the
|
||||
// household had pressed a button, been told it started, and could never learn whether it finished.
|
||||
// The operator reversed the choice for the RESTORE RECORD ONLY; notification cooldowns stay in memory.
|
||||
//
|
||||
// Shape: one JSON file in the controller's state directory (DataDir, beside settings.json), written
|
||||
// atomically (atomicWrite: tmp + rename) at BOTH ends of an op. At startup a record still marked
|
||||
// running is, by construction, a restore nothing is running any more: it becomes a terminal failure
|
||||
// (Interrupted, RestoreInterruptedMessage) and a per-app notice that stays until that app's next
|
||||
// restore. LoadRestoreRecord reports the conversion ONCE, so the caller raises restore_interrupted once.
|
||||
//
|
||||
// No path set (tests that build a bare Manager, a box with backup disabled) = the old in-memory
|
||||
// behaviour, silently: persistence is a property of the wired controller, not of every Manager.
|
||||
|
||||
// RestoreInterruptedMessage is what the household reads when the box stopped mid-restore.
|
||||
const RestoreInterruptedMessage = "A visszaállítás megszakadt (a doboz újraindult) — indítsd el újra."
|
||||
|
||||
// restoreRecordFile is the on-disk shape.
|
||||
type restoreRecordFile struct {
|
||||
Running bool `json:"running"`
|
||||
Op string `json:"op,omitempty"`
|
||||
Stack string `json:"stack,omitempty"`
|
||||
StartedAt time.Time `json:"started_at,omitempty"`
|
||||
Last *RestoreOpResult `json:"last,omitempty"`
|
||||
Interrupted map[string]RestoreOpResult `json:"interrupted,omitempty"`
|
||||
}
|
||||
|
||||
// SetRestoreRecordPath wires persistence. Call before LoadRestoreRecord and before serving requests.
|
||||
func (m *Manager) SetRestoreRecordPath(path string) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.opRecordPath = path
|
||||
}
|
||||
|
||||
// LoadRestoreRecord restores the record at startup. It returns the restore that was interrupted by the
|
||||
// stop — non-nil exactly once per interruption — so the caller can log it and raise restore_interrupted.
|
||||
func (m *Manager) LoadRestoreRecord() *RestoreOpResult {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.opRecordPath == "" {
|
||||
return nil
|
||||
}
|
||||
b, err := os.ReadFile(m.opRecordPath)
|
||||
if err != nil {
|
||||
if !os.IsNotExist(err) && m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] restore record unreadable (%v) — starting with no restore history", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
var rec restoreRecordFile
|
||||
if err := json.Unmarshal(b, &rec); err != nil {
|
||||
if m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] restore record corrupt (%v) — starting with no restore history", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
m.opLast = rec.Last
|
||||
m.opInterrupted = rec.Interrupted
|
||||
if !rec.Running || rec.Stack == "" {
|
||||
if m.logger != nil && len(m.opInterrupted) > 0 {
|
||||
m.logger.Printf("[INFO] [backup] restore record loaded: %d interrupted restore notice(s) still shown", len(m.opInterrupted))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
res := RestoreOpResult{
|
||||
Op: rec.Op, Stack: rec.Stack, OK: false,
|
||||
Message: RestoreInterruptedMessage, FinishedAt: time.Now(), Interrupted: true,
|
||||
}
|
||||
m.opLast = &res
|
||||
if m.opInterrupted == nil {
|
||||
m.opInterrupted = map[string]RestoreOpResult{}
|
||||
}
|
||||
m.opInterrupted[rec.Stack] = res
|
||||
m.opRunning = false
|
||||
if m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] restore of %s (%s, started %s) was INTERRUPTED by a controller stop — recorded as failed; the household is told to run it again",
|
||||
rec.Stack, rec.Op, rec.StartedAt.Format(time.RFC3339))
|
||||
}
|
||||
m.persistRestoreRecordLocked()
|
||||
out := res
|
||||
return &out
|
||||
}
|
||||
|
||||
// InterruptedRestore reports the standing interrupted-restore notice for one app, if any.
|
||||
func (m *Manager) InterruptedRestore(stack string) (RestoreOpResult, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
r, ok := m.opInterrupted[stack]
|
||||
return r, ok
|
||||
}
|
||||
|
||||
// InterruptedRestores lists every standing notice, sorted by app, for the restore page.
|
||||
func (m *Manager) InterruptedRestores() []RestoreOpResult {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
out := make([]RestoreOpResult, 0, len(m.opInterrupted))
|
||||
for _, r := range m.opInterrupted {
|
||||
out = append(out, r)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].Stack < out[j].Stack })
|
||||
return out
|
||||
}
|
||||
|
||||
// persistRestoreRecordLocked writes the current op-status. Caller holds m.mu. A failed write is logged
|
||||
// and the in-memory status carries on — the page still works for this process's lifetime.
|
||||
func (m *Manager) persistRestoreRecordLocked() {
|
||||
if m.opRecordPath == "" {
|
||||
return
|
||||
}
|
||||
rec := restoreRecordFile{
|
||||
Running: m.opRunning, Op: m.opName, Stack: m.opStack, StartedAt: m.opStartedAt,
|
||||
Last: m.opLast, Interrupted: m.opInterrupted,
|
||||
}
|
||||
b, err := json.Marshal(rec)
|
||||
if err == nil {
|
||||
err = atomicWrite(m.opRecordPath, b, 0o600)
|
||||
}
|
||||
if err != nil && m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] could not persist the restore record to %s: %v (kept in memory)", m.opRecordPath, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func recordManager(path string) *Manager {
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0)}
|
||||
m.SetRestoreRecordPath(path)
|
||||
return m
|
||||
}
|
||||
|
||||
// R-550 (operator ruling "fix", 2026-09-17). Chaos night round 10: a restore was accepted and the box
|
||||
// was hard-reset four seconds later. Afterwards /api/backup/restore-status answered the Go zero value
|
||||
// and nothing told the household whether the restore had finished — the op-status was in-memory only.
|
||||
//
|
||||
// The consequence asserted: a restore that was IN FLIGHT when the controller stopped is, after the
|
||||
// next start, a FAILED result for that app saying it was interrupted — and loading it reports it once,
|
||||
// so the caller can raise restore_interrupted.
|
||||
//
|
||||
// RED-PROOF: keep the op-status in memory (no record written / read) → the second Manager's status is
|
||||
// blank → "after a restart the interrupted restore left no trace".
|
||||
func TestRestoreRecord_InterruptedRestoreSurvivesRestart(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "restore-status.json")
|
||||
m1 := recordManager(path)
|
||||
m1.BeginRestoreOp("restore", "gokapi")
|
||||
// the box stops here — no EndRestoreOp
|
||||
|
||||
m2 := recordManager(path)
|
||||
got := m2.LoadRestoreRecord()
|
||||
st := m2.RestoreStatus()
|
||||
if got == nil || st.Last == nil {
|
||||
t.Fatalf("after a restart the interrupted restore left no trace: loaded=%v status=%+v", got, st)
|
||||
}
|
||||
if st.Running {
|
||||
t.Fatalf("a restore from before the restart is still shown as RUNNING — nothing is running it: %+v", st)
|
||||
}
|
||||
if st.Last.OK || !st.Last.Interrupted || st.Last.Stack != "gokapi" || st.Last.Op != "restore" {
|
||||
t.Fatalf("interrupted record = %+v, want a failed, interrupted restore of gokapi", st.Last)
|
||||
}
|
||||
// ASCII fragment of the Hungarian sentence, with a negative control.
|
||||
if !strings.Contains(st.Last.Message, "megszakadt") || strings.Contains(st.Last.Message, "zzzz-not-present") {
|
||||
t.Fatalf("message %q does not say the restore was interrupted", st.Last.Message)
|
||||
}
|
||||
if rec, ok := m2.InterruptedRestore("gokapi"); !ok || !rec.Interrupted {
|
||||
t.Fatalf("InterruptedRestore(gokapi) = %+v, %v — the restore page would not show it", rec, ok)
|
||||
}
|
||||
// Reported ONCE: a third start must not raise the event again for the same interruption.
|
||||
m3 := recordManager(path)
|
||||
if again := m3.LoadRestoreRecord(); again != nil {
|
||||
t.Fatalf("the same interruption was reported again on the next start: %+v", again)
|
||||
}
|
||||
if _, ok := m3.InterruptedRestore("gokapi"); !ok {
|
||||
t.Fatalf("the interrupted notice vanished on the next start — it must stay until gokapi is restored again")
|
||||
}
|
||||
}
|
||||
|
||||
// Control: a restore that FINISHED is still that result after a restart — never re-labelled interrupted.
|
||||
func TestRestoreRecord_FinishedRestoreStaysFinished(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "restore-status.json")
|
||||
m1 := recordManager(path)
|
||||
m1.BeginRestoreOp("restore", "mealie")
|
||||
m1.EndRestoreOp(true, "kész")
|
||||
|
||||
m2 := recordManager(path)
|
||||
if got := m2.LoadRestoreRecord(); got != nil {
|
||||
t.Fatalf("a finished restore was reported as interrupted: %+v", got)
|
||||
}
|
||||
st := m2.RestoreStatus()
|
||||
if st.Last == nil || !st.Last.OK || st.Last.Interrupted || st.Last.Message != "kész" {
|
||||
t.Fatalf("finished record after restart = %+v, want ok 'kész'", st.Last)
|
||||
}
|
||||
}
|
||||
|
||||
// The notice lasts until that app's NEXT restore — and only that app's.
|
||||
func TestRestoreRecord_NextRestoreOfThatAppClearsTheNotice(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "restore-status.json")
|
||||
m1 := recordManager(path)
|
||||
m1.BeginRestoreOp("restore", "gokapi")
|
||||
m2 := recordManager(path)
|
||||
m2.LoadRestoreRecord()
|
||||
|
||||
m2.BeginRestoreOp("restore", "mealie") // a different app
|
||||
m2.EndRestoreOp(true, "kész")
|
||||
if _, ok := m2.InterruptedRestore("gokapi"); !ok {
|
||||
t.Fatalf("restoring ANOTHER app cleared gokapi's interrupted notice")
|
||||
}
|
||||
m2.BeginRestoreOp("restore", "gokapi") // the same app, again
|
||||
if _, ok := m2.InterruptedRestore("gokapi"); ok {
|
||||
t.Fatalf("starting gokapi's restore again did not clear its interrupted notice")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user