0fe315b759
gates / gates (push) Successful in 15s
MinAgent: 0.131.0 (unchanged). Requires hub v0.117.0 for restore_interrupted. R-550 (operator ruling: fix). A design reversed and recorded: the restore op-status was in memory by choice. Now restore-status.json in DataDir, written atomically at both ends of an op. At startup a record still marked running becomes a failed, interrupted result kept per app until that app's next restore, shown on /backups/restore and the off-site wizard, and raised once as restore_interrupted. Cooldowns stay in memory. R-546. The R-543 reminder bar consults the agent's own preflight ok (every blocking item, not a copy of pbs_storage_id), cached 60 s, probed only while paused. /backup/escrow shows a waiting card that polls and reloads instead of red crosses and English diagnostics. POST /api/escrow/start refuses 409 before staging or starting - the direct path chaos night used. Unknown readiness keeps the bar. Red-proofs (each seen failing): restore record across restart; main() calls both startup functions; startup helper with loading skipped; restore page card; bar held back; waiting card; start refusal. go build/vet/test ./... green, 28 packages; controller_gates --fast all OK. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
105 lines
4.5 KiB
Go
105 lines
4.5 KiB
Go
package backup
|
|
|
|
import "time"
|
|
|
|
// Restore op-status (Part B): a lightweight, in-memory surface for the ASYNC restore family so the
|
|
// backups page can show a progress banner (running → success/failure) instead of blocking the HTTP
|
|
// request until the restore completes. It is display-only and mutex-guarded on the Manager's `mu`;
|
|
// it does NOT gate concurrency (that stays the restore functions' internal single-flight acquire).
|
|
// PERSISTED since v0.246.0 (R-550, operator ruling 2026-09-17 — a reversal of the original in-memory
|
|
// choice, for the restore record only; notification cooldowns stay in memory). See restore_record.go.
|
|
|
|
// RestoreOpResult is the terminal record of the most recent restore op.
|
|
type RestoreOpResult struct {
|
|
Op string `json:"op"` // "restore" | "tier2-restore" | "offbox-restore"
|
|
Stack string `json:"stack"`
|
|
OK bool `json:"ok"`
|
|
Message string `json:"message"`
|
|
FinishedAt time.Time `json:"finished_at"`
|
|
// Interrupted marks a restore that was still in flight when the controller stopped (R-550): found
|
|
// at the next start, recorded as a failure with RestoreInterruptedMessage.
|
|
Interrupted bool `json:"interrupted,omitempty"`
|
|
}
|
|
|
|
// RestoreResultWindow bounds how long a finished restore still counts as "what just happened".
|
|
//
|
|
// ONE EXPRESSION, TWO SURFACES (R-351). The wizard's phase strip and the list page's banner both
|
|
// need this bound, and until now only the wizard had one — it lived in internal/web as an
|
|
// unexported constant. A second copy in the JS would be exactly the "two copies that already
|
|
// differed" shape this repo keeps paying for, so the window is defined HERE, beside the status it
|
|
// bounds, and both surfaces read it from the payload.
|
|
const RestoreResultWindow = 10 * time.Minute
|
|
|
|
// RestoreOpStatus is the shape served at GET /api/backup/restore-status.
|
|
type RestoreOpStatus struct {
|
|
Running bool `json:"running"`
|
|
Op string `json:"op,omitempty"`
|
|
Stack string `json:"stack,omitempty"`
|
|
StartedAt time.Time `json:"started_at,omitempty"`
|
|
Last *RestoreOpResult `json:"last,omitempty"`
|
|
// LastRecent reports whether Last finished recently enough to still be worth showing to someone
|
|
// who was NOT watching when it happened.
|
|
//
|
|
// R-351, the measured defect: the banner's JS gated the terminal result on a page-local
|
|
// `sawRunning` flag, so a restore that finished before the page was opened — or in under one
|
|
// poll interval — was shown to nobody. The 2026-08-21 OpenGist restore finished in 8.7s and the
|
|
// operator could not tell from any screen whether it had completed; the answer existed only in
|
|
// a container log. A result nobody can see is the same defect class as no result at all.
|
|
LastRecent bool `json:"last_recent,omitempty"`
|
|
}
|
|
|
|
// BeginRestoreOp marks a restore op in flight (called by the handler just before launching the
|
|
// background goroutine). Idempotent enough for display; concurrency is enforced elsewhere.
|
|
func (m *Manager) BeginRestoreOp(op, stack string) {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
m.opRunning = true
|
|
m.opName = op
|
|
m.opStack = stack
|
|
m.opStartedAt = time.Now()
|
|
// A new restore of this app supersedes its interrupted notice (R-550).
|
|
delete(m.opInterrupted, stack)
|
|
m.persistRestoreRecordLocked()
|
|
}
|
|
|
|
// EndRestoreOp records the terminal result (called from the goroutine on completion, success or
|
|
// failure). Message carries the error (failure) or a human note like the scratch path (offbox).
|
|
func (m *Manager) EndRestoreOp(ok bool, message string) {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
m.opLast = &RestoreOpResult{
|
|
Op: m.opName,
|
|
Stack: m.opStack,
|
|
OK: ok,
|
|
Message: message,
|
|
FinishedAt: time.Now(),
|
|
}
|
|
m.opRunning = false
|
|
m.persistRestoreRecordLocked()
|
|
}
|
|
|
|
// RestoreStatus returns a deep copy of the current restore op-status for the page/API.
|
|
func (m *Manager) RestoreStatus() RestoreOpStatus {
|
|
m.mu.Lock()
|
|
defer m.mu.Unlock()
|
|
st := RestoreOpStatus{
|
|
Running: m.opRunning,
|
|
Op: m.opName,
|
|
Stack: m.opStack,
|
|
StartedAt: m.opStartedAt,
|
|
}
|
|
if m.opLast != nil {
|
|
cp := *m.opLast
|
|
st.Last = &cp
|
|
// Recency is decided here, on the clock the result was stamped with, so neither surface has
|
|
// to hold its own copy of the window. A zero FinishedAt is never recent — an unstamped
|
|
// result must not be drawn as "just now".
|
|
if !cp.FinishedAt.IsZero() {
|
|
if d := time.Since(cp.FinishedAt); d >= 0 && d < RestoreResultWindow {
|
|
st.LastRecent = true
|
|
}
|
|
}
|
|
}
|
|
return st
|
|
}
|