v0.246.0: an interrupted restore is told; the recovery-code reminder waits until the box can take it
gates / gates (push) Successful in 15s

MinAgent: 0.131.0 (unchanged). Requires hub v0.117.0 for restore_interrupted.

R-550 (operator ruling: fix). A design reversed and recorded: the restore
op-status was in memory by choice. Now restore-status.json in DataDir, written
atomically at both ends of an op. At startup a record still marked running
becomes a failed, interrupted result kept per app until that app's next
restore, shown on /backups/restore and the off-site wizard, and raised once as
restore_interrupted. Cooldowns stay in memory.

R-546. The R-543 reminder bar consults the agent's own preflight ok (every
blocking item, not a copy of pbs_storage_id), cached 60 s, probed only while
paused. /backup/escrow shows a waiting card that polls and reloads instead of
red crosses and English diagnostics. POST /api/escrow/start refuses 409 before
staging or starting - the direct path chaos night used. Unknown readiness keeps
the bar.

Red-proofs (each seen failing): restore record across restart; main() calls
both startup functions; startup helper with loading skipped; restore page card;
bar held back; waiting card; start refusal. go build/vet/test ./... green, 28
packages; controller_gates --fast all OK.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-17 10:45:29 +02:00
parent 714d5bce09
commit 0fe315b759
21 changed files with 788 additions and 10 deletions
+6
View File
@@ -527,6 +527,8 @@ func main() {
// --- Initialize backup manager (app-data only: DB dumps + Docker-volume tars) ---
var backupMgr *backup.Manager
// R-550: the restore the last stop interrupted, found when the record loads; raised once the notifier exists.
var interruptedRestore *backup.RestoreOpResult
stackProv := &stackAdapter{
mgr: stackMgr,
getStoragePaths: func() []settings.StoragePath { return sett.GetStoragePaths() },
@@ -534,6 +536,8 @@ func main() {
}
if cfg.Backup.Enabled {
backupMgr = backup.NewManager(cfg, sett, logger)
// R-550: load the persisted restore record BEFORE any page can ask for it.
interruptedRestore = loadRestoreRecordAtStartup(backupMgr, cfg.Paths.DataDir, logger)
// R-166: use the guard that already ran Recover at startup, not a second one over the same
// file (see SetAppStopGuard — one file, one owner).
backupMgr.SetAppStopGuard(appStopGuard)
@@ -580,6 +584,8 @@ func main() {
// --- Initialize notifier ---
notifier := notify.New(cfg.Hub.URL, cfg.Hub.APIKey, cfg.Customer.ID, sett, logger, cfg.Logging.Level == "debug")
// R-550: an interrupted restore is raised once, now that the notifier exists.
reportInterruptedRestore(notifier, interruptedRestore)
// R-97a: wire the quiesce loop's hub-event seam. It MUST happen here and not in
// startQuiesceLoop, because the notifier is constructed after it — and it must happen at all,
@@ -0,0 +1,48 @@
package main
import (
"fmt"
"log"
"path/filepath"
"gitea.dooplex.hu/admin/felhom-controller/internal/backup"
)
// R-550 — the restore record survives a restart, and an interruption is raised ONCE.
//
// Two functions so main() makes two plain calls a test can see (TestMainWiresRestoreRecord): the
// record must be loaded BEFORE the web server serves a restore page, and the event can only be pushed
// AFTER the notifier exists, which main() constructs later.
// restoreRecordFileName lives in DataDir, beside settings.json — the SSD state dir that survives a
// container recreation.
const restoreRecordFileName = "restore-status.json"
// restoreEventPusher is the notifier method main() hands in (an interface so a test can record it).
type restoreEventPusher interface {
PushEvent(eventType, severity, message string, details interface{})
}
// loadRestoreRecordAtStartup wires persistence and returns the restore the last stop interrupted.
func loadRestoreRecordAtStartup(mgr *backup.Manager, dataDir string, logger *log.Logger) *backup.RestoreOpResult {
if mgr == nil {
return nil
}
path := filepath.Join(dataDir, restoreRecordFileName)
mgr.SetRestoreRecordPath(path)
res := mgr.LoadRestoreRecord()
if logger != nil {
logger.Printf("[INFO] [backup] restore record wired: %s (interrupted at startup: %t)", path, res != nil)
}
return res
}
// reportInterruptedRestore raises restore_interrupted (warning; for the household — hub v0.117.0).
func reportInterruptedRestore(p restoreEventPusher, res *backup.RestoreOpResult) {
if p == nil || res == nil {
return
}
p.PushEvent("restore_interrupted", "warning",
fmt.Sprintf("A(z) %s visszaállítása megszakadt, mert a doboz újraindult — indítsd el újra.", res.Stack),
map[string]any{"stack": res.Stack, "op": res.Op})
}
@@ -0,0 +1,82 @@
package main
import (
"go/ast"
"go/parser"
"go/token"
"io"
"log"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/backup"
)
type recordedPush struct{ eventType, severity, message string }
type pushRecorder struct{ got []recordedPush }
func (r *pushRecorder) PushEvent(et, sev, msg string, _ interface{}) {
r.got = append(r.got, recordedPush{et, sev, msg})
}
// R-550 — the functions main() calls: a restore left in flight is found at startup and raised as
// restore_interrupted (warning), exactly once.
//
// RED-PROOF: make loadRestoreRecordAtStartup skip LoadRestoreRecord → nothing is pushed →
// "an interrupted restore raised no restore_interrupted".
func TestRestoreRecordAtStartup_RaisesInterruptedOnce(t *testing.T) {
dir := t.TempDir()
lg := log.New(io.Discard, "", 0)
before := new(backup.Manager)
before.SetRestoreRecordPath(dir + "/" + restoreRecordFileName)
before.BeginRestoreOp("restore", "gokapi") // the box stops here
rec := &pushRecorder{}
reportInterruptedRestore(rec, loadRestoreRecordAtStartup(new(backup.Manager), dir, lg))
if len(rec.got) != 1 || rec.got[0].eventType != "restore_interrupted" || rec.got[0].severity != "warning" {
t.Fatalf("an interrupted restore raised no restore_interrupted (warning): %+v", rec.got)
}
again := &pushRecorder{}
reportInterruptedRestore(again, loadRestoreRecordAtStartup(new(backup.Manager), dir, lg))
if len(again.got) != 0 {
t.Fatalf("the same interruption was raised again on the next start: %+v", again.got)
}
}
// The seam must be CALLED from main(): both functions, in that order.
func TestMainWiresRestoreRecord(t *testing.T) {
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, "main.go", nil, 0)
if err != nil {
t.Fatalf("parse main.go: %v", err)
}
pos := map[string]token.Pos{}
for _, d := range f.Decls {
fn, ok := d.(*ast.FuncDecl)
if !ok || fn.Name.Name != "main" {
continue
}
ast.Inspect(fn.Body, func(n ast.Node) bool {
if c, ok := n.(*ast.CallExpr); ok {
if id, ok := c.Fun.(*ast.Ident); ok {
if id.Name == "loadRestoreRecordAtStartup" || id.Name == "reportInterruptedRestore" {
if _, seen := pos[id.Name]; !seen {
pos[id.Name] = c.Pos()
}
}
}
}
return true
})
}
load, okL := pos["loadRestoreRecordAtStartup"]
report, okR := pos["reportInterruptedRestore"]
if !okL || !okR {
t.Fatalf("main() does not call both restore-record functions (load=%t report=%t) — the persistence is an inert seam", okL, okR)
}
if load > report {
t.Fatalf("main() reports before it loads — nothing would ever be reported")
}
}