R-379/R-380: put the customer's undo copy back when a database restore fails
gates / gates (push) Successful in 11s
gates / gates (push) Successful in 11s
R-379 and R-380 were one failure. Both ended with a half-restored database; the only difference was whether it looked broken. Postgres emptied and crash-looped; MariaDB applied part of the dump and reported health=healthy with a zero-row schema-version table. Measured live on demo-hp 2026-08-22. The undo copy was already taken and already good - proven by hand that day on both engines. Nothing in the product could apply it. Now it does, with the same ImportDump call, before any restart and inside the DB-only window. The WHOLE undo set, matched on this run's stamp. writeSafetyDump returned one path for an app with two databases; a rollback on that would restore one and leave the other half-written. When the rollback also fails the app is HELD STOPPED (operator ruling): a running app on a half-written database lets the customer make the damage permanent. Every start path refuses it - customer button, appstop Recover, boot sweep - via the shared driveStartGate, checked ABOVE its driveless early return because these apps have no drive. The marker is ended so nothing auto-restarts it. The row goes red. Cleared with --clear-restore-hold, an operator CLI route. --single-transaction is a belt on Postgres only; MariaDB DDL is not transactional and that is why the rollback is the fix. R-381: the engine's stderr stops reaching the customer (615 bytes on MariaDB, its middle rows out of their own database) and starts reaching the operator log, which never had it. R-382: the summary log prints the volume count it already held. Undo copies resolve to their own app, are marked IsUndo, and are capped at 3 per app, pruned from the capture side. The reported render-as-an-app symptom did NOT reproduce - the live page was read first and had zero occurrences. Tests 1468 -> 1483. Eight red-proofs; ONE PASSED and is reported: the R-381 behavioural test injected below ImportDump. A guard at that layer now convicts.
This commit is contained in:
@@ -84,6 +84,8 @@ func main() {
|
||||
abandonStatus := flag.Bool("abandon-status", false, "R-241: print the state of this box's off-site abandonment countdown (set-aside path, due date, days left) and exit. Read-only.")
|
||||
abandonExtend := flag.Int("abandon-extend", 0, "R-241 (operator): extend a running abandonment countdown by N days from now, then exit. Refuses when no countdown is running.")
|
||||
abandonStop := flag.Bool("abandon-stop", false, "R-241 (operator): stop a running abandonment countdown, then exit. The set-aside history is kept and nothing is deleted. Refuses when no countdown is running.")
|
||||
listRestoreHolds := flag.Bool("restore-holds", false, "R-379 (operator): list apps this controller is deliberately holding stopped after a failed database restore whose rollback also failed, then exit. Read-only.")
|
||||
clearRestoreHold := flag.String("clear-restore-hold", "", "R-379 (operator): clear the restore hold on APP so it can start again, then exit. Clearing does NOT repair the database — check the app's data first; the undo copies are in its unit's db-dumps dir.")
|
||||
flag.Parse()
|
||||
|
||||
if *showVersion {
|
||||
@@ -160,6 +162,48 @@ func main() {
|
||||
|
||||
// §7.5 (R-241) — the operator's abandonment levers. Grouped in one block, before the server
|
||||
// starts, exactly like the other CLI subcommands: each loads config + settings, acts, and exits.
|
||||
// R-379 (operator): the way OUT of a hold. A CLI subcommand rather than an API route, matching
|
||||
// --abandon-stop above: the controller's HTTP surface is authenticated as the CUSTOMER, and
|
||||
// clearing a hold is a decision that should follow an operator looking at the database. Reaching
|
||||
// this needs `docker exec` into the guest, which is the operator gate this project already uses.
|
||||
if *listRestoreHolds || *clearRestoreHold != "" {
|
||||
cfg, err := config.LoadPermissive(*configPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "restore-hold: loading config: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
lg := log.New(os.Stderr, "", 0)
|
||||
sett, err := settings.Load(cfg.Paths.DataDir+"/settings.json", lg)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "restore-hold: loading settings: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
if *listRestoreHolds {
|
||||
holds := sett.ListRestoreHolds()
|
||||
if len(holds) == 0 {
|
||||
fmt.Println("no restore holds in force")
|
||||
os.Exit(0)
|
||||
}
|
||||
for _, h := range holds {
|
||||
fmt.Printf("%s\theld since %s\n replay error : %s\n rollback err : %s\n", h.Stack, h.At, h.ReplayError, h.RollbackErr)
|
||||
}
|
||||
os.Exit(0)
|
||||
}
|
||||
cleared, err := sett.ClearRestoreHold(*clearRestoreHold)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "restore-hold: clearing %s: %v\n", *clearRestoreHold, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
if !cleared {
|
||||
// "there was nothing to clear" is not success — a silent 0 here would let an operator
|
||||
// believe they had unblocked an app whose name they mistyped.
|
||||
fmt.Fprintf(os.Stderr, "no restore hold in force for %q — nothing was changed\n", *clearRestoreHold)
|
||||
os.Exit(2)
|
||||
}
|
||||
fmt.Printf("restore hold cleared for %s — the app may be started again. Check its data first: the undo copies are in its unit's db-dumps dir.\n", *clearRestoreHold)
|
||||
os.Exit(0)
|
||||
}
|
||||
|
||||
if *abandonStatus || *abandonExtend > 0 || *abandonStop {
|
||||
cfg, err := config.LoadPermissive(*configPath)
|
||||
if err != nil {
|
||||
@@ -981,6 +1025,31 @@ func main() {
|
||||
}
|
||||
notifier.NotifyBackupRunFailures(rs.Message, d)
|
||||
})
|
||||
// R-379/R-380: an app HELD after a failed database restore whose rollback also failed.
|
||||
//
|
||||
// ROUTED THROUGH `backup_run_failures`, and the reuse is deliberate but imperfect — say so
|
||||
// rather than let a future reader assume it was the natural fit. That type is operator-only
|
||||
// (`notify.operatorOnlyEvents`), carries severity `error` (a VALID token — `warn` would
|
||||
// coerce to `info` and email nobody, which is R-329 and still live elsewhere), and its
|
||||
// payload is exactly {app, leg, reason}. What it is NOT is a restore event: it says "backup
|
||||
// run". **No operator-facing RESTORE event type exists**, and minting one is a two-repo
|
||||
// change with a hub deploy, which this task's baselines hold fixed. The leg is named
|
||||
// `restore-hold` so the record is unambiguous to whoever reads it.
|
||||
//
|
||||
// DELIBERATELY NOT `backup_failed`: customer-enabled by default, Hungarian "A biztonsági
|
||||
// mentés sikertelen!", and this is a deliberate hold rather than a backup failure. R-171 one
|
||||
// path over.
|
||||
backupMgr.SetRestoreHoldNotify(func(stack string, replayErr, rollbackErr error) {
|
||||
d := notify.BackupRunFailuresDetails{RunKind: "offsite-reconstitute", Failed: 1, Attempted: 1}
|
||||
reason := "rollback failed"
|
||||
if rollbackErr != nil {
|
||||
reason = rollbackErr.Error()
|
||||
}
|
||||
d.Apps = append(d.Apps, notify.RunFailureDetail{App: stack, Leg: "restore-hold", Reason: reason})
|
||||
msg := fmt.Sprintf("App %q is HELD STOPPED: its off-site database restore failed AND the rollback to the customer's own pre-restore copy also failed. "+
|
||||
"The app will not start from any path until the hold is cleared. Replay error: %v. Rollback error: %v", stack, replayErr, rollbackErr)
|
||||
notifier.NotifyBackupRunFailures(msg, d)
|
||||
})
|
||||
// 3a: the pre-push enlargement gate blocked an app's userdata push (config+DB still saved). Edge-
|
||||
// triggered by the engine (only NEW blocks notify), so the hub's per-event-type cooldown suffices —
|
||||
// no controller-side timer (the hub owns cooldown).
|
||||
@@ -1774,6 +1843,19 @@ type driveStartGate struct {
|
||||
}
|
||||
|
||||
func (g driveStartGate) MayStart(stackName string) (bool, string) {
|
||||
// R-379/R-380 HOLDER, checked FIRST and deliberately ABOVE the `HDD_PATH == ""` early return
|
||||
// below. The apps this hold exists for are DRIVELESS — a failed database restore is the case,
|
||||
// and 40 of the 53 catalogue apps have no drive at all — so a hold placed after that return
|
||||
// would never be consulted for the exact class it was built for.
|
||||
//
|
||||
// This is holder #4, and putting it here rather than in bootDriveGate is what gives it two
|
||||
// callers with one implementation: the boot sweep reaches it through bootDriveGate, and the
|
||||
// app-stop guard's Recover reaches it directly. A hold only one path honours is not a hold.
|
||||
if g.sett != nil {
|
||||
if h, ok := g.sett.GetRestoreHold(stackName); ok {
|
||||
return false, "held after a failed restore whose rollback also failed (" + h.At + ") — clear the hold to start it"
|
||||
}
|
||||
}
|
||||
if g.mgr == nil {
|
||||
return false, "no stack manager wired — the drive cannot be determined"
|
||||
}
|
||||
@@ -2444,7 +2526,9 @@ func (a *exportAdapter) GetImportRoot() string { return a.mgr.GetImportRoot() }
|
||||
// GetStackNamespaceRoot (R-203) — the felhom-data namespace root, NOT the drive path. The appbackup
|
||||
// path helpers all take this; passing HDD_PATH straight in is what made the export plan and the
|
||||
// off-site capture set describe different directories on the system-data fallback.
|
||||
func (a *exportAdapter) GetStackNamespaceRoot(name string) string { return a.mgr.StackNamespaceRoot(name) }
|
||||
func (a *exportAdapter) GetStackNamespaceRoot(name string) string {
|
||||
return a.mgr.StackNamespaceRoot(name)
|
||||
}
|
||||
|
||||
func (a *exportAdapter) GetStackHDDPath(name string) string {
|
||||
s, ok := a.mgr.GetStack(name)
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-379 SCENARIO D — a held app, later. A hold that only one start path honours is not a hold, so
|
||||
// each path is proven rather than asserted.
|
||||
|
||||
func holdTestSettings(t *testing.T) *settings.Settings {
|
||||
t.Helper()
|
||||
s, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// The SHARED gate refuses a held app — this is the path both the boot sweep (via bootDriveGate) and
|
||||
// the app-stop guard's Recover() reach.
|
||||
//
|
||||
// POSITIVE CONTROL BUILT IN: the same app, same gate, is allowed once the hold is cleared. Without
|
||||
// it "the gate refuses" could be true because the gate refuses everything.
|
||||
func TestR379_DriveStartGate_RefusesAHeldApp_AndAllowsItAfterClearing(t *testing.T) {
|
||||
sett := holdTestSettings(t)
|
||||
g := driveStartGate{sett: sett} // nil mgr on purpose: the hold must be answered BEFORE the drive
|
||||
|
||||
// Before: no hold. The nil manager makes this "cannot determine", which is NOT the hold's reason.
|
||||
if _, why := g.MayStart("docmost"); why == "" {
|
||||
t.Fatal("fixture: expected some reason from the bare gate")
|
||||
} else if strings.Contains(why, "held after a failed restore") {
|
||||
t.Fatalf("no hold exists yet, but the gate cited one: %q", why)
|
||||
}
|
||||
|
||||
if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "docmost", At: "2026-08-22T14:00:00Z"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ok, why := g.MayStart("docmost")
|
||||
if ok {
|
||||
t.Fatal("a held app must not be startable")
|
||||
}
|
||||
if !strings.Contains(why, "held after a failed restore") {
|
||||
t.Errorf("the refusal must NAME the hold, got %q", why)
|
||||
}
|
||||
|
||||
// POSITIVE CONTROL: clear it, and the gate stops citing the hold.
|
||||
cleared, err := sett.ClearRestoreHold("docmost")
|
||||
if err != nil || !cleared {
|
||||
t.Fatalf("clearing the hold failed: cleared=%v err=%v", cleared, err)
|
||||
}
|
||||
if _, why := g.MayStart("docmost"); strings.Contains(why, "held after a failed restore") {
|
||||
t.Errorf("the hold was cleared but the gate still cites it: %q", why)
|
||||
}
|
||||
}
|
||||
|
||||
// The hold is checked ABOVE the `HDD_PATH == ""` early return. The apps this exists for are
|
||||
// DRIVELESS — a failed database restore is the case, and 40 of 53 catalogue apps have no drive — so
|
||||
// a hold placed after that return would never be consulted for the exact class it was built for.
|
||||
func TestR379_HoldIsCheckedBeforeTheDrivelessEarlyReturn(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var body *ast.BlockStmt
|
||||
for _, d := range f.Decls {
|
||||
fn, ok := d.(*ast.FuncDecl)
|
||||
if ok && fn.Recv != nil && fn.Name.Name == "MayStart" && fn.Body != nil {
|
||||
// driveStartGate is the receiver we want; bootDriveGate's MayStart is separate.
|
||||
if st, ok := fn.Recv.List[0].Type.(*ast.Ident); ok && st.Name == "driveStartGate" {
|
||||
body = fn.Body
|
||||
}
|
||||
}
|
||||
}
|
||||
if body == nil {
|
||||
t.Fatal("driveStartGate.MayStart not found in main.go")
|
||||
}
|
||||
holdPos, drivelessPos := -1, -1
|
||||
ast.Inspect(body, func(n ast.Node) bool {
|
||||
if c, ok := n.(*ast.CallExpr); ok {
|
||||
if sel, ok := c.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "GetRestoreHold" && holdPos < 0 {
|
||||
holdPos = fset.Position(c.Pos()).Line
|
||||
}
|
||||
}
|
||||
// the `hdd == ""` driveless early return
|
||||
if b, ok := n.(*ast.BinaryExpr); ok && b.Op == token.EQL && drivelessPos < 0 {
|
||||
if id, ok := b.X.(*ast.Ident); ok && id.Name == "hdd" {
|
||||
drivelessPos = fset.Position(b.Pos()).Line
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
if holdPos < 0 {
|
||||
t.Fatal("driveStartGate.MayStart never consults GetRestoreHold — a held app would start from the boot sweep and from Recover()")
|
||||
}
|
||||
if drivelessPos < 0 {
|
||||
t.Fatal("the driveless early return was not found — this test can no longer prove the ordering")
|
||||
}
|
||||
if holdPos > drivelessPos {
|
||||
t.Fatalf("the hold check (line %d) is AFTER the driveless early return (line %d) — it would never fire for the 40 driveless apps this exists for", holdPos, drivelessPos)
|
||||
}
|
||||
}
|
||||
|
||||
// The production wiring: main() must hand the backup manager its hold notifier, or a held app is
|
||||
// held silently and no operator ever hears. AST, not strings.Contains — a commented-out call still
|
||||
// contains the string.
|
||||
func TestR379_MainWiresTheRestoreHoldNotifier(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "main.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var found bool
|
||||
for _, d := range f.Decls {
|
||||
fn, ok := d.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "main" || fn.Body == nil {
|
||||
continue
|
||||
}
|
||||
ast.Inspect(fn.Body, func(n ast.Node) bool {
|
||||
c, ok := n.(*ast.CallExpr)
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
if sel, ok := c.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "SetRestoreHoldNotify" {
|
||||
found = true
|
||||
return false
|
||||
}
|
||||
return true
|
||||
})
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("main() never calls SetRestoreHoldNotify — an app would be held with nobody told")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user