v0.295.0: a box that was off at its backup time catches up once (R-871, decision 109); the missed-backup banner (decision 110); a late daily timer after a host suspend is skipped
gates / gates (push) Successful in 31s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 09:27:16 +02:00
parent e730a629fd
commit 635c33d381
24 changed files with 2074 additions and 18 deletions
+85 -11
View File
@@ -43,6 +43,7 @@ import (
"gitea.dooplex.hu/admin/felhom-controller/internal/mailrelay"
"gitea.dooplex.hu/admin/felhom-controller/internal/metrics"
"gitea.dooplex.hu/admin/felhom-controller/internal/monitor"
"gitea.dooplex.hu/admin/felhom-controller/internal/nightchain"
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
"gitea.dooplex.hu/admin/felhom-controller/internal/offsiteapply"
"gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
@@ -1110,7 +1111,30 @@ func main() {
// App-data backup: daily database dumps. Disk-tier (restic snapshots,
// cross-drive, integrity check, infra backup) has moved to the host agent.
sched.Daily("db-dump", dbLeg, func(ctx context.Context) error {
// R-871 (v0.295.0, `09` decision 109): the night ledger records when each leg ran to its end, so a box that
// was off at W makes the night up ONCE when it comes back (internal/nightchain). Every leg body is wrapped
// once and shared by the scheduled run and the catch-up — the two can never drift apart.
nightLedger, nerr := nightchain.Open(filepath.Join(cfg.Paths.DataDir, "night-ledger.json"), time.Now())
if nerr != nil {
logger.Printf("[WARN] [catch-up] the night ledger cannot be read — no catch-up on this box until it can: %v", nerr)
}
webNightLedger = nightLedger
withLeg := func(leg nightchain.Leg, fn func(context.Context) error) func(context.Context) error {
return func(ctx context.Context) error {
if nightLedger == nil {
return fn(ctx)
}
nightLedger.LegLock.Lock() // a scheduled leg and a catch-up never run at the same time
defer nightLedger.LegLock.Unlock()
err := fn(ctx)
nightLedger.MarkEnded(leg, time.Now())
if leg == nightchain.LegDBDump && err == nil {
nightLedger.MarkDBDumpOK(time.Now())
}
return err
}
}
dbDumpLeg := withLeg(nightchain.LegDBDump, func(ctx context.Context) error {
err := backupMgr.RunDBDumps(ctx)
if err != nil {
notifier.NotifyDBDumpFailed("Adatbázis mentés sikertelen", err.Error())
@@ -1119,6 +1143,7 @@ func main() {
}
return err
})
sched.Daily("db-dump", dbLeg, dbDumpLeg)
// ── R-218, THE CONSUME HALF (v0.203.0) ────────────────────────────────────────────────
//
@@ -1204,10 +1229,11 @@ func main() {
})
}
})
sched.Daily("tier2-backup", tier2Leg, func(ctx context.Context) error {
tier2LegFn := withLeg(nightchain.LegTier2, func(ctx context.Context) error {
backupMgr.RunAllTier2()
return nil
})
sched.Daily("tier2-backup", tier2Leg, tier2LegFn)
// Off-box (NAS) restic-SFTP backup (Part B): the off-site leg. A failure (incl. a fail-fast dead-NAS
// error) alerts the operator via the allowlisted backup_failed event. Daily after Tier 2.
@@ -1353,16 +1379,49 @@ func main() {
// off-site job — it runs when this function's off-site half has ended, on every path: configured
// or not, ok, failed, any of RunOffboxBackup's early returns, even a panic. A box with no off-site
// target runs it at W+105m. One call site, no signal to go stale.
sched.Daily("offbox-backup", offboxLeg, func(ctx context.Context) error {
return chainUpdateLeg(ctx, func(ctx context.Context) error {
t := sett.GetOffboxTarget()
if t == nil || !t.Enabled || t.Schedule != "daily" || !backupMgr.OffboxConfigured() {
logger.Printf("[INFO] [offbox] no scheduled off-site target on this box — the off-site leg does nothing; the update leg runs now")
return nil // not configured / not scheduled
}
return backupMgr.RunOffboxBackup(ctx)
}, func(ctx context.Context) { stackMgr.RunUpdateLeg(ctx, "after-offsite") })
offsiteOnly := withLeg(nightchain.LegOffsite, func(ctx context.Context) error {
t := sett.GetOffboxTarget()
if t == nil || !t.Enabled || t.Schedule != "daily" || !backupMgr.OffboxConfigured() {
logger.Printf("[INFO] [offbox] no scheduled off-site target on this box — the off-site leg does nothing")
return nil // not configured / not scheduled
}
return backupMgr.RunOffboxBackup(ctx)
})
sched.Daily("offbox-backup", offboxLeg, func(ctx context.Context) error {
return chainUpdateLeg(ctx, offsiteOnly, func(ctx context.Context) { stackMgr.RunUpdateLeg(ctx, "after-offsite") })
})
// R-871: the catch-up — the three BACKUP legs only (no update leg: it restarts apps and waits for a real
// night). Triggered at start and on a host resume; it waits 15 minutes, and it waits for a whole-guest
// backup (and that backup waits for it). Pinned by internal/nightchain + TestR871_CatchUpWiring.
if nightLedger != nil {
catchUp := &nightchain.CatchUp{
Ledger: nightLedger,
Window: func() string {
return backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule)
},
Legs: catchUpLegs(dbDumpLeg, tier2LegFn, offsiteOnly),
QuiesceBusy: func() bool {
return quiesceLoop != nil && len(quiesceLoop.SuppressedStacks()) > 0
},
Event: func(missedAt time.Time) {
notifier.NotifyBackupCatchUp(missedAt.In(budapestLoc()).Format("15:04"))
},
Logger: logger,
Loc: budapestLoc(),
}
if quiesceLoop != nil {
quiesceLoop.SetCatchUpFn(catchUp.Running)
}
catchUp.Evaluate(ctx, "controller start")
resumeTick := time.NewTicker(time.Minute)
go nightchain.ResumeWatch(ctx, resumeTick.C,
func() time.Time { return time.Now().Round(0) },
func() time.Duration { return time.Since(startTime) },
5*time.Minute,
func(slept time.Duration) {
catchUp.Evaluate(ctx, fmt.Sprintf("host resume (suspended ~%s)", slept.Round(time.Minute)))
})
}
// R-241 — the abandonment terminal step. DAILY and not on the backup leg, deliberately: it must
// run on a box whose off-site tier is NOT configured for runs (an abandoning box may be sitting
// with escrow pending), and tying it to the backup leg would make the deletion depend on a
@@ -1815,6 +1874,8 @@ func main() {
// --- Initialize web server ---
webServer := web.NewServer(cfg, stackMgr, cpuCollector, backupMgr, sched, sett, alertMgr, notifier, updater, logger, Version)
// R-871 (decision 110): the missed-backup banner reads the night ledger and the box's own on/off record.
webServer.SetMissedBackupBanner(webNightLedger, metricsStore)
// Migration done-hook: a decommission-initiated migration finalizes the source decommission on
// success (soft-mark + agent). Wire it before RecoverMigration so a resumed one still finalizes.
stackMgr.SetMigrationDoneHook(webServer.OnMigrationDone)
@@ -3910,6 +3971,19 @@ func stopUnhealthyApps(logger *log.Logger, d unhealthyDeps, ooms []stacks.OOMCon
// returned nil (ran, or had nothing to do), returned an error, or panicked. The off-site half's error is
// returned (the scheduler logs it); a panic becomes an error. `09` §6.4.2 point 2; pinned by
// TestChainUpdateLeg_EveryPath.
// catchUpLegs is the catch-up's leg table (R-871): the three backup legs and NOTHING else — no update leg, no
// Docker step. A function so the table is one reviewable place; pinned by TestR871_CatchUpWiring.
func catchUpLegs(db, tier2, offsite func(context.Context) error) map[nightchain.Leg]func(context.Context) error {
return map[nightchain.Leg]func(context.Context) error{
nightchain.LegDBDump: db,
nightchain.LegTier2: tier2,
nightchain.LegOffsite: offsite,
}
}
// webNightLedger hands the ledger to the web server (the missed-backup banner); nil when backups are off.
var webNightLedger *nightchain.Ledger
func chainUpdateLeg(ctx context.Context, offsite func(context.Context) error, leg func(context.Context)) (err error) {
defer func() {
if r := recover(); r != nil {
@@ -0,0 +1,63 @@
package main
import (
"context"
"go/ast"
"os"
"strings"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/nightchain"
)
// TestR871_CatchUpWiring — v0.295.0 (`09` decision 109): the catch-up is built, triggered at start and on a host
// resume, and holds the whole-guest cycle — the seam-built-but-never-wired class (four earlier instances).
// COMPANION RED-PROOF: delete the `catchUp.Evaluate(ctx, "controller start")` line → this fails.
func TestR871_CatchUpWiring(t *testing.T) {
lines, _, _ := slice4CallLines(t)
for _, call := range []string{"ResumeWatch", "SetCatchUpFn", "catchUpLegs", "NotifyBackupCatchUp"} {
if len(lines[call]) == 0 {
t.Errorf("%s is never called in main.go — the catch-up is built but not wired", call)
}
}
// Both triggers, by their reason argument: the start (a literal) and the resume (inside ResumeWatch's callback).
_, f, _ := slice4CallLines(t)
start := 0
ast.Inspect(f, func(n ast.Node) bool {
if call, ok := n.(*ast.CallExpr); ok {
if sel, ok := call.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "Evaluate" && len(call.Args) == 2 {
if lit, ok := call.Args[1].(*ast.BasicLit); ok && lit.Value == `"controller start"` {
start++
}
}
}
return true
})
if start != 1 {
t.Errorf("the catch-up's START trigger (Evaluate(ctx, \"controller start\")) is called %d times, want 1", start)
}
}
// The catch-up's leg table holds the three BACKUP legs and nothing else, and the off-site body it shares with the
// night never calls the update leg (that is chained only around the SCHEDULED off-site job).
// COMPANION RED-PROOF: put `stackMgr.RunUpdateLeg` inside offsiteOnly's body → "the update leg is inside".
func TestR871_CatchUpRunsNoUpdateLeg(t *testing.T) {
noop := func(context.Context) error { return nil }
legs := catchUpLegs(noop, noop, noop)
if len(legs) != 3 || legs[nightchain.LegDBDump] == nil || legs[nightchain.LegTier2] == nil || legs[nightchain.LegOffsite] == nil {
t.Fatalf("catchUpLegs = %v, want exactly db-dump, tier2, offsite", legs)
}
src, err := os.ReadFile("main.go")
if err != nil {
t.Fatal(err)
}
s := string(src)
i := strings.Index(s, "offsiteOnly := withLeg(nightchain.LegOffsite,")
j := strings.Index(s[i:], "sched.Daily(\"offbox-backup\"")
if i < 0 || j < 0 {
t.Fatal("the shared off-site leg body was not found in main.go")
}
if body := s[i : i+j]; strings.Contains(body, "RunUpdateLeg") || strings.Contains(body, "chainUpdateLeg") {
t.Fatal("the update leg is inside the off-site body the catch-up runs")
}
}