v0.295.0: a box that was off at its backup time catches up once (R-871, decision 109); the missed-backup banner (decision 110); a late daily timer after a host suspend is skipped
gates / gates (push) Successful in 31s
gates / gates (push) Successful in 31s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -43,6 +43,7 @@ import (
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/mailrelay"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/metrics"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/monitor"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/nightchain"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/offsiteapply"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
|
||||
@@ -1110,7 +1111,30 @@ func main() {
|
||||
|
||||
// App-data backup: daily database dumps. Disk-tier (restic snapshots,
|
||||
// cross-drive, integrity check, infra backup) has moved to the host agent.
|
||||
sched.Daily("db-dump", dbLeg, func(ctx context.Context) error {
|
||||
// R-871 (v0.295.0, `09` decision 109): the night ledger records when each leg ran to its end, so a box that
|
||||
// was off at W makes the night up ONCE when it comes back (internal/nightchain). Every leg body is wrapped
|
||||
// once and shared by the scheduled run and the catch-up — the two can never drift apart.
|
||||
nightLedger, nerr := nightchain.Open(filepath.Join(cfg.Paths.DataDir, "night-ledger.json"), time.Now())
|
||||
if nerr != nil {
|
||||
logger.Printf("[WARN] [catch-up] the night ledger cannot be read — no catch-up on this box until it can: %v", nerr)
|
||||
}
|
||||
webNightLedger = nightLedger
|
||||
withLeg := func(leg nightchain.Leg, fn func(context.Context) error) func(context.Context) error {
|
||||
return func(ctx context.Context) error {
|
||||
if nightLedger == nil {
|
||||
return fn(ctx)
|
||||
}
|
||||
nightLedger.LegLock.Lock() // a scheduled leg and a catch-up never run at the same time
|
||||
defer nightLedger.LegLock.Unlock()
|
||||
err := fn(ctx)
|
||||
nightLedger.MarkEnded(leg, time.Now())
|
||||
if leg == nightchain.LegDBDump && err == nil {
|
||||
nightLedger.MarkDBDumpOK(time.Now())
|
||||
}
|
||||
return err
|
||||
}
|
||||
}
|
||||
dbDumpLeg := withLeg(nightchain.LegDBDump, func(ctx context.Context) error {
|
||||
err := backupMgr.RunDBDumps(ctx)
|
||||
if err != nil {
|
||||
notifier.NotifyDBDumpFailed("Adatbázis mentés sikertelen", err.Error())
|
||||
@@ -1119,6 +1143,7 @@ func main() {
|
||||
}
|
||||
return err
|
||||
})
|
||||
sched.Daily("db-dump", dbLeg, dbDumpLeg)
|
||||
|
||||
// ── R-218, THE CONSUME HALF (v0.203.0) ────────────────────────────────────────────────
|
||||
//
|
||||
@@ -1204,10 +1229,11 @@ func main() {
|
||||
})
|
||||
}
|
||||
})
|
||||
sched.Daily("tier2-backup", tier2Leg, func(ctx context.Context) error {
|
||||
tier2LegFn := withLeg(nightchain.LegTier2, func(ctx context.Context) error {
|
||||
backupMgr.RunAllTier2()
|
||||
return nil
|
||||
})
|
||||
sched.Daily("tier2-backup", tier2Leg, tier2LegFn)
|
||||
|
||||
// Off-box (NAS) restic-SFTP backup (Part B): the off-site leg. A failure (incl. a fail-fast dead-NAS
|
||||
// error) alerts the operator via the allowlisted backup_failed event. Daily after Tier 2.
|
||||
@@ -1353,16 +1379,49 @@ func main() {
|
||||
// off-site job — it runs when this function's off-site half has ended, on every path: configured
|
||||
// or not, ok, failed, any of RunOffboxBackup's early returns, even a panic. A box with no off-site
|
||||
// target runs it at W+105m. One call site, no signal to go stale.
|
||||
sched.Daily("offbox-backup", offboxLeg, func(ctx context.Context) error {
|
||||
return chainUpdateLeg(ctx, func(ctx context.Context) error {
|
||||
t := sett.GetOffboxTarget()
|
||||
if t == nil || !t.Enabled || t.Schedule != "daily" || !backupMgr.OffboxConfigured() {
|
||||
logger.Printf("[INFO] [offbox] no scheduled off-site target on this box — the off-site leg does nothing; the update leg runs now")
|
||||
return nil // not configured / not scheduled
|
||||
}
|
||||
return backupMgr.RunOffboxBackup(ctx)
|
||||
}, func(ctx context.Context) { stackMgr.RunUpdateLeg(ctx, "after-offsite") })
|
||||
offsiteOnly := withLeg(nightchain.LegOffsite, func(ctx context.Context) error {
|
||||
t := sett.GetOffboxTarget()
|
||||
if t == nil || !t.Enabled || t.Schedule != "daily" || !backupMgr.OffboxConfigured() {
|
||||
logger.Printf("[INFO] [offbox] no scheduled off-site target on this box — the off-site leg does nothing")
|
||||
return nil // not configured / not scheduled
|
||||
}
|
||||
return backupMgr.RunOffboxBackup(ctx)
|
||||
})
|
||||
sched.Daily("offbox-backup", offboxLeg, func(ctx context.Context) error {
|
||||
return chainUpdateLeg(ctx, offsiteOnly, func(ctx context.Context) { stackMgr.RunUpdateLeg(ctx, "after-offsite") })
|
||||
})
|
||||
// R-871: the catch-up — the three BACKUP legs only (no update leg: it restarts apps and waits for a real
|
||||
// night). Triggered at start and on a host resume; it waits 15 minutes, and it waits for a whole-guest
|
||||
// backup (and that backup waits for it). Pinned by internal/nightchain + TestR871_CatchUpWiring.
|
||||
if nightLedger != nil {
|
||||
catchUp := &nightchain.CatchUp{
|
||||
Ledger: nightLedger,
|
||||
Window: func() string {
|
||||
return backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule)
|
||||
},
|
||||
Legs: catchUpLegs(dbDumpLeg, tier2LegFn, offsiteOnly),
|
||||
QuiesceBusy: func() bool {
|
||||
return quiesceLoop != nil && len(quiesceLoop.SuppressedStacks()) > 0
|
||||
},
|
||||
Event: func(missedAt time.Time) {
|
||||
notifier.NotifyBackupCatchUp(missedAt.In(budapestLoc()).Format("15:04"))
|
||||
},
|
||||
Logger: logger,
|
||||
Loc: budapestLoc(),
|
||||
}
|
||||
if quiesceLoop != nil {
|
||||
quiesceLoop.SetCatchUpFn(catchUp.Running)
|
||||
}
|
||||
catchUp.Evaluate(ctx, "controller start")
|
||||
resumeTick := time.NewTicker(time.Minute)
|
||||
go nightchain.ResumeWatch(ctx, resumeTick.C,
|
||||
func() time.Time { return time.Now().Round(0) },
|
||||
func() time.Duration { return time.Since(startTime) },
|
||||
5*time.Minute,
|
||||
func(slept time.Duration) {
|
||||
catchUp.Evaluate(ctx, fmt.Sprintf("host resume (suspended ~%s)", slept.Round(time.Minute)))
|
||||
})
|
||||
}
|
||||
// R-241 — the abandonment terminal step. DAILY and not on the backup leg, deliberately: it must
|
||||
// run on a box whose off-site tier is NOT configured for runs (an abandoning box may be sitting
|
||||
// with escrow pending), and tying it to the backup leg would make the deletion depend on a
|
||||
@@ -1815,6 +1874,8 @@ func main() {
|
||||
|
||||
// --- Initialize web server ---
|
||||
webServer := web.NewServer(cfg, stackMgr, cpuCollector, backupMgr, sched, sett, alertMgr, notifier, updater, logger, Version)
|
||||
// R-871 (decision 110): the missed-backup banner reads the night ledger and the box's own on/off record.
|
||||
webServer.SetMissedBackupBanner(webNightLedger, metricsStore)
|
||||
// Migration done-hook: a decommission-initiated migration finalizes the source decommission on
|
||||
// success (soft-mark + agent). Wire it before RecoverMigration so a resumed one still finalizes.
|
||||
stackMgr.SetMigrationDoneHook(webServer.OnMigrationDone)
|
||||
@@ -3910,6 +3971,19 @@ func stopUnhealthyApps(logger *log.Logger, d unhealthyDeps, ooms []stacks.OOMCon
|
||||
// returned nil (ran, or had nothing to do), returned an error, or panicked. The off-site half's error is
|
||||
// returned (the scheduler logs it); a panic becomes an error. `09` §6.4.2 point 2; pinned by
|
||||
// TestChainUpdateLeg_EveryPath.
|
||||
// catchUpLegs is the catch-up's leg table (R-871): the three backup legs and NOTHING else — no update leg, no
|
||||
// Docker step. A function so the table is one reviewable place; pinned by TestR871_CatchUpWiring.
|
||||
func catchUpLegs(db, tier2, offsite func(context.Context) error) map[nightchain.Leg]func(context.Context) error {
|
||||
return map[nightchain.Leg]func(context.Context) error{
|
||||
nightchain.LegDBDump: db,
|
||||
nightchain.LegTier2: tier2,
|
||||
nightchain.LegOffsite: offsite,
|
||||
}
|
||||
}
|
||||
|
||||
// webNightLedger hands the ledger to the web server (the missed-backup banner); nil when backups are off.
|
||||
var webNightLedger *nightchain.Ledger
|
||||
|
||||
func chainUpdateLeg(ctx context.Context, offsite func(context.Context) error, leg func(context.Context)) (err error) {
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"go/ast"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/nightchain"
|
||||
)
|
||||
|
||||
// TestR871_CatchUpWiring — v0.295.0 (`09` decision 109): the catch-up is built, triggered at start and on a host
|
||||
// resume, and holds the whole-guest cycle — the seam-built-but-never-wired class (four earlier instances).
|
||||
// COMPANION RED-PROOF: delete the `catchUp.Evaluate(ctx, "controller start")` line → this fails.
|
||||
func TestR871_CatchUpWiring(t *testing.T) {
|
||||
lines, _, _ := slice4CallLines(t)
|
||||
for _, call := range []string{"ResumeWatch", "SetCatchUpFn", "catchUpLegs", "NotifyBackupCatchUp"} {
|
||||
if len(lines[call]) == 0 {
|
||||
t.Errorf("%s is never called in main.go — the catch-up is built but not wired", call)
|
||||
}
|
||||
}
|
||||
// Both triggers, by their reason argument: the start (a literal) and the resume (inside ResumeWatch's callback).
|
||||
_, f, _ := slice4CallLines(t)
|
||||
start := 0
|
||||
ast.Inspect(f, func(n ast.Node) bool {
|
||||
if call, ok := n.(*ast.CallExpr); ok {
|
||||
if sel, ok := call.Fun.(*ast.SelectorExpr); ok && sel.Sel.Name == "Evaluate" && len(call.Args) == 2 {
|
||||
if lit, ok := call.Args[1].(*ast.BasicLit); ok && lit.Value == `"controller start"` {
|
||||
start++
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
if start != 1 {
|
||||
t.Errorf("the catch-up's START trigger (Evaluate(ctx, \"controller start\")) is called %d times, want 1", start)
|
||||
}
|
||||
}
|
||||
|
||||
// The catch-up's leg table holds the three BACKUP legs and nothing else, and the off-site body it shares with the
|
||||
// night never calls the update leg (that is chained only around the SCHEDULED off-site job).
|
||||
// COMPANION RED-PROOF: put `stackMgr.RunUpdateLeg` inside offsiteOnly's body → "the update leg is inside".
|
||||
func TestR871_CatchUpRunsNoUpdateLeg(t *testing.T) {
|
||||
noop := func(context.Context) error { return nil }
|
||||
legs := catchUpLegs(noop, noop, noop)
|
||||
if len(legs) != 3 || legs[nightchain.LegDBDump] == nil || legs[nightchain.LegTier2] == nil || legs[nightchain.LegOffsite] == nil {
|
||||
t.Fatalf("catchUpLegs = %v, want exactly db-dump, tier2, offsite", legs)
|
||||
}
|
||||
src, err := os.ReadFile("main.go")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s := string(src)
|
||||
i := strings.Index(s, "offsiteOnly := withLeg(nightchain.LegOffsite,")
|
||||
j := strings.Index(s[i:], "sched.Daily(\"offbox-backup\"")
|
||||
if i < 0 || j < 0 {
|
||||
t.Fatal("the shared off-site leg body was not found in main.go")
|
||||
}
|
||||
if body := s[i : i+j]; strings.Contains(body, "RunUpdateLeg") || strings.Contains(body, "chainUpdateLeg") {
|
||||
t.Fatal("the update leg is inside the off-site body the catch-up runs")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user