v0.295.0: a box that was off at its backup time catches up once (R-871, decision 109); the missed-backup banner (decision 110); a late daily timer after a host suspend is skipped
gates / gates (push) Successful in 31s
gates / gates (push) Successful in 31s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,365 @@
|
||||
// Package nightchain makes a missed night's backups up once, when the box comes back (R-871, `09` §3 decision 109,
|
||||
// design `07` §6.1.1).
|
||||
//
|
||||
// THE PROBLEM, MEASURED (2026-10-05 Part F spike, Tester 2 — a laptop switched off at night): the controller's
|
||||
// daily jobs always schedule the NEXT future time (scheduler.nextDailyRun), and nothing remembers that a night was
|
||||
// missed. A box that is off at its window W never gets its database dumps, its second copy or its off-site copy —
|
||||
// for ever, with no alarm.
|
||||
//
|
||||
// THE MECHANISM:
|
||||
// - A LEDGER (`night-ledger.json` in the data dir) records when each of the three nightly backup legs last RAN TO
|
||||
// ITS END. It is an attempt record and is used for ONE question only — "was this leg's last scheduled time
|
||||
// missed?" — never as evidence that a backup exists (that is LastSuccess's job, R-100). A leg that ran and
|
||||
// failed was NOT missed: failures have their own alarms.
|
||||
// - On a controller START and on a host RESUME (a suspended laptop: Go timers run on CLOCK_MONOTONIC, which does
|
||||
// not count suspended time), Evaluate asks the ledger which legs missed their last scheduled time. If any did,
|
||||
// ONE catch-up is scheduled 15 minutes later (apps settle; a box switched on and off again at once does nothing).
|
||||
// - The catch-up runs ONLY the backup legs, in their normal order (database dump, second copy, off-site copy).
|
||||
// It never runs app updates or Docker steps — they restart apps and wait for a real night.
|
||||
// - Several missed nights are ONE catch-up (the question is about the LAST scheduled time only). A normal night
|
||||
// followed by a daytime restart is none. A power cut in the middle of the chain makes up only the legs that did
|
||||
// not end.
|
||||
// - A leg whose own scheduled time is less than 30 minutes away is left to its normal run.
|
||||
// - Every leg — scheduled or catch-up — holds one mutex (`Ledger.LegLock`), so a catch-up and a scheduled leg
|
||||
// never run at the same time. The whole-guest backup (agent-side, quiesce) waits while a catch-up runs
|
||||
// (quiesce.Options.CatchUpFn) and the catch-up waits while a quiesce holds the apps (QuiesceBusy).
|
||||
//
|
||||
// A ledger that does not exist yet (the first start on this release) is SEEDED at that moment and nothing before
|
||||
// it counts as missed — an upgrade must not set off a surprise catch-up on every box.
|
||||
//
|
||||
// Pinned by nightchain_test.go (TestCatchUp_*), each rule with a red-proof.
|
||||
package nightchain
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/backupwindow"
|
||||
)
|
||||
|
||||
// Leg is one nightly backup leg.
|
||||
type Leg string
|
||||
|
||||
const (
|
||||
LegDBDump Leg = "db-dump"
|
||||
LegTier2 Leg = "tier2"
|
||||
LegOffsite Leg = "offsite"
|
||||
)
|
||||
|
||||
// Order is the night chain's own order (07 §6.1).
|
||||
var Order = []Leg{LegDBDump, LegTier2, LegOffsite}
|
||||
|
||||
const (
|
||||
// DefaultDelay: the catch-up waits this long after the trigger (decision 109's "soon after it is switched on").
|
||||
DefaultDelay = 15 * time.Minute
|
||||
// leaveToNormal: a leg whose own scheduled time is closer than this runs normally, not in the catch-up.
|
||||
leaveToNormal = 30 * time.Minute
|
||||
// quiesceWaitMax: how long a catch-up waits for a whole-guest backup that holds the apps.
|
||||
quiesceWaitMax = 2 * time.Hour
|
||||
)
|
||||
|
||||
type ledgerData struct {
|
||||
SeededAt time.Time `json:"seeded_at"`
|
||||
Ended map[Leg]time.Time `json:"ended"` // ATTEMPT record: the leg ran to its end (success or failure)
|
||||
// DBDumpOK is the last SUCCESSFUL database dump (the banner's evidence; the off-site tier has its own
|
||||
// LastSuccess in settings). Written only on success (R-100's rule).
|
||||
DBDumpOK time.Time `json:"db_dump_ok,omitempty"`
|
||||
LastCatchUp time.Time `json:"last_catch_up,omitempty"`
|
||||
BannerDismissed time.Time `json:"banner_dismissed_through,omitempty"` // the missed W instant the household closed
|
||||
}
|
||||
|
||||
// Ledger is the persisted night record.
|
||||
type Ledger struct {
|
||||
mu sync.Mutex
|
||||
path string
|
||||
d ledgerData
|
||||
// LegLock is held by every leg run, scheduled or catch-up.
|
||||
LegLock sync.Mutex
|
||||
}
|
||||
|
||||
// Open reads the ledger, or seeds a new one at now.
|
||||
func Open(path string, now time.Time) (*Ledger, error) {
|
||||
l := &Ledger{path: path, d: ledgerData{Ended: map[Leg]time.Time{}}}
|
||||
b, err := os.ReadFile(path)
|
||||
switch {
|
||||
case err == nil:
|
||||
if jerr := json.Unmarshal(b, &l.d); jerr != nil {
|
||||
return nil, fmt.Errorf("nightchain: ledger unreadable: %w", jerr)
|
||||
}
|
||||
if l.d.Ended == nil {
|
||||
l.d.Ended = map[Leg]time.Time{}
|
||||
}
|
||||
case os.IsNotExist(err):
|
||||
l.d.SeededAt = now.UTC()
|
||||
if serr := l.saveLocked(); serr != nil {
|
||||
return nil, serr
|
||||
}
|
||||
default:
|
||||
return nil, fmt.Errorf("nightchain: ledger: %w", err)
|
||||
}
|
||||
return l, nil
|
||||
}
|
||||
|
||||
func (l *Ledger) saveLocked() error {
|
||||
b, _ := json.MarshalIndent(l.d, "", " ")
|
||||
if err := os.MkdirAll(filepath.Dir(l.path), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
tmp := l.path + ".tmp"
|
||||
if err := os.WriteFile(tmp, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp, l.path)
|
||||
}
|
||||
|
||||
// MarkEnded records that a leg ran to its end (success or failure).
|
||||
func (l *Ledger) MarkEnded(leg Leg, at time.Time) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
l.d.Ended[leg] = at.UTC()
|
||||
_ = l.saveLocked()
|
||||
}
|
||||
|
||||
// MarkDBDumpOK records a successful database dump.
|
||||
func (l *Ledger) MarkDBDumpOK(at time.Time) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
l.d.DBDumpOK = at.UTC()
|
||||
_ = l.saveLocked()
|
||||
}
|
||||
|
||||
func (l *Ledger) markCatchUp(at time.Time) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
l.d.LastCatchUp = at.UTC()
|
||||
_ = l.saveLocked()
|
||||
}
|
||||
|
||||
// DismissBanner records that the household closed the banner for every miss up to `through`.
|
||||
func (l *Ledger) DismissBanner(through time.Time) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
l.d.BannerDismissed = through.UTC()
|
||||
_ = l.saveLocked()
|
||||
}
|
||||
|
||||
// Snapshot returns a copy of the record (for the banner and the debug page).
|
||||
func (l *Ledger) Snapshot() (seeded, dbDumpOK, lastCatchUp, dismissed time.Time, ended map[Leg]time.Time) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
ended = map[Leg]time.Time{}
|
||||
for k, v := range l.d.Ended {
|
||||
ended[k] = v
|
||||
}
|
||||
return l.d.SeededAt, l.d.DBDumpOK, l.d.LastCatchUp, l.d.BannerDismissed, ended
|
||||
}
|
||||
|
||||
// legOffsets: each leg's time relative to W — taken from backupwindow.LegTimes so the two can never disagree.
|
||||
func legTimes(window string) map[Leg]string {
|
||||
db, t2, off := backupwindow.LegTimes(window)
|
||||
return map[Leg]string{LegDBDump: db, LegTier2: t2, LegOffsite: off}
|
||||
}
|
||||
|
||||
// LastInstant is the most recent occurrence of the Budapest clock time hhmm at or before now.
|
||||
func LastInstant(now time.Time, hhmm string, loc *time.Location) (time.Time, bool) {
|
||||
min, err := backupwindow.ParseHHMM(hhmm)
|
||||
if err != nil {
|
||||
return time.Time{}, false
|
||||
}
|
||||
n := now.In(loc)
|
||||
t := time.Date(n.Year(), n.Month(), n.Day(), min/60, min%60, 0, 0, loc)
|
||||
if t.After(n) {
|
||||
t = time.Date(n.Year(), n.Month(), n.Day()-1, min/60, min%60, 0, 0, loc)
|
||||
}
|
||||
return t, true
|
||||
}
|
||||
|
||||
// NextInstant is the next occurrence of hhmm strictly after now.
|
||||
func NextInstant(now time.Time, hhmm string, loc *time.Location) (time.Time, bool) {
|
||||
last, ok := LastInstant(now, hhmm, loc)
|
||||
if !ok {
|
||||
return time.Time{}, false
|
||||
}
|
||||
return time.Date(last.Year(), last.Month(), last.Day()+1, last.Hour(), last.Minute(), 0, 0, loc), true
|
||||
}
|
||||
|
||||
// Missed is the PURE rule: the legs (in chain order) whose last scheduled time passed after the ledger was seeded
|
||||
// and that have not run to their end since — minus a leg whose next scheduled time is under 30 minutes away.
|
||||
// lastW is the most recent missed leg instant (the banner's "the box was off at …").
|
||||
func (l *Ledger) Missed(now time.Time, window string, loc *time.Location) (legs []Leg, lastW time.Time) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
lt := legTimes(window)
|
||||
for _, leg := range Order {
|
||||
inst, ok := LastInstant(now, lt[leg], loc)
|
||||
if !ok || !inst.After(l.d.SeededAt) {
|
||||
continue
|
||||
}
|
||||
if !l.d.Ended[leg].Before(inst) {
|
||||
continue // ran to its end at or after its last scheduled time
|
||||
}
|
||||
if next, ok := NextInstant(now, lt[leg], loc); ok && next.Sub(now) < leaveToNormal {
|
||||
continue // its normal run is about to happen
|
||||
}
|
||||
legs = append(legs, leg)
|
||||
if inst.After(lastW) {
|
||||
lastW = inst
|
||||
}
|
||||
}
|
||||
return legs, lastW
|
||||
}
|
||||
|
||||
// CatchUp runs at most one catch-up at a time.
|
||||
type CatchUp struct {
|
||||
Ledger *Ledger
|
||||
Window func() string // the household's backup window W (settings > config > 02:30)
|
||||
Legs map[Leg]func(context.Context) error // the leg bodies — WITHOUT the update leg
|
||||
QuiesceBusy func() bool // a whole-guest backup holds the apps now
|
||||
Event func(missedAt time.Time) // the household's timeline line
|
||||
Logger *log.Logger
|
||||
Delay time.Duration
|
||||
Loc *time.Location
|
||||
Now func() time.Time
|
||||
// Sleep waits d or until ctx ends (false). A seam for tests.
|
||||
Sleep func(ctx context.Context, d time.Duration) bool
|
||||
|
||||
mu sync.Mutex
|
||||
pending bool
|
||||
running bool
|
||||
}
|
||||
|
||||
func (c *CatchUp) now() time.Time {
|
||||
if c.Now != nil {
|
||||
return c.Now()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
func (c *CatchUp) sleep(ctx context.Context, d time.Duration) bool {
|
||||
if c.Sleep != nil {
|
||||
return c.Sleep(ctx, d)
|
||||
}
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return false
|
||||
case <-t.C:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
func (c *CatchUp) delay() time.Duration {
|
||||
if c.Delay > 0 {
|
||||
return c.Delay
|
||||
}
|
||||
return DefaultDelay
|
||||
}
|
||||
|
||||
// Running reports whether a catch-up is running its legs now (the whole-guest backup waits for it).
|
||||
func (c *CatchUp) Running() bool {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
return c.running
|
||||
}
|
||||
|
||||
// Evaluate is the trigger (controller start, host resume). It schedules ONE catch-up when a leg was missed and
|
||||
// none is pending; it returns what it scheduled (nil = nothing missed or one already pending). The work runs in
|
||||
// its own goroutine; Wait-style callers use EvaluateSync.
|
||||
func (c *CatchUp) Evaluate(ctx context.Context, why string) []Leg {
|
||||
legs, lastW := c.Ledger.Missed(c.now(), c.Window(), c.Loc)
|
||||
if len(legs) == 0 {
|
||||
c.Logger.Printf("[INFO] [catch-up] %s: no backup leg missed its last scheduled time — nothing to make up", why)
|
||||
return nil
|
||||
}
|
||||
c.mu.Lock()
|
||||
if c.pending || c.running {
|
||||
c.mu.Unlock()
|
||||
c.Logger.Printf("[INFO] [catch-up] %s: missed %v, but a catch-up is already scheduled", why, legs)
|
||||
return nil
|
||||
}
|
||||
c.pending = true
|
||||
c.mu.Unlock()
|
||||
c.Logger.Printf("[INFO] [catch-up] %s: the box missed %v (last scheduled %s) — ONE catch-up in %s (backup legs only; app updates wait for a real night)",
|
||||
why, legs, lastW.In(c.Loc).Format("2006-01-02 15:04"), c.delay())
|
||||
go c.run(ctx, why)
|
||||
return legs
|
||||
}
|
||||
|
||||
// run waits, re-reads the ledger (a normal run may have happened meanwhile), waits out a whole-guest backup, then
|
||||
// runs the still-missed legs in chain order.
|
||||
func (c *CatchUp) run(ctx context.Context, why string) {
|
||||
defer func() {
|
||||
c.mu.Lock()
|
||||
c.pending, c.running = false, false
|
||||
c.mu.Unlock()
|
||||
}()
|
||||
if !c.sleep(ctx, c.delay()) {
|
||||
return
|
||||
}
|
||||
legs, lastW := c.Ledger.Missed(c.now(), c.Window(), c.Loc)
|
||||
if len(legs) == 0 {
|
||||
c.Logger.Printf("[INFO] [catch-up] (%s) nothing is missed any more — not run", why)
|
||||
return
|
||||
}
|
||||
for waited := time.Duration(0); c.QuiesceBusy != nil && c.QuiesceBusy(); waited += time.Minute {
|
||||
if waited >= quiesceWaitMax {
|
||||
c.Logger.Printf("[WARN] [catch-up] (%s) a whole-guest backup held the apps for %s — catch-up dropped; the next start or resume tries again", why, waited)
|
||||
return
|
||||
}
|
||||
if !c.sleep(ctx, time.Minute) {
|
||||
return
|
||||
}
|
||||
}
|
||||
c.mu.Lock()
|
||||
c.running = true
|
||||
c.mu.Unlock()
|
||||
start := c.now()
|
||||
var names []string
|
||||
for _, leg := range legs {
|
||||
fn := c.Legs[leg]
|
||||
if fn == nil {
|
||||
continue
|
||||
}
|
||||
c.Logger.Printf("[INFO] [catch-up] running the missed %s leg", leg)
|
||||
if err := fn(ctx); err != nil {
|
||||
c.Logger.Printf("[WARN] [catch-up] the %s leg ended with an error (its own alarm reports it): %v", leg, err)
|
||||
}
|
||||
names = append(names, string(leg))
|
||||
}
|
||||
c.Ledger.markCatchUp(c.now())
|
||||
sort.Strings(names)
|
||||
c.Logger.Printf("[INFO] [catch-up] done: %s in %s (missed at %s)", strings.Join(names, ", "),
|
||||
c.now().Sub(start).Round(time.Second), lastW.In(c.Loc).Format("2006-01-02 15:04"))
|
||||
if c.Event != nil {
|
||||
c.Event(lastW)
|
||||
}
|
||||
}
|
||||
|
||||
// ResumeWatch calls onResume when, between two ticks, the WALL clock advanced more than gap beyond the MONOTONIC
|
||||
// clock — the shape of a host suspend/resume: Go's timers and monotonic readings run on CLOCK_MONOTONIC, which does
|
||||
// not count suspended time, while the wall clock does. Seams: wall() is the wall clock, mono() a monotonic elapsed
|
||||
// duration (production: time.Now().Round(0) and time.Since(a fixed start)). Pinned by TestResumeWatch_*.
|
||||
func ResumeWatch(ctx context.Context, tick <-chan time.Time, wall func() time.Time, mono func() time.Duration, gap time.Duration, onResume func(slept time.Duration)) {
|
||||
pw, pm := wall(), mono()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-tick:
|
||||
}
|
||||
cw, cm := wall(), mono()
|
||||
if d := cw.Sub(pw) - (cm - pm); d > gap {
|
||||
onResume(d)
|
||||
}
|
||||
pw, pm = cw, cm
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user