635c33d381
gates / gates (push) Successful in 31s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
366 lines
13 KiB
Go
366 lines
13 KiB
Go
// Package nightchain makes a missed night's backups up once, when the box comes back (R-871, `09` §3 decision 109,
|
|
// design `07` §6.1.1).
|
|
//
|
|
// THE PROBLEM, MEASURED (2026-10-05 Part F spike, Tester 2 — a laptop switched off at night): the controller's
|
|
// daily jobs always schedule the NEXT future time (scheduler.nextDailyRun), and nothing remembers that a night was
|
|
// missed. A box that is off at its window W never gets its database dumps, its second copy or its off-site copy —
|
|
// for ever, with no alarm.
|
|
//
|
|
// THE MECHANISM:
|
|
// - A LEDGER (`night-ledger.json` in the data dir) records when each of the three nightly backup legs last RAN TO
|
|
// ITS END. It is an attempt record and is used for ONE question only — "was this leg's last scheduled time
|
|
// missed?" — never as evidence that a backup exists (that is LastSuccess's job, R-100). A leg that ran and
|
|
// failed was NOT missed: failures have their own alarms.
|
|
// - On a controller START and on a host RESUME (a suspended laptop: Go timers run on CLOCK_MONOTONIC, which does
|
|
// not count suspended time), Evaluate asks the ledger which legs missed their last scheduled time. If any did,
|
|
// ONE catch-up is scheduled 15 minutes later (apps settle; a box switched on and off again at once does nothing).
|
|
// - The catch-up runs ONLY the backup legs, in their normal order (database dump, second copy, off-site copy).
|
|
// It never runs app updates or Docker steps — they restart apps and wait for a real night.
|
|
// - Several missed nights are ONE catch-up (the question is about the LAST scheduled time only). A normal night
|
|
// followed by a daytime restart is none. A power cut in the middle of the chain makes up only the legs that did
|
|
// not end.
|
|
// - A leg whose own scheduled time is less than 30 minutes away is left to its normal run.
|
|
// - Every leg — scheduled or catch-up — holds one mutex (`Ledger.LegLock`), so a catch-up and a scheduled leg
|
|
// never run at the same time. The whole-guest backup (agent-side, quiesce) waits while a catch-up runs
|
|
// (quiesce.Options.CatchUpFn) and the catch-up waits while a quiesce holds the apps (QuiesceBusy).
|
|
//
|
|
// A ledger that does not exist yet (the first start on this release) is SEEDED at that moment and nothing before
|
|
// it counts as missed — an upgrade must not set off a surprise catch-up on every box.
|
|
//
|
|
// Pinned by nightchain_test.go (TestCatchUp_*), each rule with a red-proof.
|
|
package nightchain
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"log"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/backupwindow"
|
|
)
|
|
|
|
// Leg is one nightly backup leg.
|
|
type Leg string
|
|
|
|
const (
|
|
LegDBDump Leg = "db-dump"
|
|
LegTier2 Leg = "tier2"
|
|
LegOffsite Leg = "offsite"
|
|
)
|
|
|
|
// Order is the night chain's own order (07 §6.1).
|
|
var Order = []Leg{LegDBDump, LegTier2, LegOffsite}
|
|
|
|
const (
|
|
// DefaultDelay: the catch-up waits this long after the trigger (decision 109's "soon after it is switched on").
|
|
DefaultDelay = 15 * time.Minute
|
|
// leaveToNormal: a leg whose own scheduled time is closer than this runs normally, not in the catch-up.
|
|
leaveToNormal = 30 * time.Minute
|
|
// quiesceWaitMax: how long a catch-up waits for a whole-guest backup that holds the apps.
|
|
quiesceWaitMax = 2 * time.Hour
|
|
)
|
|
|
|
type ledgerData struct {
|
|
SeededAt time.Time `json:"seeded_at"`
|
|
Ended map[Leg]time.Time `json:"ended"` // ATTEMPT record: the leg ran to its end (success or failure)
|
|
// DBDumpOK is the last SUCCESSFUL database dump (the banner's evidence; the off-site tier has its own
|
|
// LastSuccess in settings). Written only on success (R-100's rule).
|
|
DBDumpOK time.Time `json:"db_dump_ok,omitempty"`
|
|
LastCatchUp time.Time `json:"last_catch_up,omitempty"`
|
|
BannerDismissed time.Time `json:"banner_dismissed_through,omitempty"` // the missed W instant the household closed
|
|
}
|
|
|
|
// Ledger is the persisted night record.
|
|
type Ledger struct {
|
|
mu sync.Mutex
|
|
path string
|
|
d ledgerData
|
|
// LegLock is held by every leg run, scheduled or catch-up.
|
|
LegLock sync.Mutex
|
|
}
|
|
|
|
// Open reads the ledger, or seeds a new one at now.
|
|
func Open(path string, now time.Time) (*Ledger, error) {
|
|
l := &Ledger{path: path, d: ledgerData{Ended: map[Leg]time.Time{}}}
|
|
b, err := os.ReadFile(path)
|
|
switch {
|
|
case err == nil:
|
|
if jerr := json.Unmarshal(b, &l.d); jerr != nil {
|
|
return nil, fmt.Errorf("nightchain: ledger unreadable: %w", jerr)
|
|
}
|
|
if l.d.Ended == nil {
|
|
l.d.Ended = map[Leg]time.Time{}
|
|
}
|
|
case os.IsNotExist(err):
|
|
l.d.SeededAt = now.UTC()
|
|
if serr := l.saveLocked(); serr != nil {
|
|
return nil, serr
|
|
}
|
|
default:
|
|
return nil, fmt.Errorf("nightchain: ledger: %w", err)
|
|
}
|
|
return l, nil
|
|
}
|
|
|
|
func (l *Ledger) saveLocked() error {
|
|
b, _ := json.MarshalIndent(l.d, "", " ")
|
|
if err := os.MkdirAll(filepath.Dir(l.path), 0o755); err != nil {
|
|
return err
|
|
}
|
|
tmp := l.path + ".tmp"
|
|
if err := os.WriteFile(tmp, b, 0o644); err != nil {
|
|
return err
|
|
}
|
|
return os.Rename(tmp, l.path)
|
|
}
|
|
|
|
// MarkEnded records that a leg ran to its end (success or failure).
|
|
func (l *Ledger) MarkEnded(leg Leg, at time.Time) {
|
|
l.mu.Lock()
|
|
defer l.mu.Unlock()
|
|
l.d.Ended[leg] = at.UTC()
|
|
_ = l.saveLocked()
|
|
}
|
|
|
|
// MarkDBDumpOK records a successful database dump.
|
|
func (l *Ledger) MarkDBDumpOK(at time.Time) {
|
|
l.mu.Lock()
|
|
defer l.mu.Unlock()
|
|
l.d.DBDumpOK = at.UTC()
|
|
_ = l.saveLocked()
|
|
}
|
|
|
|
func (l *Ledger) markCatchUp(at time.Time) {
|
|
l.mu.Lock()
|
|
defer l.mu.Unlock()
|
|
l.d.LastCatchUp = at.UTC()
|
|
_ = l.saveLocked()
|
|
}
|
|
|
|
// DismissBanner records that the household closed the banner for every miss up to `through`.
|
|
func (l *Ledger) DismissBanner(through time.Time) {
|
|
l.mu.Lock()
|
|
defer l.mu.Unlock()
|
|
l.d.BannerDismissed = through.UTC()
|
|
_ = l.saveLocked()
|
|
}
|
|
|
|
// Snapshot returns a copy of the record (for the banner and the debug page).
|
|
func (l *Ledger) Snapshot() (seeded, dbDumpOK, lastCatchUp, dismissed time.Time, ended map[Leg]time.Time) {
|
|
l.mu.Lock()
|
|
defer l.mu.Unlock()
|
|
ended = map[Leg]time.Time{}
|
|
for k, v := range l.d.Ended {
|
|
ended[k] = v
|
|
}
|
|
return l.d.SeededAt, l.d.DBDumpOK, l.d.LastCatchUp, l.d.BannerDismissed, ended
|
|
}
|
|
|
|
// legOffsets: each leg's time relative to W — taken from backupwindow.LegTimes so the two can never disagree.
|
|
func legTimes(window string) map[Leg]string {
|
|
db, t2, off := backupwindow.LegTimes(window)
|
|
return map[Leg]string{LegDBDump: db, LegTier2: t2, LegOffsite: off}
|
|
}
|
|
|
|
// LastInstant is the most recent occurrence of the Budapest clock time hhmm at or before now.
|
|
func LastInstant(now time.Time, hhmm string, loc *time.Location) (time.Time, bool) {
|
|
min, err := backupwindow.ParseHHMM(hhmm)
|
|
if err != nil {
|
|
return time.Time{}, false
|
|
}
|
|
n := now.In(loc)
|
|
t := time.Date(n.Year(), n.Month(), n.Day(), min/60, min%60, 0, 0, loc)
|
|
if t.After(n) {
|
|
t = time.Date(n.Year(), n.Month(), n.Day()-1, min/60, min%60, 0, 0, loc)
|
|
}
|
|
return t, true
|
|
}
|
|
|
|
// NextInstant is the next occurrence of hhmm strictly after now.
|
|
func NextInstant(now time.Time, hhmm string, loc *time.Location) (time.Time, bool) {
|
|
last, ok := LastInstant(now, hhmm, loc)
|
|
if !ok {
|
|
return time.Time{}, false
|
|
}
|
|
return time.Date(last.Year(), last.Month(), last.Day()+1, last.Hour(), last.Minute(), 0, 0, loc), true
|
|
}
|
|
|
|
// Missed is the PURE rule: the legs (in chain order) whose last scheduled time passed after the ledger was seeded
|
|
// and that have not run to their end since — minus a leg whose next scheduled time is under 30 minutes away.
|
|
// lastW is the most recent missed leg instant (the banner's "the box was off at …").
|
|
func (l *Ledger) Missed(now time.Time, window string, loc *time.Location) (legs []Leg, lastW time.Time) {
|
|
l.mu.Lock()
|
|
defer l.mu.Unlock()
|
|
lt := legTimes(window)
|
|
for _, leg := range Order {
|
|
inst, ok := LastInstant(now, lt[leg], loc)
|
|
if !ok || !inst.After(l.d.SeededAt) {
|
|
continue
|
|
}
|
|
if !l.d.Ended[leg].Before(inst) {
|
|
continue // ran to its end at or after its last scheduled time
|
|
}
|
|
if next, ok := NextInstant(now, lt[leg], loc); ok && next.Sub(now) < leaveToNormal {
|
|
continue // its normal run is about to happen
|
|
}
|
|
legs = append(legs, leg)
|
|
if inst.After(lastW) {
|
|
lastW = inst
|
|
}
|
|
}
|
|
return legs, lastW
|
|
}
|
|
|
|
// CatchUp runs at most one catch-up at a time.
|
|
type CatchUp struct {
|
|
Ledger *Ledger
|
|
Window func() string // the household's backup window W (settings > config > 02:30)
|
|
Legs map[Leg]func(context.Context) error // the leg bodies — WITHOUT the update leg
|
|
QuiesceBusy func() bool // a whole-guest backup holds the apps now
|
|
Event func(missedAt time.Time) // the household's timeline line
|
|
Logger *log.Logger
|
|
Delay time.Duration
|
|
Loc *time.Location
|
|
Now func() time.Time
|
|
// Sleep waits d or until ctx ends (false). A seam for tests.
|
|
Sleep func(ctx context.Context, d time.Duration) bool
|
|
|
|
mu sync.Mutex
|
|
pending bool
|
|
running bool
|
|
}
|
|
|
|
func (c *CatchUp) now() time.Time {
|
|
if c.Now != nil {
|
|
return c.Now()
|
|
}
|
|
return time.Now()
|
|
}
|
|
|
|
func (c *CatchUp) sleep(ctx context.Context, d time.Duration) bool {
|
|
if c.Sleep != nil {
|
|
return c.Sleep(ctx, d)
|
|
}
|
|
t := time.NewTimer(d)
|
|
defer t.Stop()
|
|
select {
|
|
case <-ctx.Done():
|
|
return false
|
|
case <-t.C:
|
|
return true
|
|
}
|
|
}
|
|
|
|
func (c *CatchUp) delay() time.Duration {
|
|
if c.Delay > 0 {
|
|
return c.Delay
|
|
}
|
|
return DefaultDelay
|
|
}
|
|
|
|
// Running reports whether a catch-up is running its legs now (the whole-guest backup waits for it).
|
|
func (c *CatchUp) Running() bool {
|
|
c.mu.Lock()
|
|
defer c.mu.Unlock()
|
|
return c.running
|
|
}
|
|
|
|
// Evaluate is the trigger (controller start, host resume). It schedules ONE catch-up when a leg was missed and
|
|
// none is pending; it returns what it scheduled (nil = nothing missed or one already pending). The work runs in
|
|
// its own goroutine; Wait-style callers use EvaluateSync.
|
|
func (c *CatchUp) Evaluate(ctx context.Context, why string) []Leg {
|
|
legs, lastW := c.Ledger.Missed(c.now(), c.Window(), c.Loc)
|
|
if len(legs) == 0 {
|
|
c.Logger.Printf("[INFO] [catch-up] %s: no backup leg missed its last scheduled time — nothing to make up", why)
|
|
return nil
|
|
}
|
|
c.mu.Lock()
|
|
if c.pending || c.running {
|
|
c.mu.Unlock()
|
|
c.Logger.Printf("[INFO] [catch-up] %s: missed %v, but a catch-up is already scheduled", why, legs)
|
|
return nil
|
|
}
|
|
c.pending = true
|
|
c.mu.Unlock()
|
|
c.Logger.Printf("[INFO] [catch-up] %s: the box missed %v (last scheduled %s) — ONE catch-up in %s (backup legs only; app updates wait for a real night)",
|
|
why, legs, lastW.In(c.Loc).Format("2006-01-02 15:04"), c.delay())
|
|
go c.run(ctx, why)
|
|
return legs
|
|
}
|
|
|
|
// run waits, re-reads the ledger (a normal run may have happened meanwhile), waits out a whole-guest backup, then
|
|
// runs the still-missed legs in chain order.
|
|
func (c *CatchUp) run(ctx context.Context, why string) {
|
|
defer func() {
|
|
c.mu.Lock()
|
|
c.pending, c.running = false, false
|
|
c.mu.Unlock()
|
|
}()
|
|
if !c.sleep(ctx, c.delay()) {
|
|
return
|
|
}
|
|
legs, lastW := c.Ledger.Missed(c.now(), c.Window(), c.Loc)
|
|
if len(legs) == 0 {
|
|
c.Logger.Printf("[INFO] [catch-up] (%s) nothing is missed any more — not run", why)
|
|
return
|
|
}
|
|
for waited := time.Duration(0); c.QuiesceBusy != nil && c.QuiesceBusy(); waited += time.Minute {
|
|
if waited >= quiesceWaitMax {
|
|
c.Logger.Printf("[WARN] [catch-up] (%s) a whole-guest backup held the apps for %s — catch-up dropped; the next start or resume tries again", why, waited)
|
|
return
|
|
}
|
|
if !c.sleep(ctx, time.Minute) {
|
|
return
|
|
}
|
|
}
|
|
c.mu.Lock()
|
|
c.running = true
|
|
c.mu.Unlock()
|
|
start := c.now()
|
|
var names []string
|
|
for _, leg := range legs {
|
|
fn := c.Legs[leg]
|
|
if fn == nil {
|
|
continue
|
|
}
|
|
c.Logger.Printf("[INFO] [catch-up] running the missed %s leg", leg)
|
|
if err := fn(ctx); err != nil {
|
|
c.Logger.Printf("[WARN] [catch-up] the %s leg ended with an error (its own alarm reports it): %v", leg, err)
|
|
}
|
|
names = append(names, string(leg))
|
|
}
|
|
c.Ledger.markCatchUp(c.now())
|
|
sort.Strings(names)
|
|
c.Logger.Printf("[INFO] [catch-up] done: %s in %s (missed at %s)", strings.Join(names, ", "),
|
|
c.now().Sub(start).Round(time.Second), lastW.In(c.Loc).Format("2006-01-02 15:04"))
|
|
if c.Event != nil {
|
|
c.Event(lastW)
|
|
}
|
|
}
|
|
|
|
// ResumeWatch calls onResume when, between two ticks, the WALL clock advanced more than gap beyond the MONOTONIC
|
|
// clock — the shape of a host suspend/resume: Go's timers and monotonic readings run on CLOCK_MONOTONIC, which does
|
|
// not count suspended time, while the wall clock does. Seams: wall() is the wall clock, mono() a monotonic elapsed
|
|
// duration (production: time.Now().Round(0) and time.Since(a fixed start)). Pinned by TestResumeWatch_*.
|
|
func ResumeWatch(ctx context.Context, tick <-chan time.Time, wall func() time.Time, mono func() time.Duration, gap time.Duration, onResume func(slept time.Duration)) {
|
|
pw, pm := wall(), mono()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-tick:
|
|
}
|
|
cw, cm := wall(), mono()
|
|
if d := cw.Sub(pw) - (cm - pm); d > gap {
|
|
onResume(d)
|
|
}
|
|
pw, pm = cw, cm
|
|
}
|
|
}
|