Files
felhom-controller/controller/internal/web/night_chain.go
T

111 lines
4.1 KiB
Go

package web
import (
"context"
"net/http"
"sync/atomic"
"time"
)
// ── R-705 (v0.279.0): "run tonight's chain now" — a debug action ─────────────────────────────────────
//
// The night runs four legs in ONE order (07 §6.1, 09 §6.4.2): the database/volume dump at W, the
// second-drive copy at W+60m, the off-site copy at W+105m, and the automatic update leg chained after the
// off-site one. Until now the only way to see that chain by day was to move the backup window and wait
// two hours. This runs the same four legs, in the same order, one at a time, NOW. The whole-guest backup
// (the agent's) is not part of it — it has no controller trigger (R-705 keeps that half open).
//
// Refused while any backup/restore op or guarded update runs, and while a chain is already running.
// nightChain is the four legs; each field is a seam so the order and the refusal are testable without
// Docker or restic. production: newNightChain(s).
type nightChain struct {
dump func(ctx context.Context) error
tier2 func()
offsite func(ctx context.Context) error // nil: no off-site target on this box
leg func(ctx context.Context)
busy func() (bool, string)
logf func(format string, args ...interface{})
}
var nightChainRunning atomic.Bool
func (s *Server) newNightChain() nightChain {
c := nightChain{
dump: s.backupMgr.RunDBDumps,
tier2: s.backupMgr.RunAllTier2,
leg: func(ctx context.Context) { s.stackMgr.RunUpdateLegNow(ctx, "manual-chain") },
logf: s.logger.Printf,
busy: func() (bool, string) {
if s.backupMgr.IsRunning() || s.backupMgr.RestoreStatus().Running {
return true, "a backup or restore is running"
}
if s.stackMgr.AnyUpdating() {
return true, "a guarded update is running"
}
return false, ""
},
}
if s.backupMgr.OffboxRunnable() {
c.offsite = s.backupMgr.RunOffboxBackup
}
return c
}
// start refuses or launches; it reports which legs will run.
func (c nightChain) start() (bool, string, []string) {
if busy, why := c.busy(); busy {
return false, why, nil
}
if !nightChainRunning.CompareAndSwap(false, true) {
return false, "the night's chain is already running", nil
}
legs := []string{"db-dump", "tier2", "offsite", "update-leg"}
if c.offsite == nil {
legs = []string{"db-dump", "tier2", "update-leg"}
}
go c.run()
return true, "", legs
}
func (c nightChain) run() {
defer nightChainRunning.Store(false)
ctx := context.Background()
t0 := time.Now()
step := func(name string, fn func() error) {
s := time.Now()
c.logf("[INFO] [night-chain] %s: started", name)
if err := fn(); err != nil {
c.logf("[WARN] [night-chain] %s: ended with an error after %s: %v — the chain goes on, as the night does", name, time.Since(s).Round(time.Second), err)
return
}
c.logf("[INFO] [night-chain] %s: done in %s", name, time.Since(s).Round(time.Second))
}
c.logf("[INFO] [night-chain] manual run of tonight's chain: dump → second drive → off-site → update leg")
step("db-dump", func() error { return c.dump(ctx) })
step("tier2", func() error { c.tier2(); return nil })
if c.offsite != nil {
step("offsite", func() error { return c.offsite(ctx) })
} else {
c.logf("[INFO] [night-chain] offsite: no off-site target on this box — skipped, as at night")
}
step("update-leg", func() error { c.leg(ctx); return nil })
c.logf("[INFO] [night-chain] finished in %s", time.Since(t0).Round(time.Second))
}
func (s *Server) debugRunNightChain(w http.ResponseWriter, r *http.Request) {
if s.backupMgr == nil || s.stackMgr == nil {
writeDebugJSON(w, http.StatusBadRequest, false, "Backup manager nincs konfigurálva", nil)
return
}
s.backupMgr.MarkManualRun()
ok, why, legs := s.newNightChain().start()
if !ok {
s.logger.Printf("[WARN] [night-chain] manual run REFUSED: %s", why)
writeDebugJSON(w, http.StatusConflict, false, "refused: "+why, nil)
return
}
s.logger.Printf("[INFO] [night-chain] manual run started from %s: %v", r.RemoteAddr, legs)
writeDebugJSON(w, http.StatusAccepted, true, "started", map[string]interface{}{"legs": legs})
}