Files
felhom-controller/controller/internal/quiesce/quiesce.go
T
Claude Code 9e5ea56853 v0.175.0 — R-82: a tier that overruns the quiesce bound defers the rest
Operator ruling 2026-07-26: let the first backup run as long as needed; other
backups shouldn't start until finished.

A first FULL offsite snapshot runs for hours, far past max_quiesce. When that
bound elapses the app resumes (unchanged), but the loop then started the NEXT
tier while the first was still uploading. Now it breaks and defers the rest to
a later poll — vzdump still holds the guest lock, so the second start would be
refused by the agent (409, v0.99.0) or fail on the lock, and a failed backup
never satisfies a cadence, so the tier would retry into the same wall forever.

pollTier returns (phase, stillRunning, err). The app still resumes exactly once.

Red-proof observed and restored; full suite green.
2026-07-26 15:06:10 +02:00

545 lines
22 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Package quiesce implements the slice-8B app-consistent backup loop (doc 03 §6/§8): the
// in-guest controller polls the host agent's GET /backup/due, and when due it QUIESCES (stops its
// app stacks) → POST /backup → polls GET /backup/status to completion → UNQUIESCES (restarts
// exactly the stacks it stopped). An agent-initiated vzdump is crash-consistent only (an LXC has
// no fsfreeze); stopping the stacks first makes the captured state clean-shutdown-consistent.
//
// The correctness centerpiece is crash-safety: a stranded-down app is worse than a crash-consistent
// backup. So: a persisted marker is written BEFORE stopping anything; unquiesce is guaranteed (it
// runs even when the backup errors or times out); a max-quiesce bound restarts the app no matter
// what; and on controller startup Recover() restarts any stacks left stopped by a mid-quiesce crash.
package quiesce
import (
"context"
"encoding/json"
"errors"
"fmt"
"log"
"os"
"path/filepath"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/backupwindow"
)
// ErrBackupInProgress is returned by TriggerNow when a scheduled or manual quiesce cycle is already
// running (single-flight). The caller (the "Mentés most" handler) surfaces it as a benign 409.
var ErrBackupInProgress = errors.New("quiesce: a backup cycle is already in progress")
// Backend is the agent local-API surface the loop needs (satisfied by an adapter over
// *agentapi.Client). Kept minimal (bool/int/string) so the loop is testable with plain fakes.
// Due also returns the age of the newest successful backup in seconds (nil = none yet) — the
// window gate's safety valve reads it so a box powered on only outside its window never starves.
type Backend interface {
Due(ctx context.Context) (due bool, ageSecs *int64, err error)
StartBackup(ctx context.Context) (jobID string, err error)
BackupStatus(ctx context.Context) (phase string, err error)
}
// Stacks is the stack-control surface (satisfied by *stacks.Manager). RunningAppStacks must return
// only deployed, non-protected, currently-up stacks (so unquiesce restarts exactly those).
type Stacks interface {
RunningAppStacks() []string
StopStack(name string) error
StartStack(name string) error
}
// Backup status phases (mirror the agent's vocabulary).
const (
phaseSnapshotted = "snapshotted" // 8B.2: storage snapshot taken → app may resume early
phaseDone = "done"
phaseFailed = "failed"
)
// Marker is the persisted quiesce state — the crash-safety + single-flight record. It is written
// (atomically, 0600) BEFORE any stack is stopped, so a controller crash mid-quiesce leaves a
// durable "these stacks were stopped, restart them" note that Recover honors at next startup.
type Marker struct {
Active bool `json:"active"`
StartedAt time.Time `json:"started_at"`
StoppedStacks []string `json:"stopped_stacks"`
JobID string `json:"job_id"`
}
// Options configures a Loop.
type Options struct {
Backend Backend
Stacks Stacks
MarkerPath string // persisted marker (e.g. <data_dir>/quiesce-state.json)
Poll time.Duration // how often to check /backup/due
StatusPoll time.Duration // how often to poll /backup/status while quiesced
MaxQuiesce time.Duration // hard bound on app downtime (unquiesce no matter what)
Logger *log.Logger
// WindowStartFn returns the CURRENT effective backup-window start "HH:MM" (customer-configurable,
// so it is read fresh each poll — a window change must take effect without restart). When nil the
// window gate is disabled and a due cycle runs whenever the agent says due (pre-v0.168.0 behavior).
WindowStartFn func() string
// Cadence is the agent's backup cadence, used only by the gate's safety valve (run regardless of
// the window once the last successful backup is older than Cadence+24h). Defaults to 24h.
Cadence time.Duration
}
// Loop is the quiesce background loop.
type Loop struct {
backend Backend
stacks Stacks
markerPath string
poll time.Duration
statusPoll time.Duration
maxQuiesce time.Duration
logger *log.Logger
now func() time.Time
// windowStartFn (nil = gate disabled) + cadence drive the scheduled-cycle window gate (Part 3).
windowStartFn func() string
cadence time.Duration
// mu single-flights the quiesce cycle across the scheduled loop AND the manual trigger, so the
// two can never stop the same stacks concurrently (the persisted marker covers crash-safety across
// restarts; this covers concurrency within the process — which a manual trigger introduces).
mu sync.Mutex
// degradeOnce reports the pre-R-82 agent fallback exactly once per process (see tiers.go).
degradeOnce sync.Once
}
// New builds a Loop with sane defaults for any unset duration.
func New(o Options) *Loop {
if o.Poll <= 0 {
o.Poll = 5 * time.Minute
}
if o.StatusPoll <= 0 {
o.StatusPoll = 10 * time.Second
}
if o.MaxQuiesce <= 0 {
o.MaxQuiesce = 30 * time.Minute
}
if o.Logger == nil {
o.Logger = log.Default()
}
if o.Cadence <= 0 {
o.Cadence = 24 * time.Hour
}
return &Loop{
backend: o.Backend, stacks: o.Stacks, markerPath: o.MarkerPath,
poll: o.Poll, statusPoll: o.StatusPoll, maxQuiesce: o.MaxQuiesce,
logger: o.Logger, now: time.Now,
windowStartFn: o.WindowStartFn, cadence: o.Cadence,
}
}
// Recover restarts any stacks left stopped by a controller crash mid-quiesce, then clears the
// marker. Call ONCE at startup, before Run. Idempotent — StartStack on an already-running stack is
// tolerated; an absent/inactive marker is a no-op.
func (l *Loop) Recover() {
m, ok := l.readMarker()
if !ok || !m.Active {
return
}
l.logger.Printf("[WARN] [quiesce] crash recovery: a quiesce was in progress (job %q, %d stack(s) stopped) — restarting them",
m.JobID, len(m.StoppedStacks))
l.restartAll(m.StoppedStacks)
if err := l.clearMarker(); err != nil {
l.logger.Printf("[ERROR] [quiesce] crash recovery: clear marker: %v", err)
}
}
// Run polls for a due backup and runs the quiesce cycle, until ctx is cancelled.
func (l *Loop) Run(ctx context.Context) {
l.logger.Printf("[INFO] [quiesce] loop started (poll %s, max-quiesce %s)", l.poll, l.maxQuiesce)
ticker := time.NewTicker(l.poll)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
l.logger.Printf("[INFO] [quiesce] loop stopping")
return
case <-ticker.C:
if err := l.runOnce(ctx); err != nil && ctx.Err() == nil {
l.logger.Printf("[ERROR] [quiesce] cycle error: %v", err)
}
}
}
}
// runOnce performs one due-check → (if due) quiesce → backup → poll → unquiesce cycle. Unquiesce
// is guaranteed via the deferred closure: a backup error, a status-poll error, the max-quiesce
// bound, or context cancellation all still restart the stacks and clear the marker.
func (l *Loop) runOnce(ctx context.Context) error {
// Single-flight: skip the scheduled check if a cycle (scheduled or manual) is already running.
if !l.mu.TryLock() {
l.logger.Printf("[INFO] [quiesce] a backup cycle is already running — skipping this scheduled check")
return nil
}
defer l.mu.Unlock()
// Defensive single-flight: never quiesce on top of an active marker (Recover clears one left
// by a crash; the mutex above serializes within the process).
if m, ok := l.readMarker(); ok && m.Active {
l.logger.Printf("[WARN] [quiesce] a marker is already active — skipping this cycle")
return nil
}
// R-82: resolve EVERY due tier up front. This is the dedup rule (tiers.go): both tiers due on
// the weekly night yields ONE window with two backups, never two stop/start cycles.
dueTiers, _, err := l.resolveDueTiers(ctx)
if err != nil {
return fmt.Errorf("check due: %w", err)
}
if len(dueTiers) == 0 {
return nil
}
// Window gate (Part 3) — SCHEDULED path only. TriggerNow calls quiesceAndPoll directly and is
// never gated. Disabled when no window fn is wired (pre-v0.168.0 behavior).
//
// With several tiers due, the gate is evaluated against the OLDEST (most overdue) tier's age,
// so the safety valve — "run regardless of the window once the last success is older than
// cadence+24h" — cannot be suppressed by a fresher sibling tier.
if l.windowStartFn != nil {
window := l.windowStartFn()
if !scheduledRunAllowed(l.now().In(budapestLocation()), window, oldestAge(dueTiers), l.cadence) {
from, to := gateBounds(window)
l.logger.Printf("[DEBUG] [quiesce] scheduled backup due but outside the backup window [%s%s) — deferring to the next poll inside it", from, to)
return nil
}
}
return l.quiesceAndPollTiers(ctx, dueTiers)
}
// oldestAge returns the largest (most overdue) age among the due tiers; nil when any tier has never
// backed up (nil age = "never", which is maximally overdue and must win).
func oldestAge(tiers []dueTier) *int64 {
var oldest *int64
for _, t := range tiers {
if t.ageSecs == nil {
return nil // never backed up — the strongest claim on the safety valve
}
if oldest == nil || *t.ageSecs > *oldest {
oldest = t.ageSecs
}
}
return oldest
}
// TriggerNow forces an app-consistent backup NOW (the manual "Mentés most" action), bypassing the
// /backup/due check. It runs the SAME quiesce flow the scheduled loop uses (stop stacks → POST
// /backup → poll → resume), so it is app-consistent and crash-safe (marker-protected). Single-flight
// via the same mutex: it returns ErrBackupInProgress if a scheduled or manual cycle is already
// running. The cycle runs ASYNCHRONOUSLY (it can take minutes) on a background context bounded by
// maxQuiesce; the caller polls /backup/status for progress. The controller — not the agent — owns
// quiescing (the agent's vzdump is crash-consistent only), so this MUST go through the loop.
func (l *Loop) TriggerNow() error {
if !l.mu.TryLock() {
return ErrBackupInProgress
}
if m, ok := l.readMarker(); ok && m.Active {
l.mu.Unlock()
return ErrBackupInProgress
}
go func() {
defer l.mu.Unlock()
// Detached from any request context; bounded so a hung backup still unquiesces.
ctx, cancel := context.WithTimeout(context.Background(), l.maxQuiesce+5*time.Minute)
defer cancel()
l.logger.Printf("[INFO] [quiesce] manual backup requested — quiescing now")
// Manual runs bypass due-ness (that is the point) but must still cover EVERY tier, in one
// window. A manual "Mentés most" that silently skipped the DR tier would be the same
// applied-and-empty fault in a different costume.
if err := l.quiesceAndPollTiers(ctx, l.allTiersForManualRun(ctx)); err != nil {
l.logger.Printf("[ERROR] [quiesce] manual backup cycle error: %v", err)
}
}()
return nil
}
// quiesceAndPoll performs the marked, guaranteed-unquiesce cycle: write marker → stop running app
// stacks → POST /backup → poll /backup/status → restart exactly the stacks it stopped. The caller
// MUST hold l.mu. Unquiesce is guaranteed via the deferred closure (backup error, status-poll error,
// the max-quiesce bound, or context cancellation all still restart the stacks and clear the marker).
func (l *Loop) quiesceAndPoll(ctx context.Context) error {
return l.quiesceAndPollTiers(ctx, []dueTier{{target: ""}})
}
// allTiersForManualRun lists every tier a manual run should cover: all advertised tiers on an R-82
// agent, or the single untargeted tier otherwise. Due-ness is deliberately NOT consulted.
func (l *Loop) allTiersForManualRun(ctx context.Context) []dueTier {
tb, ok := l.backend.(TieredBackend)
if !ok {
return []dueTier{{target: ""}}
}
tiers, err := tb.Tiers(ctx)
if err != nil || len(tiers) == 0 {
if errors.Is(err, ErrTiersUnsupported) {
l.logTierDegradeOnce()
} else if err != nil {
l.logger.Printf("[WARN] [quiesce] manual run: tier list unavailable (%v) — using the untargeted tier", err)
}
return []dueTier{{target: ""}}
}
out := make([]dueTier, 0, len(tiers))
for _, t := range tiers {
out = append(out, dueTier{target: t.Target})
}
return out
}
// quiesceAndPollTiers is the R-82 multi-tier cycle: ONE marker, ONE stop, N backups run
// SEQUENTIALLY inside the window, ONE resume, then the tail polled to completion.
//
// Why sequential: vzdump takes a guest lock, so a second backup cannot start until the first
// finishes. Why the app stays down until the LAST tier snapshots: the whole point of quiescing is
// app-consistency, and resuming after tier 1's snapshot would leave tier 2 capturing a RUNNING app.
// Consequence, stated plainly because it is user-visible: on the both-due night downtime is
// (first tier's full backup) + (last tier's snapshot), not one snapshot. Tier ORDER therefore
// matters — see resolveDueTiers.
//
// Crash-safety is unchanged and non-negotiable: the marker is written BEFORE anything stops,
// unquiesce is guaranteed by defer, and it fires exactly once no matter which tier fails. A crash
// between two backups leaves the marker on disk and Recover() restarts the stacks at startup.
func (l *Loop) quiesceAndPollTiers(ctx context.Context, tiers []dueTier) error {
if len(tiers) == 0 {
return nil
}
running := l.stacks.RunningAppStacks()
marker := Marker{Active: true, StartedAt: l.now(), StoppedStacks: running}
if err := l.writeMarker(marker); err != nil {
return fmt.Errorf("write quiesce marker (refusing to stop stacks unprotected): %w", err)
}
// GUARANTEED unquiesce + marker clear — runs on every exit path below.
unquiesced := false
unquiesce := func(reason string) {
if unquiesced {
return
}
unquiesced = true
l.logger.Printf("[INFO] [quiesce] unquiescing (%s): restarting %d stack(s)", reason, len(running))
l.restartAll(running)
if err := l.clearMarker(); err != nil {
l.logger.Printf("[ERROR] [quiesce] clear marker: %v", err)
}
}
defer unquiesce("deferred")
l.logger.Printf("[INFO] [quiesce] backup due on %d tier(s) — quiescing %d stack(s): %v",
len(tiers), len(running), running)
for _, st := range running {
if err := l.stacks.StopStack(st); err != nil {
l.logger.Printf("[ERROR] [quiesce] stop %s: %v (continuing)", st, err)
}
}
deadline := l.now().Add(l.maxQuiesce)
var firstErr error
// ONE window, N tiers, sequential. The app resumes only after the LAST tier snapshots.
for i, t := range tiers {
label := tierLabel(t.target)
last := i == len(tiers)-1
jobID, err := l.startBackupOn(ctx, t.target)
if err != nil {
l.logger.Printf("[ERROR] [quiesce] start backup on tier %s: %v", label, err)
if firstErr == nil {
firstErr = fmt.Errorf("start backup on %s: %w", label, err)
}
// A tier that will not start must not hold the app down for the others.
if last {
unquiesce("last tier failed to start")
}
continue
}
marker.JobID = jobID
_ = l.writeMarker(marker) // best-effort: record the CURRENT tier's job id for diagnosis
l.logger.Printf("[INFO] [quiesce] tier %s: backup job %s started — polling", label, jobID)
phase, stillRunning, perr := l.pollTier(ctx, t.target, jobID, label, deadline, last, &unquiesced, unquiesce)
if perr != nil && firstErr == nil {
firstErr = perr
}
if phase == phaseFailed {
l.logger.Printf("[WARN] [quiesce] tier %s: backup job %s failed", label, jobID)
}
if stillRunning {
// The max-quiesce guard fired while THIS tier's backup is still going (a first full
// offsite snapshot legitimately runs for hours). The app is already back up. We must
// NOT start the next tier: ONE BACKUP AT A TIME PER GUEST (operator ruling
// 2026-07-26) — vzdump still holds the guest lock, so a second start would be refused
// by the agent (409) or fail on the lock and record a spurious failure. The remaining
// tiers simply run on a later poll, once this one has finished.
l.logger.Printf("[INFO] [quiesce] tier %s still running past the quiesce bound — deferring %d remaining tier(s) to a later cycle",
label, len(tiers)-i-1)
break
}
}
// Belt: if every tier failed to start, nothing above unquiesced. The deferred call covers it,
// but doing it here keeps the "resume as soon as possible" property explicit.
unquiesce("cycle complete")
return firstErr
}
// pollTier polls ONE tier's job. It returns the terminal (or snapshotted-at-deadline) phase.
//
// The app is resumed ONLY when this is the LAST tier — that is what keeps every tier
// app-consistent while still costing exactly one stop/start pair. For a non-last tier the loop
// waits for a TERMINAL phase (done/failed), because vzdump holds the guest lock and the next tier
// cannot start until this one truly finishes.
// Returns (phase, stillRunning, err). stillRunning=true means the quiesce bound elapsed while the
// backup is STILL going — the caller must not start another tier (one backup at a time per guest).
func (l *Loop) pollTier(ctx context.Context, target, jobID, label string, deadline time.Time,
last bool, unquiesced *bool, unquiesce func(string)) (string, bool, error) {
for {
if !l.now().Before(deadline) {
l.logger.Printf("[WARN] [quiesce] max-quiesce-duration (%s) exceeded on tier %s (job %s) — unquiescing while the backup continues on the agent",
l.maxQuiesce, label, jobID)
unquiesce("max-quiesce guard")
return "", true, nil
}
phase, err := l.backupStatusOn(ctx, target)
if err != nil {
unquiesce("status poll failed")
return "", false, fmt.Errorf("poll backup status on %s: %w", label, err)
}
switch phase {
case phaseSnapshotted:
// 8B.2 early resume — but ONLY on the last tier. Resuming here on a non-last tier would
// leave the following tier capturing a running app, losing app-consistency for exactly
// the DR tier we most want it on.
if last && !*unquiesced {
l.logger.Printf("[INFO] [quiesce] tier %s: job %s snapshotted — resuming app early (8B.2)", label, jobID)
unquiesce("snapshotted (early resume, last tier)")
}
case phaseDone:
if last {
l.logger.Printf("[INFO] [quiesce] tier %s: backup job %s done", label, jobID)
unquiesce("backup done")
} else {
l.logger.Printf("[INFO] [quiesce] tier %s: backup job %s done — next tier may start (app still quiesced)", label, jobID)
}
return phaseDone, false, nil
case phaseFailed:
if last {
unquiesce("backup failed")
}
return phaseFailed, false, nil
}
select {
case <-ctx.Done():
unquiesce("controller shutting down")
return "", false, ctx.Err()
case <-time.After(l.statusPoll):
}
}
}
// ---- window gate (Part 3, v0.168.0) -----------------------------------------------------
var (
quiesceBudapest *time.Location
quiesceBudapestOnce sync.Once
)
func budapestLocation() *time.Location {
quiesceBudapestOnce.Do(func() {
loc, err := time.LoadLocation("Europe/Budapest")
if err != nil {
quiesceBudapest = time.UTC
return
}
quiesceBudapest = loc
})
return quiesceBudapest
}
const (
gateOpenOffsetMin = 120 // gate opens at W+2h
gateSpanMin = 240 // 4h span → [W+2h, W+6h)
)
// scheduledRunAllowed decides whether a DUE, scheduled whole-guest backup may run at `now` (passed by
// the caller as Budapest wall-clock — only its hour/minute are read). True when now is inside the gate
// window [W+2h, W+6h); otherwise true ONLY if the safety valve holds — the newest successful backup is
// missing (nil) or older than cadence+24h — so a box powered on only outside its window never starves.
// An unparseable window fails OPEN (allow) rather than block backups forever.
func scheduledRunAllowed(now time.Time, windowStart string, lastAgeSecs *int64, cadence time.Duration) bool {
startMin, err := backupwindow.ParseHHMM(windowStart)
if err != nil {
return true
}
nowMin := now.Hour()*60 + now.Minute()
if within(nowMin, mod1440(startMin+gateOpenOffsetMin), gateSpanMin) {
return true
}
// Outside the window: only the safety valve may run it.
if lastAgeSecs == nil {
return true // no recorded backup yet — never withhold the first one
}
return time.Duration(*lastAgeSecs)*time.Second > cadence+24*time.Hour
}
// gateBounds returns the gate window [W+2h, W+6h) as HH:MM for the deferral log line.
func gateBounds(windowStart string) (from, to string) {
return backupwindow.GateWindow(windowStart)
}
func mod1440(m int) int { return ((m % 1440) + 1440) % 1440 }
// within reports whether minute-of-day p falls in [start, start+span) modulo 24h (wrap-safe).
func within(p, start, span int) bool {
return mod1440(p-start) < span
}
func (l *Loop) restartAll(stacks []string) {
for _, s := range stacks {
if err := l.stacks.StartStack(s); err != nil {
l.logger.Printf("[ERROR] [quiesce] restart %s: %v", s, err)
}
}
}
// ---- marker persistence (atomic, 0600) --------------------------------------------------
func (l *Loop) writeMarker(m Marker) error {
m.Active = true
data, err := json.MarshalIndent(m, "", " ")
if err != nil {
return err
}
if err := os.MkdirAll(filepath.Dir(l.markerPath), 0o755); err != nil {
return err
}
tmp := l.markerPath + ".tmp"
if err := os.WriteFile(tmp, data, 0o600); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, l.markerPath)
}
func (l *Loop) readMarker() (Marker, bool) {
data, err := os.ReadFile(l.markerPath)
if err != nil {
return Marker{}, false
}
var m Marker
if err := json.Unmarshal(data, &m); err != nil {
// S3: a corrupt marker is NOT silently dropped — log it LOUD and quarantine the bad file (a real
// corrupted-mid-quiesce marker would otherwise skip stack-recovery with no trace). Still return
// false: "no usable marker" ⇒ no recovery is the correct contract.
l.logger.Printf("[WARN] [quiesce] marker at %s is corrupt (%v) — quarantining; stacks not auto-recovered from it", l.markerPath, err)
_ = os.Rename(l.markerPath, fmt.Sprintf("%s.corrupt-%d", l.markerPath, l.now().Unix()))
return Marker{}, false
}
return m, true
}
func (l *Loop) clearMarker() error {
err := os.Remove(l.markerPath)
if os.IsNotExist(err) {
return nil
}
return err
}