R-894: after a restart the agent remembers the last backup per tier
gates / gates (push) Successful in 35s

An unreadable storage right after an agent restart read the off-site tier
DUE (the in-memory record was empty). The newest success per tier is now
kept on disk and read ONLY when the storage cannot be read: fresh -> not
due, older than the cadence -> due, none -> due (unknown) as before. A
storage that answers stays the ground truth.

Ships with v0.150.0 after the 2026-10-07 read-back; nothing delivered.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-06 20:33:27 +02:00
parent 7e82f325b8
commit 74b5eae5b0
9 changed files with 492 additions and 28 deletions
+144
View File
@@ -0,0 +1,144 @@
package localapi
import (
"context"
"io"
"log/slog"
"net/http"
"path/filepath"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/backup"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
)
// R-894 — after an agent restart, an UNREADABLE storage must fall back to the last success saved on
// disk, not to "never". Measured 2026-10-05 on demo-hp: a restart at 04:57, the off-site storage
// unreachable at 06:25, the 7-day tier (last copy 4 days old) read DUE, vzdump failed.
//
// Every test here builds a NEW server and a NEW BackupSuccessState from the same file — that is the
// restart. The in-memory store (fakeStore) is always fresh, as after a real restart.
// r894Server builds a two-tier server whose off-site tier answers the storage listing with lister.
func r894Server(t *testing.T, path string, pbsSvc BackupService) *Server {
t.Helper()
srv, err := NewServer(Options{
ListenAddr: "127.0.0.1:0", Guests: &fakeGuests{}, Backups: &fakeBackups{}, Store: &fakeStore{},
Storage: fakeStorage{targets: []hub.StorageTarget{{Name: "local"}, {Name: "felhom-pbs"}}},
Tokens: staticTokens{"A": 8200},
BackupTiers: []BackupTier{
{TargetID: "local", Cadence: 24 * time.Hour, Primary: true, Service: &fakeBackups{}},
{TargetID: "felhom-pbs", Cadence: 7 * 24 * time.Hour, Service: pbsSvc},
},
LastKnownBackups: backup.NewBackupSuccessState(path),
Logger: slog.New(slog.NewTextHandler(io.Discard, nil)),
})
if err != nil {
t.Fatal(err)
}
srv.baseCtx = context.Background()
srv.now = func() time.Time { return testNow }
return srv
}
// unreadable is the off-site storage as demo-hp saw it: "Can't connect to 10.77.0.1:8007".
func unreadable() archiveLister {
return archiveLister{fakeBackups: &fakeBackups{}, err: errStorageRead}
}
// THE R-894 CASE, end to end. Agent 1 takes an off-site backup through POST /backup (the fake
// runner's success is 12 h before testNow). The agent restarts. The storage cannot be read. The tier
// must read NOT due, from the copy saved on disk.
//
// COMPANION RED-PROOF (observed): delete the `lookup == archiveUnknown && s.lastKnown != nil` block in
// handleBackupDue → this fails with "after a restart an unreadable storage must fall back to the saved
// copy (12 h old, 7-day tier) — NOT due; got {… Due:true … AgeState:unknown …}". Restored.
func TestBackupDue_R894_RestartThenUnreadableStorage_FreshSavedCopyIsNotDue(t *testing.T) {
path := filepath.Join(t.TempDir(), "backup-success-state.json")
// Agent 1: a real backup job through the endpoint the controller calls.
first := r894Server(t, path, &fakeBackups{})
if rr := do(t, first.Handler(), "POST", "/backup?target=felhom-pbs", "A", ""); rr.Code != http.StatusAccepted {
t.Fatalf("POST /backup: %d %s", rr.Code, rr.Body.String())
}
waitFor(t, func() bool {
_, ok := backup.NewBackupSuccessState(path).LastKnownSuccess("felhom-pbs", 8200)
return ok
})
// Agent 2: a restart (new server, new state from the same file), and the storage is unreachable.
got := dueFor(t, r894Server(t, path, unreadable()).Handler(), "felhom-pbs")
if got.Due {
t.Fatalf("after a restart an unreadable storage must fall back to the saved copy (12 h old, 7-day tier) — NOT due; got %+v", got)
}
if got.AgeState != AgeStateKnown || got.AgeSecs == nil || *got.AgeSecs != int64((12*time.Hour).Seconds()) {
t.Fatalf("the age must come from the saved copy (12 h, known); got %+v", got)
}
}
// The deliberate rule stays: an unreadable storage must not suppress a backup that IS due. A saved
// copy older than the cadence reads DUE.
//
// COMPANION RED-PROOF (observed): make the fallback answer not-due whenever a saved copy exists
// (`if fromDisk { …Due:false… }` before the cadence check) → this fails with "a saved copy 9 days old
// under a 7-day cadence MUST read due". Restored.
func TestBackupDue_R894_RestartThenUnreadableStorage_OldSavedCopyIsDue(t *testing.T) {
path := filepath.Join(t.TempDir(), "backup-success-state.json")
st := backup.NewBackupSuccessState(path)
if err := st.RecordBackupSuccess("felhom-pbs", backupAt("felhom-pbs", 8200, 9*24*time.Hour, true)); err != nil {
t.Fatal(err)
}
got := dueFor(t, r894Server(t, path, unreadable()).Handler(), "felhom-pbs")
if !got.Due {
t.Fatalf("a saved copy 9 days old under a 7-day cadence MUST read due; got %+v", got)
}
if got.AgeState != AgeStateKnown {
t.Fatalf("the age is known (from disk); got %+v", got)
}
}
// No saved copy → the pre-R-894 answer, byte for byte: DUE, age UNKNOWN (never ABSENT — the controller
// fires its window-gate valve only on absent, R-88).
func TestBackupDue_R894_RestartThenUnreadableStorage_NoSavedCopyIsDueUnknown(t *testing.T) {
path := filepath.Join(t.TempDir(), "backup-success-state.json")
got := dueFor(t, r894Server(t, path, unreadable()).Handler(), "felhom-pbs")
if !got.Due || got.AgeState != AgeStateUnknown || got.AgeSecs != nil {
t.Fatalf("no saved copy + unreadable storage must stay DUE with age unknown; got %+v", got)
}
}
// A storage that ANSWERS is the ground truth: an archive absent there makes the tier due even when the
// file remembers a fresh success (a pruned or deleted copy must be made again).
//
// COMPANION RED-PROOF (observed): drop `lookup == archiveUnknown &&` from the fallback condition → this
// fails with "the storage answered 'no archive' — the saved copy must NOT stand in for it". Restored.
func TestBackupDue_R894_SavedCopyIgnoredWhenStorageAnswers(t *testing.T) {
path := filepath.Join(t.TempDir(), "backup-success-state.json")
st := backup.NewBackupSuccessState(path)
if err := st.RecordBackupSuccess("felhom-pbs", backupAt("felhom-pbs", 8200, time.Hour, true)); err != nil {
t.Fatal(err)
}
absent := archiveLister{fakeBackups: &fakeBackups{}, found: false}
got := dueFor(t, r894Server(t, path, absent).Handler(), "felhom-pbs")
if !got.Due {
t.Fatalf("the storage answered 'no archive' — the saved copy must NOT stand in for it; got %+v", got)
}
}
// A FAILED backup is never saved: it must not make a tier look fresh after a restart.
func TestBackupDue_R894_FailedBackupIsNotSaved(t *testing.T) {
path := filepath.Join(t.TempDir(), "backup-success-state.json")
first := r894Server(t, path, &fakeBackups{failErr: "could not activate storage 'felhom-pbs'"})
if rr := do(t, first.Handler(), "POST", "/backup?target=felhom-pbs", "A", ""); rr.Code != http.StatusAccepted {
t.Fatalf("POST /backup: %d %s", rr.Code, rr.Body.String())
}
waitFor(t, func() bool { return len(first.store.Backups(context.Background())) == 1 })
if _, ok := backup.NewBackupSuccessState(path).LastKnownSuccess("felhom-pbs", 8200); ok {
t.Fatal("a failed backup must never be saved as a success")
}
got := dueFor(t, r894Server(t, path, unreadable()).Handler(), "felhom-pbs")
if !got.Due {
t.Fatalf("after a failed backup and a restart the tier must still be due; got %+v", got)
}
}
+61 -22
View File
@@ -85,6 +85,13 @@ type BackupStore interface {
RestoreTests(ctx context.Context) []hub.RestoreTest
}
// LastKnownBackupStore (R-894) is the on-disk newest-success-per-tier record. Satisfied by
// *backup.BackupSuccessState.
type LastKnownBackupStore interface {
RecordBackupSuccess(target string, b hub.Backup) error
LastKnownSuccess(target string, vmid int) (time.Time, bool)
}
// StorageView yields the host's observed storage targets (for mapping a mount's storage id →
// fast/slow class). Satisfied by *storage.Observer.
type StorageView interface {
@@ -164,6 +171,10 @@ type Options struct {
// PRIMARY tier, inside the backup goroutine and BEFORE the host-wide heavy-op gate is released — so the OS leg
// that it starts can never overlap another backup or a restore-test (`11` C10). OPTIONAL — nil → nothing runs.
AfterPrimaryBackup func(ctx context.Context, vmid int)
// LastKnownBackups (R-894) keeps the newest successful backup per tier ON DISK, so the due-check's
// fallback for an UNREADABLE storage after an agent restart is the last known copy, not "never".
// nil = the pre-R-894 behaviour (in-memory record only).
LastKnownBackups LastKnownBackupStore
// Privileged runs the fenced root wrappers (E-2a: felhom-backup-target-apply). OPTIONAL — when
// nil, POST /backup/target reports "not configured". Satisfied by *proxmox.ExecRunner.
Privileged PrivilegedRunner
@@ -229,7 +240,6 @@ type Options struct {
// POST /escrow/recover-offsite-password. OPTIONAL — nil → that route reports "not configured"
// (503) instead of failing obscurely. Satisfied by escrow.OffsiteKeyRecoverer.
EscrowRecovery EscrowRecoverer
}
// defaultBackupCadence is the fallback /backup/due window when none is configured.
@@ -279,21 +289,22 @@ type Server struct {
// is the pre-R-82 shape.
tiers []BackupTier
// inFlight (R-85) is shared with the restore-test scheduler so the two never run together.
inFlight *backup.InFlight
inFlight *backup.InFlight
afterPrimaryBackup func(ctx context.Context, vmid int) // the OS leg (agent v0.140.0); nil = none
logger *slog.Logger
now func() time.Time
lastKnown LastKnownBackupStore // R-894: on-disk newest success per tier; nil = none
logger *slog.Logger
now func() time.Time
disks DiskOps // slice 8C (optional)
diskGate StorageGate // slice 8C (optional)
guestList GuestLister // slice 8C (optional)
guestAttach GuestAttacher // slice 10 P2 (optional)
mem MemoryOps // v0.90.0 R-24 guest RAM resize (optional)
memMu sync.Mutex // single-flight around a resize apply (one customer per host)
netStorage NetworkStorageOps // Part A1: NAS network mounts (optional)
netMountRoot string // the user-data namespace root for the network-mount role gate
smbCredsDir string // where SMB creds files are written (out-of-band, 0600)
escrowStagePath string // fork-4: 0600 staging file for the pushed restic repo password
disks DiskOps // slice 8C (optional)
diskGate StorageGate // slice 8C (optional)
guestList GuestLister // slice 8C (optional)
guestAttach GuestAttacher // slice 10 P2 (optional)
mem MemoryOps // v0.90.0 R-24 guest RAM resize (optional)
memMu sync.Mutex // single-flight around a resize apply (one customer per host)
netStorage NetworkStorageOps // Part A1: NAS network mounts (optional)
netMountRoot string // the user-data namespace root for the network-mount role gate
smbCredsDir string // where SMB creds files are written (out-of-band, 0600)
escrowStagePath string // fork-4: 0600 staging file for the pushed restic repo password
// crashGuardStatePath (R-856) is the host crash guard's state file read by GET /host/crash-guard;
// empty = defaultCrashGuardStatePath. A seam: tests point it at a fixture.
crashGuardStatePath string
@@ -301,11 +312,11 @@ type Server struct {
// identity blob from the hub, unseal it with the customer's recovery code, return ONLY the
// offsite repository password. OPTIONAL — nil (no hub client configured) makes
// POST /escrow/recover-offsite-password answer 503 rather than pretending.
escrowRecovery EscrowRecoverer
intent IntentRecorder // slice 10 P3 (optional)
guestBinds *GuestBindStore // F9 startup bind re-assert record (optional)
formatJobs *FormatJobStore // F20-BUG3 detached-format job record (optional)
staleLock StaleLockController // F2-b startup stale-lock recovery (optional)
escrowRecovery EscrowRecoverer
intent IntentRecorder // slice 10 P3 (optional)
guestBinds *GuestBindStore // F9 startup bind re-assert record (optional)
formatJobs *FormatJobStore // F20-BUG3 detached-format job record (optional)
staleLock StaleLockController // F2-b startup stale-lock recovery (optional)
// guestPower (F-REBOOT) is per-guest start-attempt state for the guest-power watchdog.
// Guarded by guestPowerMu in guestpower.go; in-memory on purpose (see guestPowerState).
guestPower map[int]guestPowerState
@@ -478,6 +489,7 @@ func NewServer(o Options) (*Server, error) {
s.tiers = normalizeBackupTiers(o.BackupTiers, o.Backups, cadence)
s.inFlight = o.InFlight
s.afterPrimaryBackup = o.AfterPrimaryBackup
s.lastKnown = o.LastKnownBackups
if s.backups == nil && len(s.tiers) > 0 {
s.backups = s.tiers[0].Service
}
@@ -905,6 +917,13 @@ func (s *Server) handleBackup(w http.ResponseWriter, r *http.Request, vmid int)
s.logger.Info("local-api: backup job complete", "vmid", vmid, "target", tier.TargetID, "job", jobID, "archive", b.Archive)
}
s.store.RecordBackup(b)
if b.Success && s.lastKnown != nil {
if err := s.lastKnown.RecordBackupSuccess(tier.TargetID, b); err != nil {
// Not fatal: the backup exists. Only the fallback after a restart loses this copy.
s.logger.Warn("local-api: could not save the backup on disk for the due-check fallback (R-894)",
"vmid", vmid, "target", tier.TargetID, "err", err)
}
}
s.finishJob(key, jobID, b)
// OS leg (agent v0.140.0): after the night's whole-guest copy exists, still holding the heavy-op gate.
if b.Success && tier.Primary && s.afterPrimaryBackup != nil {
@@ -1067,6 +1086,20 @@ func (s *Server) handleBackupDue(w http.ResponseWriter, r *http.Request, vmid in
newest, haveNewest = t, true
unparseable = false // ground truth supersedes an unreadable in-memory timestamp
}
// R-894: the storage could not be read → the last success saved ON DISK stands in for the in-memory
// record a restart emptied. ONLY on archiveUnknown: a storage that answers is the ground truth, and an
// archive absent there must make the tier due even when the file remembers one (a pruned copy).
// A saved copy older than the cadence still reads DUE below — an unreadable storage never suppresses
// a backup that is due.
fromDisk := false
if lookup == archiveUnknown && s.lastKnown != nil {
if saved, ok := s.lastKnown.LastKnownSuccess(tier.TargetID, vmid); ok && (!haveNewest || saved.After(newest)) {
newest, haveNewest, fromDisk = saved, true, true
unparseable = false
s.logger.Info("local-api: backup storage unreadable — due-check uses the last success saved on disk (R-894)",
"vmid", vmid, "target", tier.TargetID, "saved", saved.UTC().Format(time.RFC3339))
}
}
if !haveNewest {
// R-88 Part 2: THREE distinct reasons for a nil age, each with its own state. Only ABSENT is a
// positive claim of "never backed up"; only that one may license the controller to bypass its
@@ -1089,13 +1122,17 @@ func (s *Server) handleBackupDue(w http.ResponseWriter, r *http.Request, vmid in
}
age := s.now().Sub(newest)
ageSecs := int64(age.Seconds())
suffix := ""
if fromDisk {
suffix = " (storage unreadable — age from the last success saved on disk)"
}
if age >= tier.Cadence {
writeOK(w, BackupDueResponse{VMID: vmid, Due: true, AgeSecs: &ageSecs, AgeState: AgeStateKnown,
Reason: "older than cadence", Target: echo})
Reason: "older than cadence" + suffix, Target: echo})
return
}
writeOK(w, BackupDueResponse{VMID: vmid, Due: false, AgeSecs: &ageSecs, AgeState: AgeStateKnown,
Reason: "within cadence window", Target: echo})
Reason: "within cadence window" + suffix, Target: echo})
}
// BackupTiersResponse is GET /backup/tiers (R-82): the tiers this agent serves, primary first.
@@ -1483,4 +1520,6 @@ func writeStatus(w http.ResponseWriter, code int, ok bool, data any, errMsg stri
}
// SetAfterPrimaryBackup wires the hook that runs after a successful primary-tier backup (the OS leg, agent v0.140.0).
func (s *Server) SetAfterPrimaryBackup(fn func(ctx context.Context, vmid int)) { s.afterPrimaryBackup = fn }
func (s *Server) SetAfterPrimaryBackup(fn func(ctx context.Context, vmid int)) {
s.afterPrimaryBackup = fn
}