R-894: after a restart the agent remembers the last backup per tier
gates / gates (push) Successful in 35s

An unreadable storage right after an agent restart read the off-site tier
DUE (the in-memory record was empty). The newest success per tier is now
kept on disk and read ONLY when the storage cannot be read: fresh -> not
due, older than the cadence -> due, none -> due (unknown) as before. A
storage that answers stays the ground truth.

Ships with v0.150.0 after the 2026-10-07 read-back; nothing delivered.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-06 20:33:27 +02:00
parent 7e82f325b8
commit 74b5eae5b0
9 changed files with 492 additions and 28 deletions
+131
View File
@@ -0,0 +1,131 @@
package backup
import (
"encoding/json"
"os"
"path/filepath"
"sort"
"strconv"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
)
// BackupSuccessState persists the newest SUCCESSFUL whole-guest backup per tier and guest (R-894).
//
// Why it exists. The due-check (`localapi` handleBackupDue) asks the tier's storage when a backup last
// landed (R-84) and falls back to the in-memory record when the storage cannot be read. The in-memory
// record is empty after an agent restart (Store, R-348), so "storage unreadable" right after a restart
// read as "no record — DUE". Measured 2026-10-05 on demo-hp: the agent restarted at 04:57, the off-site
// storage answered "Can't connect" at 06:25, the 7-day tier — last copy 2026-10-01 — read DUE, the
// controller asked, and vzdump failed. This file is the last known copy the fallback reads instead.
//
// It is read ONLY when the storage cannot be read. A storage that answers is the ground truth and wins,
// in both directions: an archive found there counts, and an archive absent there is absent even when
// this file remembers a success (a pruned or deleted archive must make the tier due — the same reason
// R-84 chose the storage over a persisted record). Pinned by
// TestBackupDue_R894_SavedCopyIgnoredWhenStorageAnswers.
//
// Only SUCCESSES are written (the RestoreTestState rule): a failure must stay due and be retried, so a
// record of a failure has no reader.
type BackupSuccessState struct {
path string
mu sync.Mutex
last map[string]savedSuccess // key(target, vmid) → the newest success
}
type savedSuccess struct {
target string
vmid int
at time.Time
}
// backupSuccessJSON is one entry on disk.
type backupSuccessJSON struct {
Target string `json:"target"`
VMID int `json:"vmid"`
StartedAt string `json:"started_at"`
}
func backupStateKey(target string, vmid int) string { return target + "/" + strconv.Itoa(vmid) }
// NewBackupSuccessState opens (or creates) the state at path. A missing or unreadable file degrades to
// "nothing known" — the pre-R-894 behaviour, which is DUE — and never wedges the daemon.
func NewBackupSuccessState(path string) *BackupSuccessState {
s := &BackupSuccessState{path: path, last: map[string]savedSuccess{}}
data, err := os.ReadFile(path)
if err != nil {
return s
}
var entries []backupSuccessJSON
if json.Unmarshal(data, &entries) != nil {
return s
}
for _, e := range entries {
t, perr := time.Parse(time.RFC3339, e.StartedAt)
if perr != nil {
continue // one unreadable entry must not lose the others
}
s.last[backupStateKey(e.Target, e.VMID)] = savedSuccess{target: e.Target, vmid: e.VMID, at: t.UTC()}
}
return s
}
// RecordBackupSuccess saves b when it is a success newer than the one on file. target is the tier the
// job ran on (the due-check's key); a failure or an unparseable time is ignored.
func (s *BackupSuccessState) RecordBackupSuccess(target string, b hub.Backup) error {
if s == nil || !b.Success {
return nil
}
t, err := time.Parse(time.RFC3339, b.StartedAt)
if err != nil {
return nil
}
s.mu.Lock()
defer s.mu.Unlock()
k := backupStateKey(target, b.VMID)
if old, ok := s.last[k]; ok && !t.After(old.at) {
return nil
}
s.last[k] = savedSuccess{target: target, vmid: b.VMID, at: t.UTC()}
return s.saveLocked()
}
// LastKnownSuccess returns the newest saved success for this tier and guest (ok=false = none on file).
func (s *BackupSuccessState) LastKnownSuccess(target string, vmid int) (time.Time, bool) {
if s == nil {
return time.Time{}, false
}
s.mu.Lock()
defer s.mu.Unlock()
e, ok := s.last[backupStateKey(target, vmid)]
return e.at, ok
}
func (s *BackupSuccessState) saveLocked() error {
entries := make([]backupSuccessJSON, 0, len(s.last))
for _, e := range s.last {
entries = append(entries, backupSuccessJSON{Target: e.target, VMID: e.vmid, StartedAt: e.at.Format(time.RFC3339)})
}
// Deterministic file content (Go's map order is random).
sort.Slice(entries, func(i, j int) bool {
if entries[i].Target != entries[j].Target {
return entries[i].Target < entries[j].Target
}
return entries[i].VMID < entries[j].VMID
})
data, err := json.MarshalIndent(entries, "", " ")
if err != nil {
return err
}
if err := os.MkdirAll(filepath.Dir(s.path), 0o755); err != nil {
return err
}
tmp := s.path + ".tmp"
if err := os.WriteFile(tmp, data, 0o600); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, s.path)
}
+59
View File
@@ -0,0 +1,59 @@
package backup
import (
"os"
"path/filepath"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
)
// R-894: the on-disk newest success per tier survives a restart (a new state from the same file).
func TestBackupSuccessState_SurvivesRestart(t *testing.T) {
path := filepath.Join(t.TempDir(), "backup-success-state.json")
s := NewBackupSuccessState(path)
at := time.Date(2026, 10, 1, 20, 15, 0, 0, time.UTC)
if err := s.RecordBackupSuccess("felhom-pbs", hub.Backup{VMID: 9201, Success: true, StartedAt: at.Format(time.RFC3339)}); err != nil {
t.Fatal(err)
}
got, ok := NewBackupSuccessState(path).LastKnownSuccess("felhom-pbs", 9201)
if !ok || !got.Equal(at) {
t.Fatalf("after a restart the saved copy must read back; got %v ok=%v", got, ok)
}
if _, ok := NewBackupSuccessState(path).LastKnownSuccess("local", 9201); ok {
t.Fatal("another tier must not borrow this tier's copy")
}
}
// Only a NEWER success replaces the saved one; failures and unparseable times are ignored.
func TestBackupSuccessState_KeepsNewestSuccessOnly(t *testing.T) {
path := filepath.Join(t.TempDir(), "s.json")
s := NewBackupSuccessState(path)
newer := time.Date(2026, 10, 5, 0, 0, 0, 0, time.UTC)
older := newer.Add(-48 * time.Hour)
for _, b := range []hub.Backup{
{VMID: 1, Success: true, StartedAt: newer.Format(time.RFC3339)},
{VMID: 1, Success: true, StartedAt: older.Format(time.RFC3339)}, // older: ignored
{VMID: 1, Success: false, StartedAt: newer.Add(time.Hour).Format(time.RFC3339)}, // failure: ignored
{VMID: 1, Success: true, StartedAt: "not-a-time"}, // unparseable: ignored
} {
if err := s.RecordBackupSuccess("t", b); err != nil {
t.Fatal(err)
}
}
if got, _ := NewBackupSuccessState(path).LastKnownSuccess("t", 1); !got.Equal(newer) {
t.Fatalf("want the newest success %v, got %v", newer, got)
}
}
// A corrupt file degrades to "nothing known" (the pre-R-894 DUE answer), never a crash.
func TestBackupSuccessState_CorruptFileIsEmpty(t *testing.T) {
path := filepath.Join(t.TempDir(), "s.json")
if err := os.WriteFile(path, []byte("{not json"), 0o600); err != nil {
t.Fatal(err)
}
if _, ok := NewBackupSuccessState(path).LastKnownSuccess("t", 1); ok {
t.Fatal("a corrupt file must read as nothing known")
}
}
+3
View File
@@ -30,6 +30,9 @@ import (
// 2026-08-20, two consecutive host-reports with `0 backups` while `pvesm list` showed archives on both tiers. What
// is unaffected is the hub's VERDICT: it looks back 7 days over stored reports (felhom.eu hub/internal/monitor/
// deadline.go backupEvidenceLookback) and the storage stays the ground truth (R-84).
// The due-check's fallback for an UNREADABLE storage no longer reads this store alone (R-894): the newest
// success per tier is also on disk (BackupSuccessState), so a restart followed by an unreachable storage
// reads the last known copy, not "never".
type Store struct {
mu sync.Mutex
byTarget map[string]hub.Backup // latest backup per target id