R-894: after a restart the agent remembers the last backup per tier
gates / gates (push) Successful in 35s
gates / gates (push) Successful in 35s
An unreadable storage right after an agent restart read the off-site tier DUE (the in-memory record was empty). The newest success per tier is now kept on disk and read ONLY when the storage cannot be read: fresh -> not due, older than the cadence -> due, none -> due (unknown) as before. A storage that answers stays the ground truth. Ships with v0.150.0 after the 2026-10-07 read-back; nothing delivered. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -0,0 +1,131 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
||||
)
|
||||
|
||||
// BackupSuccessState persists the newest SUCCESSFUL whole-guest backup per tier and guest (R-894).
|
||||
//
|
||||
// Why it exists. The due-check (`localapi` handleBackupDue) asks the tier's storage when a backup last
|
||||
// landed (R-84) and falls back to the in-memory record when the storage cannot be read. The in-memory
|
||||
// record is empty after an agent restart (Store, R-348), so "storage unreadable" right after a restart
|
||||
// read as "no record — DUE". Measured 2026-10-05 on demo-hp: the agent restarted at 04:57, the off-site
|
||||
// storage answered "Can't connect" at 06:25, the 7-day tier — last copy 2026-10-01 — read DUE, the
|
||||
// controller asked, and vzdump failed. This file is the last known copy the fallback reads instead.
|
||||
//
|
||||
// It is read ONLY when the storage cannot be read. A storage that answers is the ground truth and wins,
|
||||
// in both directions: an archive found there counts, and an archive absent there is absent even when
|
||||
// this file remembers a success (a pruned or deleted archive must make the tier due — the same reason
|
||||
// R-84 chose the storage over a persisted record). Pinned by
|
||||
// TestBackupDue_R894_SavedCopyIgnoredWhenStorageAnswers.
|
||||
//
|
||||
// Only SUCCESSES are written (the RestoreTestState rule): a failure must stay due and be retried, so a
|
||||
// record of a failure has no reader.
|
||||
type BackupSuccessState struct {
|
||||
path string
|
||||
mu sync.Mutex
|
||||
last map[string]savedSuccess // key(target, vmid) → the newest success
|
||||
}
|
||||
|
||||
type savedSuccess struct {
|
||||
target string
|
||||
vmid int
|
||||
at time.Time
|
||||
}
|
||||
|
||||
// backupSuccessJSON is one entry on disk.
|
||||
type backupSuccessJSON struct {
|
||||
Target string `json:"target"`
|
||||
VMID int `json:"vmid"`
|
||||
StartedAt string `json:"started_at"`
|
||||
}
|
||||
|
||||
func backupStateKey(target string, vmid int) string { return target + "/" + strconv.Itoa(vmid) }
|
||||
|
||||
// NewBackupSuccessState opens (or creates) the state at path. A missing or unreadable file degrades to
|
||||
// "nothing known" — the pre-R-894 behaviour, which is DUE — and never wedges the daemon.
|
||||
func NewBackupSuccessState(path string) *BackupSuccessState {
|
||||
s := &BackupSuccessState{path: path, last: map[string]savedSuccess{}}
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return s
|
||||
}
|
||||
var entries []backupSuccessJSON
|
||||
if json.Unmarshal(data, &entries) != nil {
|
||||
return s
|
||||
}
|
||||
for _, e := range entries {
|
||||
t, perr := time.Parse(time.RFC3339, e.StartedAt)
|
||||
if perr != nil {
|
||||
continue // one unreadable entry must not lose the others
|
||||
}
|
||||
s.last[backupStateKey(e.Target, e.VMID)] = savedSuccess{target: e.Target, vmid: e.VMID, at: t.UTC()}
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// RecordBackupSuccess saves b when it is a success newer than the one on file. target is the tier the
|
||||
// job ran on (the due-check's key); a failure or an unparseable time is ignored.
|
||||
func (s *BackupSuccessState) RecordBackupSuccess(target string, b hub.Backup) error {
|
||||
if s == nil || !b.Success {
|
||||
return nil
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339, b.StartedAt)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
k := backupStateKey(target, b.VMID)
|
||||
if old, ok := s.last[k]; ok && !t.After(old.at) {
|
||||
return nil
|
||||
}
|
||||
s.last[k] = savedSuccess{target: target, vmid: b.VMID, at: t.UTC()}
|
||||
return s.saveLocked()
|
||||
}
|
||||
|
||||
// LastKnownSuccess returns the newest saved success for this tier and guest (ok=false = none on file).
|
||||
func (s *BackupSuccessState) LastKnownSuccess(target string, vmid int) (time.Time, bool) {
|
||||
if s == nil {
|
||||
return time.Time{}, false
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
e, ok := s.last[backupStateKey(target, vmid)]
|
||||
return e.at, ok
|
||||
}
|
||||
|
||||
func (s *BackupSuccessState) saveLocked() error {
|
||||
entries := make([]backupSuccessJSON, 0, len(s.last))
|
||||
for _, e := range s.last {
|
||||
entries = append(entries, backupSuccessJSON{Target: e.target, VMID: e.vmid, StartedAt: e.at.Format(time.RFC3339)})
|
||||
}
|
||||
// Deterministic file content (Go's map order is random).
|
||||
sort.Slice(entries, func(i, j int) bool {
|
||||
if entries[i].Target != entries[j].Target {
|
||||
return entries[i].Target < entries[j].Target
|
||||
}
|
||||
return entries[i].VMID < entries[j].VMID
|
||||
})
|
||||
data, err := json.MarshalIndent(entries, "", " ")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(s.path), 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
tmp := s.path + ".tmp"
|
||||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||||
os.Remove(tmp)
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp, s.path)
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
||||
)
|
||||
|
||||
// R-894: the on-disk newest success per tier survives a restart (a new state from the same file).
|
||||
func TestBackupSuccessState_SurvivesRestart(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "backup-success-state.json")
|
||||
s := NewBackupSuccessState(path)
|
||||
at := time.Date(2026, 10, 1, 20, 15, 0, 0, time.UTC)
|
||||
if err := s.RecordBackupSuccess("felhom-pbs", hub.Backup{VMID: 9201, Success: true, StartedAt: at.Format(time.RFC3339)}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, ok := NewBackupSuccessState(path).LastKnownSuccess("felhom-pbs", 9201)
|
||||
if !ok || !got.Equal(at) {
|
||||
t.Fatalf("after a restart the saved copy must read back; got %v ok=%v", got, ok)
|
||||
}
|
||||
if _, ok := NewBackupSuccessState(path).LastKnownSuccess("local", 9201); ok {
|
||||
t.Fatal("another tier must not borrow this tier's copy")
|
||||
}
|
||||
}
|
||||
|
||||
// Only a NEWER success replaces the saved one; failures and unparseable times are ignored.
|
||||
func TestBackupSuccessState_KeepsNewestSuccessOnly(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "s.json")
|
||||
s := NewBackupSuccessState(path)
|
||||
newer := time.Date(2026, 10, 5, 0, 0, 0, 0, time.UTC)
|
||||
older := newer.Add(-48 * time.Hour)
|
||||
for _, b := range []hub.Backup{
|
||||
{VMID: 1, Success: true, StartedAt: newer.Format(time.RFC3339)},
|
||||
{VMID: 1, Success: true, StartedAt: older.Format(time.RFC3339)}, // older: ignored
|
||||
{VMID: 1, Success: false, StartedAt: newer.Add(time.Hour).Format(time.RFC3339)}, // failure: ignored
|
||||
{VMID: 1, Success: true, StartedAt: "not-a-time"}, // unparseable: ignored
|
||||
} {
|
||||
if err := s.RecordBackupSuccess("t", b); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if got, _ := NewBackupSuccessState(path).LastKnownSuccess("t", 1); !got.Equal(newer) {
|
||||
t.Fatalf("want the newest success %v, got %v", newer, got)
|
||||
}
|
||||
}
|
||||
|
||||
// A corrupt file degrades to "nothing known" (the pre-R-894 DUE answer), never a crash.
|
||||
func TestBackupSuccessState_CorruptFileIsEmpty(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "s.json")
|
||||
if err := os.WriteFile(path, []byte("{not json"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, ok := NewBackupSuccessState(path).LastKnownSuccess("t", 1); ok {
|
||||
t.Fatal("a corrupt file must read as nothing known")
|
||||
}
|
||||
}
|
||||
@@ -30,6 +30,9 @@ import (
|
||||
// 2026-08-20, two consecutive host-reports with `0 backups` while `pvesm list` showed archives on both tiers. What
|
||||
// is unaffected is the hub's VERDICT: it looks back 7 days over stored reports (felhom.eu hub/internal/monitor/
|
||||
// deadline.go backupEvidenceLookback) and the storage stays the ground truth (R-84).
|
||||
// The due-check's fallback for an UNREADABLE storage no longer reads this store alone (R-894): the newest
|
||||
// success per tier is also on disk (BackupSuccessState), so a restart followed by an unreachable storage
|
||||
// reads the last known copy, not "never".
|
||||
type Store struct {
|
||||
mu sync.Mutex
|
||||
byTarget map[string]hub.Backup // latest backup per target id
|
||||
|
||||
Reference in New Issue
Block a user