Files
felhom-agent/internal/backup/backup_state.go
T
admin 74b5eae5b0
gates / gates (push) Successful in 35s
R-894: after a restart the agent remembers the last backup per tier
An unreadable storage right after an agent restart read the off-site tier
DUE (the in-memory record was empty). The newest success per tier is now
kept on disk and read ONLY when the storage cannot be read: fresh -> not
due, older than the cadence -> due, none -> due (unknown) as before. A
storage that answers stays the ground truth.

Ships with v0.150.0 after the 2026-10-07 read-back; nothing delivered.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
2026-10-06 20:33:27 +02:00

132 lines
4.4 KiB
Go

package backup
import (
"encoding/json"
"os"
"path/filepath"
"sort"
"strconv"
"sync"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
)
// BackupSuccessState persists the newest SUCCESSFUL whole-guest backup per tier and guest (R-894).
//
// Why it exists. The due-check (`localapi` handleBackupDue) asks the tier's storage when a backup last
// landed (R-84) and falls back to the in-memory record when the storage cannot be read. The in-memory
// record is empty after an agent restart (Store, R-348), so "storage unreadable" right after a restart
// read as "no record — DUE". Measured 2026-10-05 on demo-hp: the agent restarted at 04:57, the off-site
// storage answered "Can't connect" at 06:25, the 7-day tier — last copy 2026-10-01 — read DUE, the
// controller asked, and vzdump failed. This file is the last known copy the fallback reads instead.
//
// It is read ONLY when the storage cannot be read. A storage that answers is the ground truth and wins,
// in both directions: an archive found there counts, and an archive absent there is absent even when
// this file remembers a success (a pruned or deleted archive must make the tier due — the same reason
// R-84 chose the storage over a persisted record). Pinned by
// TestBackupDue_R894_SavedCopyIgnoredWhenStorageAnswers.
//
// Only SUCCESSES are written (the RestoreTestState rule): a failure must stay due and be retried, so a
// record of a failure has no reader.
type BackupSuccessState struct {
path string
mu sync.Mutex
last map[string]savedSuccess // key(target, vmid) → the newest success
}
type savedSuccess struct {
target string
vmid int
at time.Time
}
// backupSuccessJSON is one entry on disk.
type backupSuccessJSON struct {
Target string `json:"target"`
VMID int `json:"vmid"`
StartedAt string `json:"started_at"`
}
func backupStateKey(target string, vmid int) string { return target + "/" + strconv.Itoa(vmid) }
// NewBackupSuccessState opens (or creates) the state at path. A missing or unreadable file degrades to
// "nothing known" — the pre-R-894 behaviour, which is DUE — and never wedges the daemon.
func NewBackupSuccessState(path string) *BackupSuccessState {
s := &BackupSuccessState{path: path, last: map[string]savedSuccess{}}
data, err := os.ReadFile(path)
if err != nil {
return s
}
var entries []backupSuccessJSON
if json.Unmarshal(data, &entries) != nil {
return s
}
for _, e := range entries {
t, perr := time.Parse(time.RFC3339, e.StartedAt)
if perr != nil {
continue // one unreadable entry must not lose the others
}
s.last[backupStateKey(e.Target, e.VMID)] = savedSuccess{target: e.Target, vmid: e.VMID, at: t.UTC()}
}
return s
}
// RecordBackupSuccess saves b when it is a success newer than the one on file. target is the tier the
// job ran on (the due-check's key); a failure or an unparseable time is ignored.
func (s *BackupSuccessState) RecordBackupSuccess(target string, b hub.Backup) error {
if s == nil || !b.Success {
return nil
}
t, err := time.Parse(time.RFC3339, b.StartedAt)
if err != nil {
return nil
}
s.mu.Lock()
defer s.mu.Unlock()
k := backupStateKey(target, b.VMID)
if old, ok := s.last[k]; ok && !t.After(old.at) {
return nil
}
s.last[k] = savedSuccess{target: target, vmid: b.VMID, at: t.UTC()}
return s.saveLocked()
}
// LastKnownSuccess returns the newest saved success for this tier and guest (ok=false = none on file).
func (s *BackupSuccessState) LastKnownSuccess(target string, vmid int) (time.Time, bool) {
if s == nil {
return time.Time{}, false
}
s.mu.Lock()
defer s.mu.Unlock()
e, ok := s.last[backupStateKey(target, vmid)]
return e.at, ok
}
func (s *BackupSuccessState) saveLocked() error {
entries := make([]backupSuccessJSON, 0, len(s.last))
for _, e := range s.last {
entries = append(entries, backupSuccessJSON{Target: e.target, VMID: e.vmid, StartedAt: e.at.Format(time.RFC3339)})
}
// Deterministic file content (Go's map order is random).
sort.Slice(entries, func(i, j int) bool {
if entries[i].Target != entries[j].Target {
return entries[i].Target < entries[j].Target
}
return entries[i].VMID < entries[j].VMID
})
data, err := json.MarshalIndent(entries, "", " ")
if err != nil {
return err
}
if err := os.MkdirAll(filepath.Dir(s.path), 0o755); err != nil {
return err
}
tmp := s.path + ".tmp"
if err := os.WriteFile(tmp, data, 0o600); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, s.path)
}