74b5eae5b0
gates / gates (push) Successful in 35s
An unreadable storage right after an agent restart read the off-site tier DUE (the in-memory record was empty). The newest success per tier is now kept on disk and read ONLY when the storage cannot be read: fresh -> not due, older than the cadence -> due, none -> due (unknown) as before. A storage that answers stays the ground truth. Ships with v0.150.0 after the 2026-10-07 read-back; nothing delivered. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
132 lines
4.4 KiB
Go
132 lines
4.4 KiB
Go
package backup
|
|
|
|
import (
|
|
"encoding/json"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strconv"
|
|
"sync"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
|
)
|
|
|
|
// BackupSuccessState persists the newest SUCCESSFUL whole-guest backup per tier and guest (R-894).
|
|
//
|
|
// Why it exists. The due-check (`localapi` handleBackupDue) asks the tier's storage when a backup last
|
|
// landed (R-84) and falls back to the in-memory record when the storage cannot be read. The in-memory
|
|
// record is empty after an agent restart (Store, R-348), so "storage unreadable" right after a restart
|
|
// read as "no record — DUE". Measured 2026-10-05 on demo-hp: the agent restarted at 04:57, the off-site
|
|
// storage answered "Can't connect" at 06:25, the 7-day tier — last copy 2026-10-01 — read DUE, the
|
|
// controller asked, and vzdump failed. This file is the last known copy the fallback reads instead.
|
|
//
|
|
// It is read ONLY when the storage cannot be read. A storage that answers is the ground truth and wins,
|
|
// in both directions: an archive found there counts, and an archive absent there is absent even when
|
|
// this file remembers a success (a pruned or deleted archive must make the tier due — the same reason
|
|
// R-84 chose the storage over a persisted record). Pinned by
|
|
// TestBackupDue_R894_SavedCopyIgnoredWhenStorageAnswers.
|
|
//
|
|
// Only SUCCESSES are written (the RestoreTestState rule): a failure must stay due and be retried, so a
|
|
// record of a failure has no reader.
|
|
type BackupSuccessState struct {
|
|
path string
|
|
mu sync.Mutex
|
|
last map[string]savedSuccess // key(target, vmid) → the newest success
|
|
}
|
|
|
|
type savedSuccess struct {
|
|
target string
|
|
vmid int
|
|
at time.Time
|
|
}
|
|
|
|
// backupSuccessJSON is one entry on disk.
|
|
type backupSuccessJSON struct {
|
|
Target string `json:"target"`
|
|
VMID int `json:"vmid"`
|
|
StartedAt string `json:"started_at"`
|
|
}
|
|
|
|
func backupStateKey(target string, vmid int) string { return target + "/" + strconv.Itoa(vmid) }
|
|
|
|
// NewBackupSuccessState opens (or creates) the state at path. A missing or unreadable file degrades to
|
|
// "nothing known" — the pre-R-894 behaviour, which is DUE — and never wedges the daemon.
|
|
func NewBackupSuccessState(path string) *BackupSuccessState {
|
|
s := &BackupSuccessState{path: path, last: map[string]savedSuccess{}}
|
|
data, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return s
|
|
}
|
|
var entries []backupSuccessJSON
|
|
if json.Unmarshal(data, &entries) != nil {
|
|
return s
|
|
}
|
|
for _, e := range entries {
|
|
t, perr := time.Parse(time.RFC3339, e.StartedAt)
|
|
if perr != nil {
|
|
continue // one unreadable entry must not lose the others
|
|
}
|
|
s.last[backupStateKey(e.Target, e.VMID)] = savedSuccess{target: e.Target, vmid: e.VMID, at: t.UTC()}
|
|
}
|
|
return s
|
|
}
|
|
|
|
// RecordBackupSuccess saves b when it is a success newer than the one on file. target is the tier the
|
|
// job ran on (the due-check's key); a failure or an unparseable time is ignored.
|
|
func (s *BackupSuccessState) RecordBackupSuccess(target string, b hub.Backup) error {
|
|
if s == nil || !b.Success {
|
|
return nil
|
|
}
|
|
t, err := time.Parse(time.RFC3339, b.StartedAt)
|
|
if err != nil {
|
|
return nil
|
|
}
|
|
s.mu.Lock()
|
|
defer s.mu.Unlock()
|
|
k := backupStateKey(target, b.VMID)
|
|
if old, ok := s.last[k]; ok && !t.After(old.at) {
|
|
return nil
|
|
}
|
|
s.last[k] = savedSuccess{target: target, vmid: b.VMID, at: t.UTC()}
|
|
return s.saveLocked()
|
|
}
|
|
|
|
// LastKnownSuccess returns the newest saved success for this tier and guest (ok=false = none on file).
|
|
func (s *BackupSuccessState) LastKnownSuccess(target string, vmid int) (time.Time, bool) {
|
|
if s == nil {
|
|
return time.Time{}, false
|
|
}
|
|
s.mu.Lock()
|
|
defer s.mu.Unlock()
|
|
e, ok := s.last[backupStateKey(target, vmid)]
|
|
return e.at, ok
|
|
}
|
|
|
|
func (s *BackupSuccessState) saveLocked() error {
|
|
entries := make([]backupSuccessJSON, 0, len(s.last))
|
|
for _, e := range s.last {
|
|
entries = append(entries, backupSuccessJSON{Target: e.target, VMID: e.vmid, StartedAt: e.at.Format(time.RFC3339)})
|
|
}
|
|
// Deterministic file content (Go's map order is random).
|
|
sort.Slice(entries, func(i, j int) bool {
|
|
if entries[i].Target != entries[j].Target {
|
|
return entries[i].Target < entries[j].Target
|
|
}
|
|
return entries[i].VMID < entries[j].VMID
|
|
})
|
|
data, err := json.MarshalIndent(entries, "", " ")
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if err := os.MkdirAll(filepath.Dir(s.path), 0o755); err != nil {
|
|
return err
|
|
}
|
|
tmp := s.path + ".tmp"
|
|
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
|
os.Remove(tmp)
|
|
return err
|
|
}
|
|
return os.Rename(tmp, s.path)
|
|
}
|