7861bf9dde
gates / gates (push) Successful in 28s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
276 lines
12 KiB
Go
276 lines
12 KiB
Go
package backup
|
||
|
||
import (
|
||
"context"
|
||
"encoding/json"
|
||
"fmt"
|
||
"sort"
|
||
"strconv"
|
||
"strings"
|
||
"time"
|
||
)
|
||
|
||
// ── Decision 68 (v0.289.0): the box prunes its own repository ONLY inside a hub-opened window ──────────
|
||
//
|
||
// The hub-provisioned tier's key is append-only (decision 69): every delete is refused by the provider.
|
||
// To keep retention working, the box ASKS the hub for a clean-up window after its run. The hub grants
|
||
// one at most weekly (or on an operator's one-shot grant) by PREPENDING a deleting line for the box's own
|
||
// key — OpenSSH uses the first matching line, measured on the provider — and closes it when the box
|
||
// reports, or after 20 minutes on its own.
|
||
//
|
||
// THE FAKE-SNAPSHOT GUARD (R-822) runs BEFORE any forget, inside the window. Measured in the lab: 13
|
||
// future-dated empty snapshots added through an add-only key make the box's own policy select EVERY real
|
||
// snapshot for removal. So, refuse when:
|
||
// - any snapshot is dated in the future (beyond offsiteGuardSkew), or after the hub's newest-allowed
|
||
// bound (the moment the window opened, plus the same skew);
|
||
// - the plan would remove a snapshot whose calendar day is fewer than keepDaily days before today and
|
||
// that is not superseded the same day — the honest policy (--keep-daily keepDaily) never does that,
|
||
// while a poisoning shape does exactly that. v0.294.0 (R-867): the line is DERIVED from keepDaily, the
|
||
// same constant the policy is built from. v0.289–0.293 used a fixed 8-day AGE, which sat inside the
|
||
// keep window: the snapshot a keep-7-dailies policy drops each night is 7 days + seconds old, so every
|
||
// window on a box with more than 7 nightly snapshots refused and mailed an error (measured
|
||
// demo-felhom 2026-10-05). Pinned by TestOffsiteGuard_RealPolicy* (the policy itself, simulated as
|
||
// restic 0.14.0 applies it, over 15+ nightly snapshots and a month boundary).
|
||
// And a plan larger than MaxRemove (the hub's number for one week) REFUSES (v0.290.0, per the 2026-10-04
|
||
// brief, replacing v0.289's cap). The cost, recorded (R-96 rule 4): after a long gap without windows the
|
||
// honest backlog exceeds a week and the guard refuses until the operator grants a window by hand — R-833.
|
||
//
|
||
// The NAS tier (Transport "") is unchanged: the household's own disk, pruned by the box as before.
|
||
//
|
||
// Pinned by TestOffsiteGuard_* (offbox_window_test.go), including the lab's 13-fake shape.
|
||
|
||
const offsiteGuardSkew = time.Hour
|
||
|
||
// The ruled retention policy (SP-2), as constants: BOTH the policy's arguments and the guard's day line
|
||
// are built from these, so the guard can never sit inside the keep window (R-867).
|
||
const (
|
||
keepDaily = 7
|
||
keepWeekly = 4
|
||
keepMonthly = 6
|
||
)
|
||
|
||
// OffsiteWindow is the hub's answer to "may I prune now?".
|
||
type OffsiteWindow struct {
|
||
Granted bool
|
||
ID int64
|
||
NewestAllowed time.Time
|
||
MaxRemove int
|
||
Reason string // why not granted (logged)
|
||
}
|
||
|
||
// OffsiteWindowResult is what the box reports when it is done (the hub closes the window on it).
|
||
type OffsiteWindowResult struct {
|
||
ID int64 `json:"window_id"`
|
||
CountBefore int `json:"count_before"`
|
||
CountAfter int `json:"count_after"`
|
||
Removed int `json:"removed"`
|
||
Outcome string `json:"outcome"` // pruned | nothing | guard-refused | error
|
||
Reason string `json:"reason,omitempty"`
|
||
}
|
||
|
||
// OffsiteWindowClient is the hub side (offsiteapply.HubWindowClient in production).
|
||
type OffsiteWindowClient interface {
|
||
Open(ctx context.Context, countBefore int) (OffsiteWindow, error)
|
||
Close(ctx context.Context, r OffsiteWindowResult) error
|
||
}
|
||
|
||
// SetOffsiteWindowClient wires the hub's window (decision 68). nil → no box-side retention on the pinned tier.
|
||
func (m *Manager) SetOffsiteWindowClient(c OffsiteWindowClient) { m.offsiteWindow = c }
|
||
|
||
// retentionPolicy is the ruled policy, unchanged since SP-2 (`--group-by host,tags`).
|
||
var retentionPolicy = []string{"--group-by", "host,tags",
|
||
"--keep-daily", strconv.Itoa(keepDaily), "--keep-weekly", strconv.Itoa(keepWeekly), "--keep-monthly", strconv.Itoa(keepMonthly)}
|
||
|
||
// calendarDaysBefore: how many calendar days s lies before now, both read in s's own zone — the zone
|
||
// restic 0.14.0 buckets a snapshot's day in (the offset stored with the snapshot). Edge, recorded: in the
|
||
// hour after local midnight on a DST change, a box whose snapshots carry two different offsets can read one
|
||
// day short; the guard then REFUSES (the safe direction) and the next week's window passes.
|
||
func calendarDaysBefore(s, now time.Time) int {
|
||
loc := s.Location()
|
||
a, b := s.In(loc), now.In(loc)
|
||
da := time.Date(a.Year(), a.Month(), a.Day(), 0, 0, 0, 0, time.UTC)
|
||
db := time.Date(b.Year(), b.Month(), b.Day(), 0, 0, 0, 0, time.UTC)
|
||
return int(db.Sub(da).Hours() / 24)
|
||
}
|
||
|
||
type guardSnap struct {
|
||
ID string `json:"id"`
|
||
ShortID string `json:"short_id"`
|
||
Time time.Time `json:"time"`
|
||
Hostname string `json:"hostname"`
|
||
Tags []string `json:"tags"`
|
||
}
|
||
|
||
// group is the `--group-by host,tags` key.
|
||
func (g guardSnap) group() string {
|
||
t := append([]string{}, g.Tags...)
|
||
sort.Strings(t)
|
||
return g.Hostname + "|" + strings.Join(t, ",")
|
||
}
|
||
|
||
// supersededSameDay: a NEWER snapshot of the same group exists on the same UTC day (a manual run after
|
||
// the night's) — the one benign reason the honest policy removes a young snapshot (R-824, measured on
|
||
// demo-hp 2026-10-03).
|
||
func supersededSameDay(s guardSnap, all []guardSnap) bool {
|
||
day := s.Time.UTC().Format("2006-01-02")
|
||
for _, o := range all {
|
||
if o.ID != s.ID && o.group() == s.group() && o.Time.After(s.Time) && o.Time.UTC().Format("2006-01-02") == day {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
// offsiteGuard is the PURE decision: from all snapshots and the policy's remove-plan, either the ids to
|
||
// remove (oldest first) or a refusal reason. v0.294.0 (R-867): "young" means fewer than keepDaily calendar
|
||
// days before today, not an age in hours. v0.290.0 (R-824): a YOUNG snapshot that a newer same-day
|
||
// snapshot of its group supersedes is EXCLUDED (kept for a later window, when it is old) instead of
|
||
// refusing the run — v0.289.x refused every window after any manual run. A young removal WITHOUT that
|
||
// explanation still refuses: it is the poisoning signature. Future-dated snapshots, snapshots newer than
|
||
// the hub allows, and a plan larger than a week's removal (maxRemove) refuse.
|
||
func offsiteGuard(all, plan []guardSnap, now, newestAllowed time.Time, maxRemove int) ([]string, string) {
|
||
for _, s := range all {
|
||
if s.Time.After(now.Add(offsiteGuardSkew)) {
|
||
return nil, fmt.Sprintf("snapshot %s is dated in the future (%s)", s.ShortID, s.Time.UTC().Format(time.RFC3339))
|
||
}
|
||
if !newestAllowed.IsZero() && s.Time.After(newestAllowed.Add(offsiteGuardSkew)) {
|
||
return nil, fmt.Sprintf("snapshot %s (%s) is newer than the hub allows (%s)", s.ShortID, s.Time.UTC().Format(time.RFC3339), newestAllowed.UTC().Format(time.RFC3339))
|
||
}
|
||
}
|
||
var keep []guardSnap
|
||
for _, s := range plan {
|
||
if calendarDaysBefore(s.Time, now) < keepDaily {
|
||
if supersededSameDay(s, all) {
|
||
continue // excluded: removed in a later window, once keepDaily days old
|
||
}
|
||
return nil, fmt.Sprintf("the policy would remove snapshot %s from %s — within the last %d days kept daily and not superseded the same day, which honest retention never does",
|
||
s.ShortID, s.Time.UTC().Format(time.RFC3339), keepDaily)
|
||
}
|
||
keep = append(keep, s)
|
||
}
|
||
if maxRemove >= 0 && len(keep) > maxRemove {
|
||
return nil, fmt.Sprintf("the plan would remove %d snapshots, more than one week's retention may (%d)", len(keep), maxRemove)
|
||
}
|
||
sort.Slice(keep, func(i, j int) bool { return keep[i].Time.Before(keep[j].Time) })
|
||
ids := make([]string, 0, len(keep))
|
||
for _, s := range keep {
|
||
ids = append(ids, s.ID)
|
||
}
|
||
return ids, ""
|
||
}
|
||
|
||
func (m *Manager) listGuardSnaps(ctx context.Context, base, env []string) ([]guardSnap, error) {
|
||
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
defer cancel()
|
||
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
||
if err != nil {
|
||
return nil, fmt.Errorf("list snapshots: %w: %s", err, truncate(out))
|
||
}
|
||
var snaps []guardSnap
|
||
if err := json.Unmarshal(out, &snaps); err != nil {
|
||
return nil, fmt.Errorf("parse snapshots: %w", err)
|
||
}
|
||
return snaps, nil
|
||
}
|
||
|
||
func (m *Manager) planRemovals(ctx context.Context, base, env []string) ([]guardSnap, error) {
|
||
pctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
||
defer cancel()
|
||
args := append(append(append([]string{}, base...), "forget"), retentionPolicy...)
|
||
args = append(args, "--dry-run", "--json")
|
||
out, err := m.runner()(pctx, env, args...)
|
||
if err != nil {
|
||
return nil, fmt.Errorf("forget --dry-run: %w: %s", err, truncate(out))
|
||
}
|
||
// restic 0.14.0 prints the JSON array on stdout; the runner may combine stderr — take the array.
|
||
js := string(out)
|
||
if i := strings.Index(js, "["); i > 0 {
|
||
js = js[i:]
|
||
}
|
||
var groups []struct {
|
||
Remove []guardSnap `json:"remove"`
|
||
}
|
||
if err := json.Unmarshal([]byte(strings.TrimSpace(js)), &groups); err != nil {
|
||
return nil, fmt.Errorf("parse forget plan: %w", err)
|
||
}
|
||
var plan []guardSnap
|
||
for _, g := range groups {
|
||
plan = append(plan, g.Remove...)
|
||
}
|
||
return plan, nil
|
||
}
|
||
|
||
// offsiteWindowRetention is the ONE retention step for both callers (after a run, over quota).
|
||
func (m *Manager) offsiteWindowRetention(ctx context.Context, base, env []string, why string) {
|
||
t := m.settings.GetOffboxTarget()
|
||
if !t.Pinned() {
|
||
// The household's own SFTP NAS: the box prunes as it always did (SP-2 policy).
|
||
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||
defer cancel()
|
||
args := append(append([]string{"forget"}, retentionPolicy...), "--prune")
|
||
if out, ferr := m.resticStep(fctx, env, base, "prune", args...); ferr != nil {
|
||
m.logger.Printf("[WARN] [offbox] forget --prune failed (%s; backups are safe): %v: %s", why, ferr, truncate(out))
|
||
}
|
||
return
|
||
}
|
||
if m.offsiteWindow == nil {
|
||
m.logger.Printf("[INFO] [offbox] retention skipped (%s): the off-site key is append-only and no clean-up window client is wired — nothing deleted (decision 68)", why)
|
||
return
|
||
}
|
||
snaps, err := m.listGuardSnaps(ctx, base, env)
|
||
if err != nil {
|
||
m.logger.Printf("[WARN] [offbox] retention skipped (%s): %v", why, err)
|
||
return
|
||
}
|
||
w, err := m.offsiteWindow.Open(ctx, len(snaps))
|
||
if err != nil {
|
||
m.logger.Printf("[WARN] [offbox] retention skipped (%s): asking the hub for a window failed: %v", why, err)
|
||
return
|
||
}
|
||
if !w.Granted {
|
||
m.logger.Printf("[INFO] [offbox] retention skipped (%s): no clean-up window now (%s) — nothing deleted (decision 68)", why, w.Reason)
|
||
return
|
||
}
|
||
start := time.Now()
|
||
res := OffsiteWindowResult{ID: w.ID, CountBefore: len(snaps), CountAfter: len(snaps)}
|
||
defer func() {
|
||
cctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||
defer cancel()
|
||
if cerr := m.offsiteWindow.Close(cctx, res); cerr != nil {
|
||
m.logger.Printf("[WARN] [offbox] closing clean-up window %d with the hub failed (the hub closes it by itself in 20 min): %v", w.ID, cerr)
|
||
}
|
||
}()
|
||
plan, err := m.planRemovals(ctx, base, env)
|
||
if err != nil {
|
||
res.Outcome, res.Reason = "error", err.Error()
|
||
m.logger.Printf("[WARN] [offbox] clean-up window %d: %v", w.ID, err)
|
||
return
|
||
}
|
||
ids, refuse := offsiteGuard(snaps, plan, time.Now(), w.NewestAllowed, w.MaxRemove)
|
||
if refuse != "" {
|
||
res.Outcome, res.Reason = "guard-refused", refuse
|
||
m.logger.Printf("[ERROR] [offbox] clean-up window %d: the fake-snapshot guard REFUSED — nothing deleted: %s (R-822)", w.ID, refuse)
|
||
return
|
||
}
|
||
if len(ids) == 0 {
|
||
res.Outcome = "nothing"
|
||
m.logger.Printf("[INFO] [offbox] clean-up window %d: the policy removes nothing", w.ID)
|
||
return
|
||
}
|
||
fctx, cancel := context.WithTimeout(ctx, offboxBackupTimeout)
|
||
defer cancel()
|
||
args := append(append([]string{"forget"}, ids...), "--prune")
|
||
if out, ferr := m.resticStep(fctx, env, base, "prune", args...); ferr != nil {
|
||
res.Outcome, res.Reason = "error", truncate(out)
|
||
m.logger.Printf("[WARN] [offbox] clean-up window %d: forget --prune failed (backups are safe): %v: %s", w.ID, ferr, truncate(out))
|
||
} else {
|
||
res.Outcome = "pruned"
|
||
}
|
||
if after, lerr := m.listGuardSnaps(ctx, base, env); lerr == nil {
|
||
res.CountAfter = len(after)
|
||
}
|
||
res.Removed = res.CountBefore - res.CountAfter
|
||
m.logger.Printf("[INFO] [offbox] clean-up window %d (%s): %d of %d planned snapshot(s) removed, %d -> %d, in %s",
|
||
w.ID, why, res.Removed, len(plan), res.CountBefore, res.CountAfter, time.Since(start).Round(time.Second))
|
||
}
|