controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s
gates / gates (push) Successful in 14s
Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1 (own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable counts as absent with a WARN) and leans on the first FRESH copy; the backup_max_age rule applies to whichever tier is chosen. No copy anywhere: back up first. Refused only when nothing exists and no backup can be taken. RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured unit proven current. The hold names the tier (második meghajtó / saját meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text. A successful off-site restore now lifts an update hold. The backups page still uses Tier2UnitRestorePoint unchanged. Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
@@ -173,6 +173,13 @@ type Manager struct {
|
||||
// updatingCheck (slice 4) — nil-safe; see isHeld / SetUpdatingCheck.
|
||||
updatingCheck func(stackName string) bool
|
||||
|
||||
// R-475 update-precondition seams, one per tier. Nil → the real Tier2UnitRestorePoint /
|
||||
// ListRestorePoints / OffsiteInventoryList. They let a test reach Tier 1 and Tier 3 without a drive
|
||||
// or a restic repository; see UpdateRestorePoints.
|
||||
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
|
||||
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
|
||||
updateOffsiteInvFn func(ctx context.Context) (OffsiteInventory, error)
|
||||
|
||||
// R-354 volume-REPLAY seam — the mirror of the F17 DB seams above, so the off-site path's new
|
||||
// volume leg is unit-testable without Docker. Nil → the real restoreDockerVolumesFrom.
|
||||
volumeReplayFrom func(stackName, dumpDir string) (int, error)
|
||||
|
||||
@@ -344,7 +344,11 @@ func (m *Manager) RestoreHoldFor(stack string) (bool, string) {
|
||||
if h.CopyDate != "" {
|
||||
copyDate = fmtHoldTime(h.CopyDate)
|
||||
}
|
||||
return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), copyDate)
|
||||
// R-475: name the tier when the hold recorded one; an older hold keeps its own sentence.
|
||||
if label := UpdateTierLabel(h.CopyTier); label != "" && h.CopyDate != "" {
|
||||
return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), label, copyDate)
|
||||
}
|
||||
return true, fmt.Sprintf(UpdateHoldLegacyFmt, stack, fmtHoldTime(h.At), copyDate)
|
||||
}
|
||||
when := h.At
|
||||
if t, err := time.Parse(time.RFC3339, h.At); err == nil {
|
||||
@@ -824,6 +828,11 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
|
||||
if err := restartStack(); err != nil {
|
||||
return res, fmt.Errorf("a(z) %s újraindítása sikertelen a fájlok visszaállítása után: %w", stack, err)
|
||||
}
|
||||
// R-475: an update hold may now name the OFF-SITE copy, and this is the route back it names. The
|
||||
// unit restore clears it in RestoreFromRecoveryUnitAt; this path never went through that function,
|
||||
// so without this line a successful off-site restore would leave the app refusing its next start.
|
||||
// Pinned by TestR475_OffsiteRestoreClearsAnUpdateHold.
|
||||
m.clearUpdateHoldAfterRestore(stack)
|
||||
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
|
||||
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
|
||||
}
|
||||
|
||||
@@ -68,7 +68,7 @@ func TestSlice4_UpdateHoldTextNamesTheTimeAndTheCopy(t *testing.T) {
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
at := time.Date(2026, 9, 13, 8, 0, 0, 0, time.UTC)
|
||||
copyAt := time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)
|
||||
if err := m.HoldAfterFailedUpdate("bookstack", at, copyAt); err != nil {
|
||||
if err := m.HoldAfterFailedUpdate("bookstack", at, copyAt, UpdateTierSecondDrive); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h, ok := sett.GetRestoreHold("bookstack")
|
||||
@@ -77,7 +77,7 @@ func TestSlice4_UpdateHoldTextNamesTheTimeAndTheCopy(t *testing.T) {
|
||||
}
|
||||
held, why := m.RestoreHoldFor("bookstack")
|
||||
// Budapest is UTC+2 in September: 08:00Z → 10:00, 01:30Z → 03:30.
|
||||
want := fmt.Sprintf(UpdateHoldFmt, "bookstack", "2026-09-13 10:00", "2026-09-13 03:30")
|
||||
want := fmt.Sprintf(UpdateHoldFmt, "bookstack", "2026-09-13 10:00", "második meghajtó", "2026-09-13 03:30")
|
||||
if !held || why != want {
|
||||
t.Errorf("hold text =\n%q\nwant\n%q", why, want)
|
||||
}
|
||||
@@ -98,7 +98,7 @@ func TestSlice4_RestoreHoldTextIsUnchanged(t *testing.T) {
|
||||
func TestSlice4_ASuccessfulRestoreClearsOnlyAnUpdateHold(t *testing.T) {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
_ = m.HoldAfterFailedUpdate("upd", time.Now(), time.Now())
|
||||
_ = m.HoldAfterFailedUpdate("upd", time.Now(), time.Now(), UpdateTierLocal)
|
||||
_ = sett.SetRestoreHold(settings.RestoreHold{Stack: "rst", At: "2026-08-22T14:00:00Z"})
|
||||
m.clearUpdateHoldAfterRestore("upd")
|
||||
m.clearUpdateHoldAfterRestore("rst")
|
||||
@@ -136,7 +136,7 @@ func TestSlice4_UpdateBusy(t *testing.T) {
|
||||
func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "held", "free")
|
||||
h.m.settings = slice4Settings(t)
|
||||
if err := h.m.HoldAfterFailedUpdate("held", time.Now(), time.Now()); err != nil {
|
||||
if err := h.m.HoldAfterFailedUpdate("held", time.Now(), time.Now(), UpdateTierSecondDrive); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h.m.runVolumeDumps()
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
@@ -93,6 +94,149 @@ func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
|
||||
return t, true
|
||||
}
|
||||
|
||||
// ── R-475: any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0) ────
|
||||
//
|
||||
// Until v0.239.0 the update's precondition was Tier2UnitRestorePoint alone, so an app with no second
|
||||
// drive could never be updated — even with a fresh recovery unit on its own drive and an off-site
|
||||
// snapshot from last night. The ruling: every backup counts. Tier2UnitRestorePoint itself is NOT
|
||||
// changed; the backups page still calls it for the „Teljes visszaállítás" action, which really does
|
||||
// restore from the second drive only.
|
||||
|
||||
// Backup tiers, as the update precondition and the hold sentence name them.
|
||||
const (
|
||||
UpdateTierLocal = 1 // the app's own recovery unit on its drive — „helyi" on the restore page
|
||||
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
|
||||
UpdateTierOffsite = 3 // the off-site restic repository
|
||||
)
|
||||
|
||||
// updateTierOrder is the preference order the ruling set: the second drive, then the app's own unit,
|
||||
// then off-site. The first tier holding a copy the caller ACCEPTS is chosen.
|
||||
var updateTierOrder = []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite}
|
||||
|
||||
// UpdateTierLabel is a tier's name in the customer's hold sentence. "" for an unknown tier.
|
||||
func UpdateTierLabel(tier int) string {
|
||||
switch tier {
|
||||
case UpdateTierSecondDrive:
|
||||
return "második meghajtó"
|
||||
case UpdateTierLocal:
|
||||
return "saját meghajtó"
|
||||
case UpdateTierOffsite:
|
||||
return "távoli mentés"
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// updateOffsiteCheckTimeout bounds the off-site lookup. An update must not stall on an unreachable
|
||||
// Storage Box: past this the off-site copy counts as ABSENT (with a WARN), and the update carries on
|
||||
// with backing up first. A var only so a test can shorten it.
|
||||
var updateOffsiteCheckTimeout = 15 * time.Second
|
||||
|
||||
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
|
||||
type UpdateTierPoint struct {
|
||||
Tier int
|
||||
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
|
||||
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
|
||||
At time.Time
|
||||
}
|
||||
|
||||
// UpdateRestorePoints walks the tiers in preference order (2, 1, 3) and returns the FIRST copy that
|
||||
// accept admits (nil accepts any), whether one was found, and every copy it looked at on the way.
|
||||
//
|
||||
// It stops at the first accepted copy, so a box with a fresh second-drive copy never touches the
|
||||
// network. The AGE rule is the caller's (stacks applies backup_max_age through accept) — that is what
|
||||
// makes "the age rule applies to whichever tier is chosen" one rule, not three (R-475 Scenario M).
|
||||
func (m *Manager) UpdateRestorePoints(ctx context.Context, stackName string, accept func(UpdateTierPoint) bool) (UpdateTierPoint, bool, []UpdateTierPoint) {
|
||||
var seen []UpdateTierPoint
|
||||
for _, tier := range updateTierOrder {
|
||||
p, ok := m.updateTierPoint(ctx, stackName, tier)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
seen = append(seen, p)
|
||||
if accept == nil || accept(p) {
|
||||
return p, true, seen
|
||||
}
|
||||
}
|
||||
return UpdateTierPoint{}, false, seen
|
||||
}
|
||||
|
||||
func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier int) (UpdateTierPoint, bool) {
|
||||
switch tier {
|
||||
case UpdateTierSecondDrive:
|
||||
get := m.updateTier2PointFn
|
||||
if get == nil {
|
||||
get = m.Tier2UnitRestorePoint
|
||||
}
|
||||
rp, err := get(stackName)
|
||||
if err != nil {
|
||||
if m.isDebug() {
|
||||
m.logger.Printf("[DEBUG] [backup] update precondition for %s: no Tier-2 copy (%v)", stackName, err)
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
at, ok := rp.ProvenCopyTime()
|
||||
return UpdateTierPoint{Tier: tier, At: at}, ok
|
||||
case UpdateTierLocal:
|
||||
list := m.updateTier1PointsFn
|
||||
if list == nil {
|
||||
list = m.ListRestorePoints
|
||||
}
|
||||
pts, _ := list(stackName)
|
||||
for _, rp := range pts {
|
||||
if at, err := time.Parse(time.RFC3339, rp.Time); err == nil {
|
||||
return UpdateTierPoint{Tier: tier, At: at}, true
|
||||
}
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
case UpdateTierOffsite:
|
||||
inv := m.updateOffsiteInvFn
|
||||
if inv == nil {
|
||||
if m.settings == nil || !m.OffboxConfigured() {
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
inv = m.OffsiteInventoryList
|
||||
}
|
||||
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
|
||||
defer cancel()
|
||||
got, err := inv(cctx)
|
||||
if err != nil {
|
||||
if !errors.Is(err, errNoOffsiteTarget) {
|
||||
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
for _, a := range got.Apps {
|
||||
if a.App == stackName && !a.LatestAt.IsZero() {
|
||||
return UpdateTierPoint{Tier: tier, At: a.LatestAt}, true
|
||||
}
|
||||
}
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
|
||||
// CanBackUpApp reports whether "back up first" can run for this app at all right now — the second
|
||||
// half of R-475 Scenario L: an app with no copy anywhere is refused only when this is false too.
|
||||
// Cheap and read-only; RunAppBackupNow re-checks everything when it actually runs.
|
||||
func (m *Manager) CanBackUpApp(stackName string) (bool, string) {
|
||||
if m == nil {
|
||||
return false, "backup is not enabled on this box"
|
||||
}
|
||||
if m.stackProvider == nil {
|
||||
return false, "stack provider not configured"
|
||||
}
|
||||
if m.migrationActive() {
|
||||
return false, "a data migration is running"
|
||||
}
|
||||
drivePath := m.GetAppDrivePath(stackName)
|
||||
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
||||
return false, "the app's drive cannot be resolved"
|
||||
}
|
||||
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
|
||||
return false, fmt.Sprintf("the app's drive %s is not available", drivePath)
|
||||
}
|
||||
return true, ""
|
||||
}
|
||||
|
||||
// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an
|
||||
// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer
|
||||
// sentence is chosen by the caller.
|
||||
@@ -142,6 +286,7 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName)
|
||||
start := time.Now()
|
||||
var nsRoot string
|
||||
legErr := func() error {
|
||||
defer m.releaseRunning()
|
||||
defer m.beginAdmissionRun()()
|
||||
@@ -156,7 +301,7 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
|
||||
if !m.admitApp(stackName) {
|
||||
return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath)
|
||||
}
|
||||
nsRoot := m.namespaceRoot(drivePath)
|
||||
nsRoot = m.namespaceRoot(drivePath)
|
||||
|
||||
discover := m.discoverDBs
|
||||
if discover == nil {
|
||||
@@ -203,16 +348,35 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
|
||||
return legErr
|
||||
}
|
||||
|
||||
m.updatePreBackupTail(stackName, nsRoot, time.Now())
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
|
||||
return nil
|
||||
}
|
||||
|
||||
// updatePreBackupTail is what "back up first" does after the capture succeeded (R-475).
|
||||
//
|
||||
// 1. It marks the app's OWN unit as proven current NOW. CaptureRecoveryUnit leaves the manifest alone
|
||||
// when nothing changed (the checksum skip), and Tier 1's age is the newest artifact's mtime — so on
|
||||
// an app with no database and no named volume a fresh "back up first" would leave Tier 1 as old as
|
||||
// its last definition change, and the update would be refused forever. That is the trap
|
||||
// ProvenCopyTime documents for Tier 2, one tier down. The capture has just compared the unit with
|
||||
// the live definition, so "current as of now" is exactly what it established.
|
||||
// 2. It runs the Tier-2 copy, and a Tier-2 failure is a WARN, not a failure: the update may lean on
|
||||
// any tier, and the Tier-1 unit it can lean on was just written. Pinned by
|
||||
// TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh.
|
||||
func (m *Manager) updatePreBackupTail(stackName, nsRoot string, now time.Time) {
|
||||
if nsRoot != "" {
|
||||
if err := os.Chtimes(RecoveryUnitManifestPath(nsRoot, stackName), now, now); err != nil {
|
||||
m.logger.Printf("[WARN] [backup] update pre-backup for %s: could not mark the recovery unit as proven current (%v) — its own-unit copy may read older than it is", stackName, err)
|
||||
}
|
||||
}
|
||||
runOne := m.perAppTier2
|
||||
if runOne == nil {
|
||||
runOne = m.RunTier2
|
||||
}
|
||||
if err := runOne(stackName); err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v", stackName, err)
|
||||
return fmt.Errorf("a másodlagos másolat elkészítése sikertelen: %w", err)
|
||||
m.logger.Printf("[WARN] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v — not fatal: the app's own recovery unit was just captured, and an update may lean on any tier (R-475)", stackName, err)
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
|
||||
return nil
|
||||
}
|
||||
|
||||
// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the
|
||||
@@ -244,9 +408,18 @@ func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) (
|
||||
}
|
||||
|
||||
// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app,
|
||||
// the time of the failure, and the PROVEN date of the copy it can be restored from. One named string so
|
||||
// a test asserts it verbatim instead of retyping Hungarian (R-364).
|
||||
// the time of the failure, the TIER of the copy it can be restored from (UpdateTierLabel), and that
|
||||
// copy's PROVEN date. One named string so a test asserts it verbatim instead of retyping Hungarian
|
||||
// (R-364). Since v0.239.0 (R-475) it names the tier: the copy may be on any of three, and each is
|
||||
// restored from a different place on the Mentések page.
|
||||
const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
|
||||
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
|
||||
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s."
|
||||
|
||||
// UpdateHoldLegacyFmt is the v0.237.0–v0.238.1 sentence, kept for a hold written before the tier was
|
||||
// recorded (CopyTier 0) — every such hold named a Tier-2 copy, but it did not SAY so, and rewriting
|
||||
// it now would state a fact the record does not hold.
|
||||
const UpdateHoldLegacyFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
|
||||
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
|
||||
"Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon."
|
||||
|
||||
@@ -276,7 +449,7 @@ func fmtHoldTime(rfc3339 string) string {
|
||||
// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app
|
||||
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
|
||||
// failure on the page.
|
||||
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time) error {
|
||||
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error {
|
||||
if m == nil || m.settings == nil {
|
||||
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
|
||||
}
|
||||
@@ -287,11 +460,12 @@ func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate
|
||||
}
|
||||
if !copyDate.IsZero() {
|
||||
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
|
||||
h.CopyTier = copyTier
|
||||
}
|
||||
if err := m.settings.SetRestoreHold(h); err != nil {
|
||||
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
|
||||
}
|
||||
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: %s)", stackName, h.CopyDate)
|
||||
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate)
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,317 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-475 — the backup side of "any backup tier lets an app update" (controller v0.239.0).
|
||||
|
||||
var r475T0 = time.Date(2026, 9, 13, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
func r475Manager() (*Manager, *bytes.Buffer) {
|
||||
var buf bytes.Buffer
|
||||
return &Manager{logger: log.New(&buf, "", 0)}, &buf
|
||||
}
|
||||
|
||||
func tier2At(at time.Time) func(string) (Tier2RestorePoint, error) {
|
||||
return func(string) (Tier2RestorePoint, error) {
|
||||
ts := at.UTC().Format(time.RFC3339)
|
||||
return Tier2RestorePoint{Restorable: true, CopyDateProven: true, CopyLastSuccess: ts, CopyDate: ts}, nil
|
||||
}
|
||||
}
|
||||
func noTier2(string) (Tier2RestorePoint, error) {
|
||||
return Tier2RestorePoint{}, errors.New("no Tier-2 copy recorded for this app")
|
||||
}
|
||||
func tier1At(at time.Time) func(string) ([]RestorePoint, bool) {
|
||||
return func(string) ([]RestorePoint, bool) {
|
||||
return []RestorePoint{{Time: at.UTC().Format(time.RFC3339), ShortID: "helyi", Tier: 1}}, true
|
||||
}
|
||||
}
|
||||
func noTier1(string) ([]RestorePoint, bool) { return []RestorePoint{}, true }
|
||||
func offsiteWith(app string, at time.Time) func(context.Context) (OffsiteInventory, error) {
|
||||
return func(context.Context) (OffsiteInventory, error) {
|
||||
return OffsiteInventory{Apps: []OffsiteInventoryApp{{App: app, LatestAt: at}}}, nil
|
||||
}
|
||||
}
|
||||
func noOffsiteTarget(context.Context) (OffsiteInventory, error) {
|
||||
return OffsiteInventory{}, errNoOffsiteTarget
|
||||
}
|
||||
|
||||
func tiersOf(ps []UpdateTierPoint) []int {
|
||||
out := make([]int, 0, len(ps))
|
||||
for _, p := range ps {
|
||||
out = append(out, p.Tier)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// G — the order is 2, 1, 3, and the walk STOPS at the first accepted copy (so a box with a fresh
|
||||
// second-drive copy never reaches the network).
|
||||
func TestR475_G_TierOrderIsSecondDriveThenOwnUnitThenOffsite(t *testing.T) {
|
||||
m, _ := r475Manager()
|
||||
offsiteCalls := 0
|
||||
m.updateTier2PointFn = tier2At(r475T0.Add(-1 * time.Hour))
|
||||
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
|
||||
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
|
||||
offsiteCalls++
|
||||
return offsiteWith("gokapi", r475T0.Add(-3*time.Hour))(ctx)
|
||||
}
|
||||
ctx := context.Background()
|
||||
|
||||
p, ok, seen := m.UpdateRestorePoints(ctx, "gokapi", nil)
|
||||
if !ok || p.Tier != UpdateTierSecondDrive || fmt.Sprint(tiersOf(seen)) != "[2]" || offsiteCalls != 0 {
|
||||
t.Errorf("all three present: want tier 2 and a stop; got %+v ok=%v seen=%v offsite calls=%d", p, ok, tiersOf(seen), offsiteCalls)
|
||||
}
|
||||
p, ok, seen = m.UpdateRestorePoints(ctx, "gokapi", func(p UpdateTierPoint) bool { return p.Tier != UpdateTierSecondDrive })
|
||||
if !ok || p.Tier != UpdateTierLocal || fmt.Sprint(tiersOf(seen)) != "[2 1]" {
|
||||
t.Errorf("tier 2 refused: want tier 1; got %+v seen=%v", p, tiersOf(seen))
|
||||
}
|
||||
p, ok, seen = m.UpdateRestorePoints(ctx, "gokapi", func(p UpdateTierPoint) bool { return p.Tier == UpdateTierOffsite })
|
||||
if !ok || p.Tier != UpdateTierOffsite || !p.At.Equal(r475T0.Add(-3*time.Hour)) || fmt.Sprint(tiersOf(seen)) != "[2 1 3]" {
|
||||
t.Errorf("tiers 2 and 1 refused: want tier 3; got %+v seen=%v", p, tiersOf(seen))
|
||||
}
|
||||
if _, ok, seen = m.UpdateRestorePoints(ctx, "gokapi", func(UpdateTierPoint) bool { return false }); ok || len(seen) != 3 {
|
||||
t.Errorf("nothing accepted: want not found with all three seen; ok=%v seen=%v", ok, tiersOf(seen))
|
||||
}
|
||||
}
|
||||
|
||||
func TestR475_H_OwnUnitOnly(t *testing.T) {
|
||||
m, _ := r475Manager()
|
||||
m.updateTier2PointFn = noTier2
|
||||
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
|
||||
m.updateOffsiteInvFn = noOffsiteTarget
|
||||
p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
|
||||
if !ok || p.Tier != UpdateTierLocal || !p.At.Equal(r475T0.Add(-2*time.Hour)) {
|
||||
t.Fatalf("got %+v ok=%v", p, ok)
|
||||
}
|
||||
// A Tier-2 record that was only ATTEMPTED is not a copy (R-101) — the own unit still wins.
|
||||
m.updateTier2PointFn = func(string) (Tier2RestorePoint, error) {
|
||||
return Tier2RestorePoint{Restorable: true, CopyDateProven: false}, nil
|
||||
}
|
||||
if p, ok, _ = m.UpdateRestorePoints(context.Background(), "gokapi", nil); !ok || p.Tier != UpdateTierLocal {
|
||||
t.Errorf("an unproven Tier-2 record must be skipped; got %+v", p)
|
||||
}
|
||||
// No unit on disk (ListRestorePoints' empty list) is no copy.
|
||||
m.updateTier1PointsFn = noTier1
|
||||
if _, ok, _ = m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok {
|
||||
t.Error("no copy on any tier must be not found")
|
||||
}
|
||||
}
|
||||
|
||||
func TestR475_I_OffsiteOnly(t *testing.T) {
|
||||
m, _ := r475Manager()
|
||||
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
|
||||
m.updateOffsiteInvFn = offsiteWith("gokapi", r475T0.Add(-5*time.Hour))
|
||||
if p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); !ok || p.Tier != UpdateTierOffsite {
|
||||
t.Fatalf("got %+v ok=%v", p, ok)
|
||||
}
|
||||
// Another app's snapshot is not this app's copy.
|
||||
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "nextcloud", nil); ok {
|
||||
t.Error("a snapshot tagged for a different app must not count")
|
||||
}
|
||||
}
|
||||
|
||||
// J — an unreachable off-site repository is ABSENT with a WARN, and it is bounded in time.
|
||||
func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
|
||||
m, buf := r475Manager()
|
||||
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
|
||||
m.updateOffsiteInvFn = func(context.Context) (OffsiteInventory, error) {
|
||||
return OffsiteInventory{}, errors.New("ssh: connect to host: connection timed out")
|
||||
}
|
||||
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok {
|
||||
t.Error("an unreachable off-site copy must count as absent")
|
||||
}
|
||||
if !strings.Contains(buf.String(), "[WARN]") || !strings.Contains(buf.String(), "counted as ABSENT") {
|
||||
t.Errorf("an unreachable off-site copy must WARN; log = %q", buf.String())
|
||||
}
|
||||
|
||||
// Control: a box with NO off-site target is plainly absent — that is not a fault, so no WARN.
|
||||
buf.Reset()
|
||||
m.updateOffsiteInvFn = noOffsiteTarget
|
||||
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok || strings.Contains(buf.String(), "WARN") {
|
||||
t.Errorf("no off-site target: want absent and silent; ok=%v log=%q", ok, buf.String())
|
||||
}
|
||||
|
||||
// The bound: a repository that never answers is given up on at the timeout.
|
||||
old := updateOffsiteCheckTimeout
|
||||
updateOffsiteCheckTimeout = 50 * time.Millisecond
|
||||
defer func() { updateOffsiteCheckTimeout = old }()
|
||||
buf.Reset()
|
||||
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
|
||||
<-ctx.Done()
|
||||
return OffsiteInventory{}, ctx.Err()
|
||||
}
|
||||
start := time.Now()
|
||||
_, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
|
||||
if took := time.Since(start); ok || took > 5*time.Second || !strings.Contains(buf.String(), "counted as ABSENT") {
|
||||
t.Errorf("a hanging off-site check must end at its bound as absent with a WARN; ok=%v took=%s log=%q", ok, took, buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The hold names the tier. The labels are the ruling's exact words.
|
||||
func TestR475_HoldTextNamesTheTier(t *testing.T) {
|
||||
at := time.Date(2026, 9, 13, 8, 0, 0, 0, time.UTC)
|
||||
copyAt := time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)
|
||||
for tier, label := range map[int]string{
|
||||
UpdateTierSecondDrive: "második meghajtó",
|
||||
UpdateTierLocal: "saját meghajtó",
|
||||
UpdateTierOffsite: "távoli mentés",
|
||||
} {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
if err := m.HoldAfterFailedUpdate("gokapi", at, copyAt, tier); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
held, why := m.RestoreHoldFor("gokapi")
|
||||
want := fmt.Sprintf(UpdateHoldFmt, "gokapi", "2026-09-13 10:00", label, "2026-09-13 03:30")
|
||||
if !held || why != want {
|
||||
t.Errorf("tier %d: hold text =\n%q\nwant\n%q", tier, why, want)
|
||||
}
|
||||
if !strings.HasSuffix(why, "ebből a biztonsági mentésből: "+label+", 2026-09-13 03:30.") {
|
||||
t.Errorf("tier %d: the sentence must END naming the tier and the date, got %q", tier, why)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestR475_AHoldWrittenBeforeTheTierKeepsItsSentence(t *testing.T) {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "uptime-kuma", At: "2026-09-13T10:18:02Z",
|
||||
Reason: settings.HoldReasonUpdateFailed, CopyDate: "2026-09-13T10:09:51Z"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, why := m.RestoreHoldFor("uptime-kuma")
|
||||
// The exact sentence quoted in audits/slice4-2026-09-13 (13-H-refusals.txt), live on v0.238.1.
|
||||
if want := fmt.Sprintf(UpdateHoldLegacyFmt, "uptime-kuma", "2026-09-13 12:18", "2026-09-13 12:09"); why != want {
|
||||
t.Errorf("a v0.238.1 hold must keep its sentence:\n%q\nwant\n%q", why, want)
|
||||
}
|
||||
}
|
||||
|
||||
// RunAppBackupNow's tail: a Tier-2 failure no longer fails "back up first", and the own unit it
|
||||
// captured reads as fresh even when the capture found nothing to rewrite.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): aim the Chtimes at a path that does not exist — this test fails on
|
||||
// the mtime: "back up first" would then leave a quiet app's own unit as old as its last definition change.
|
||||
func TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh(t *testing.T) {
|
||||
nsRoot := t.TempDir()
|
||||
mp := RecoveryUnitManifestPath(nsRoot, "gokapi")
|
||||
if err := os.MkdirAll(filepath.Dir(mp), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(mp, []byte(`{"app_name":"gokapi"}`), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
old := time.Now().Add(-30 * time.Hour)
|
||||
if err := os.Chtimes(mp, old, old); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m, buf := r475Manager()
|
||||
tier2Ran := false
|
||||
m.perAppTier2 = func(string) error { tier2Ran = true; return errors.New("no second drive with room") }
|
||||
|
||||
now := time.Now()
|
||||
m.updatePreBackupTail("gokapi", nsRoot, now)
|
||||
|
||||
if !tier2Ran {
|
||||
t.Fatal("the Tier-2 copy must still be attempted")
|
||||
}
|
||||
if !strings.Contains(buf.String(), "[WARN]") || !strings.Contains(buf.String(), "Tier 2 copy FAILED") {
|
||||
t.Errorf("a Tier-2 failure must be logged as a WARN; log = %q", buf.String())
|
||||
}
|
||||
fi, err := os.Stat(mp)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if d := fi.ModTime().Sub(now); d < -time.Second || d > time.Second {
|
||||
t.Errorf("the own unit must read as proven NOW (mtime %s, now %s)", fi.ModTime(), now)
|
||||
}
|
||||
|
||||
// Control: the same unit through ListRestorePoints' own rule reads fresh — the newest artifact.
|
||||
buf.Reset()
|
||||
m.perAppTier2 = func(string) error { return nil }
|
||||
m.updatePreBackupTail("gokapi", nsRoot, now)
|
||||
if strings.Contains(buf.String(), "WARN") {
|
||||
t.Errorf("a successful Tier-2 copy must not WARN; log = %q", buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The tail is really what RunAppBackupNow runs — a helper tested and never called is the "seam built
|
||||
// but never wired" shape this project has shipped repeatedly.
|
||||
func TestR475_RunAppBackupNowUsesTheTolerantTail(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "update_guard.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var calls []string
|
||||
found := false
|
||||
for _, d := range f.Decls {
|
||||
fn, ok := d.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "RunAppBackupNow" {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
ast.Inspect(fn.Body, func(n ast.Node) bool {
|
||||
if sel, ok := n.(*ast.SelectorExpr); ok {
|
||||
calls = append(calls, sel.Sel.Name)
|
||||
}
|
||||
return true
|
||||
})
|
||||
}
|
||||
joined := " " + strings.Join(calls, " ") + " "
|
||||
if !found || !strings.Contains(joined, " updatePreBackupTail ") {
|
||||
t.Fatalf("RunAppBackupNow must end in updatePreBackupTail; selectors = %s", joined)
|
||||
}
|
||||
if strings.Contains(joined, " RunTier2 ") || strings.Contains(joined, " perAppTier2 ") {
|
||||
t.Error("RunAppBackupNow must not run the Tier-2 copy itself any more — only through the tolerant tail")
|
||||
}
|
||||
}
|
||||
|
||||
func TestR475_CanBackUpApp(t *testing.T) {
|
||||
var nilM *Manager
|
||||
if ok, why := nilM.CanBackUpApp("gokapi"); ok || why == "" {
|
||||
t.Error("no backup manager: cannot back up, with a reason")
|
||||
}
|
||||
m, _ := r475Manager()
|
||||
if ok, why := m.CanBackUpApp("gokapi"); ok || !strings.Contains(why, "stack provider") {
|
||||
t.Errorf("no stack provider: cannot back up; got ok=%v why=%q", ok, why)
|
||||
}
|
||||
}
|
||||
|
||||
// Tier 3's route back is the off-site restore, and it never went through RestoreFromRecoveryUnitAt —
|
||||
// so it must lift an update hold itself, or a hold naming „távoli mentés" could never be cleared.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): remove the clearUpdateHoldAfterRestore call from
|
||||
// ReconstituteFromOffsite — this test fails with the hold still in place.
|
||||
func TestR475_OffsiteRestoreClearsAnUpdateHold(t *testing.T) {
|
||||
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", "")
|
||||
prov.composePath = writeLiveCompose(t, noDBCompose)
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return nil, nil }
|
||||
if err := m.HoldAfterFailedUpdate("immich", r475T0, r475T0.Add(-time.Hour), UpdateTierOffsite); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if held, why := m.RestoreHoldFor("immich"); !held || !strings.Contains(why, "távoli mentés") {
|
||||
t.Fatalf("fixture: the app must be held naming the off-site copy, got held=%v %q", held, why)
|
||||
}
|
||||
if _, err := m.ReconstituteFromOffsite(context.Background(), "immich", false); err != nil {
|
||||
t.Fatalf("reconstitute: %v", err)
|
||||
}
|
||||
if held, _ := m.RestoreHoldFor("immich"); held {
|
||||
t.Error("a successful off-site restore is the route back the hold names — the hold must be lifted")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user