controller v0.239.0: any backup tier lets an app update (R-475)
gates / gates (push) Successful in 14s

Operator ruling 2026-09-13. The update precondition walks Tier 2, Tier 1
(own recovery unit, "helyi") and Tier 3 (off-site, 15 s bound; unreachable
counts as absent with a WARN) and leans on the first FRESH copy; the
backup_max_age rule applies to whichever tier is chosen. No copy anywhere:
back up first. Refused only when nothing exists and no backup can be taken.
RunAppBackupNow tolerates a Tier-2 failure (WARN) and marks the captured
unit proven current. The hold names the tier (második meghajtó / saját
meghajtó / távoli mentés) and the date; pre-v0.239.0 holds keep their text.
A successful off-site restore now lifts an update hold. The backups page
still uses Tier2UnitRestorePoint unchanged.

Scenarios G-M tested; red-proofs M, L, the tail and the off-site clear in
felhom.eu documentation/audits/rulings-r472-r475-2026-09-13/.
This commit is contained in:
2026-09-13 17:16:24 +02:00
parent f946b0d0ca
commit b93c1543da
16 changed files with 1028 additions and 114 deletions
+7
View File
@@ -173,6 +173,13 @@ type Manager struct {
// updatingCheck (slice 4) — nil-safe; see isHeld / SetUpdatingCheck.
updatingCheck func(stackName string) bool
// R-475 update-precondition seams, one per tier. Nil → the real Tier2UnitRestorePoint /
// ListRestorePoints / OffsiteInventoryList. They let a test reach Tier 1 and Tier 3 without a drive
// or a restic repository; see UpdateRestorePoints.
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
updateOffsiteInvFn func(ctx context.Context) (OffsiteInventory, error)
// R-354 volume-REPLAY seam — the mirror of the F17 DB seams above, so the off-site path's new
// volume leg is unit-testable without Docker. Nil → the real restoreDockerVolumesFrom.
volumeReplayFrom func(stackName, dumpDir string) (int, error)
@@ -344,7 +344,11 @@ func (m *Manager) RestoreHoldFor(stack string) (bool, string) {
if h.CopyDate != "" {
copyDate = fmtHoldTime(h.CopyDate)
}
return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), copyDate)
// R-475: name the tier when the hold recorded one; an older hold keeps its own sentence.
if label := UpdateTierLabel(h.CopyTier); label != "" && h.CopyDate != "" {
return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), label, copyDate)
}
return true, fmt.Sprintf(UpdateHoldLegacyFmt, stack, fmtHoldTime(h.At), copyDate)
}
when := h.At
if t, err := time.Parse(time.RFC3339, h.At); err == nil {
@@ -824,6 +828,11 @@ func (m *Manager) ReconstituteFromOffsite(ctx context.Context, stack string, ack
if err := restartStack(); err != nil {
return res, fmt.Errorf("a(z) %s újraindítása sikertelen a fájlok visszaállítása után: %w", stack, err)
}
// R-475: an update hold may now name the OFF-SITE copy, and this is the route back it names. The
// unit restore clears it in RestoreFromRecoveryUnitAt; this path never went through that function,
// so without this line a successful off-site restore would leave the app refusing its next start.
// Pinned by TestR475_OffsiteRestoreClearsAnUpdateHold.
m.clearUpdateHoldAfterRestore(stack)
if err := m.waitForHealthy(stack, 90*time.Second); err != nil {
m.logger.Printf("[WARN] [offbox] %s reconstituted but health check failed: %v", stack, err)
}
@@ -68,7 +68,7 @@ func TestSlice4_UpdateHoldTextNamesTheTimeAndTheCopy(t *testing.T) {
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
at := time.Date(2026, 9, 13, 8, 0, 0, 0, time.UTC)
copyAt := time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)
if err := m.HoldAfterFailedUpdate("bookstack", at, copyAt); err != nil {
if err := m.HoldAfterFailedUpdate("bookstack", at, copyAt, UpdateTierSecondDrive); err != nil {
t.Fatal(err)
}
h, ok := sett.GetRestoreHold("bookstack")
@@ -77,7 +77,7 @@ func TestSlice4_UpdateHoldTextNamesTheTimeAndTheCopy(t *testing.T) {
}
held, why := m.RestoreHoldFor("bookstack")
// Budapest is UTC+2 in September: 08:00Z → 10:00, 01:30Z → 03:30.
want := fmt.Sprintf(UpdateHoldFmt, "bookstack", "2026-09-13 10:00", "2026-09-13 03:30")
want := fmt.Sprintf(UpdateHoldFmt, "bookstack", "2026-09-13 10:00", "második meghajtó", "2026-09-13 03:30")
if !held || why != want {
t.Errorf("hold text =\n%q\nwant\n%q", why, want)
}
@@ -98,7 +98,7 @@ func TestSlice4_RestoreHoldTextIsUnchanged(t *testing.T) {
func TestSlice4_ASuccessfulRestoreClearsOnlyAnUpdateHold(t *testing.T) {
sett := slice4Settings(t)
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
_ = m.HoldAfterFailedUpdate("upd", time.Now(), time.Now())
_ = m.HoldAfterFailedUpdate("upd", time.Now(), time.Now(), UpdateTierLocal)
_ = sett.SetRestoreHold(settings.RestoreHold{Stack: "rst", At: "2026-08-22T14:00:00Z"})
m.clearUpdateHoldAfterRestore("upd")
m.clearUpdateHoldAfterRestore("rst")
@@ -136,7 +136,7 @@ func TestSlice4_UpdateBusy(t *testing.T) {
func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) {
h := newAdmissionHarness(t, "held", "free")
h.m.settings = slice4Settings(t)
if err := h.m.HoldAfterFailedUpdate("held", time.Now(), time.Now()); err != nil {
if err := h.m.HoldAfterFailedUpdate("held", time.Now(), time.Now(), UpdateTierSecondDrive); err != nil {
t.Fatal(err)
}
h.m.runVolumeDumps()
+183 -9
View File
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"time"
@@ -93,6 +94,149 @@ func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
return t, true
}
// ── R-475: any backup tier lets an app update (operator ruling 2026-09-13, controller v0.239.0) ────
//
// Until v0.239.0 the update's precondition was Tier2UnitRestorePoint alone, so an app with no second
// drive could never be updated — even with a fresh recovery unit on its own drive and an off-site
// snapshot from last night. The ruling: every backup counts. Tier2UnitRestorePoint itself is NOT
// changed; the backups page still calls it for the „Teljes visszaállítás" action, which really does
// restore from the second drive only.
// Backup tiers, as the update precondition and the hold sentence name them.
const (
UpdateTierLocal = 1 // the app's own recovery unit on its drive — „helyi" on the restore page
UpdateTierSecondDrive = 2 // the Tier-2 mirror on another drive
UpdateTierOffsite = 3 // the off-site restic repository
)
// updateTierOrder is the preference order the ruling set: the second drive, then the app's own unit,
// then off-site. The first tier holding a copy the caller ACCEPTS is chosen.
var updateTierOrder = []int{UpdateTierSecondDrive, UpdateTierLocal, UpdateTierOffsite}
// UpdateTierLabel is a tier's name in the customer's hold sentence. "" for an unknown tier.
func UpdateTierLabel(tier int) string {
switch tier {
case UpdateTierSecondDrive:
return "második meghajtó"
case UpdateTierLocal:
return "saját meghajtó"
case UpdateTierOffsite:
return "távoli mentés"
}
return ""
}
// updateOffsiteCheckTimeout bounds the off-site lookup. An update must not stall on an unreachable
// Storage Box: past this the off-site copy counts as ABSENT (with a WARN), and the update carries on
// with backing up first. A var only so a test can shorten it.
var updateOffsiteCheckTimeout = 15 * time.Second
// UpdateTierPoint is one proven, restorable copy of an app on one tier.
type UpdateTierPoint struct {
Tier int
// At is when the data in that copy was last proven written: Tier 2 ProvenCopyTime, Tier 1 the
// newest artifact of the unit (ListRestorePoints), Tier 3 the newest snapshot for the app.
At time.Time
}
// UpdateRestorePoints walks the tiers in preference order (2, 1, 3) and returns the FIRST copy that
// accept admits (nil accepts any), whether one was found, and every copy it looked at on the way.
//
// It stops at the first accepted copy, so a box with a fresh second-drive copy never touches the
// network. The AGE rule is the caller's (stacks applies backup_max_age through accept) — that is what
// makes "the age rule applies to whichever tier is chosen" one rule, not three (R-475 Scenario M).
func (m *Manager) UpdateRestorePoints(ctx context.Context, stackName string, accept func(UpdateTierPoint) bool) (UpdateTierPoint, bool, []UpdateTierPoint) {
var seen []UpdateTierPoint
for _, tier := range updateTierOrder {
p, ok := m.updateTierPoint(ctx, stackName, tier)
if !ok {
continue
}
seen = append(seen, p)
if accept == nil || accept(p) {
return p, true, seen
}
}
return UpdateTierPoint{}, false, seen
}
func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier int) (UpdateTierPoint, bool) {
switch tier {
case UpdateTierSecondDrive:
get := m.updateTier2PointFn
if get == nil {
get = m.Tier2UnitRestorePoint
}
rp, err := get(stackName)
if err != nil {
if m.isDebug() {
m.logger.Printf("[DEBUG] [backup] update precondition for %s: no Tier-2 copy (%v)", stackName, err)
}
return UpdateTierPoint{}, false
}
at, ok := rp.ProvenCopyTime()
return UpdateTierPoint{Tier: tier, At: at}, ok
case UpdateTierLocal:
list := m.updateTier1PointsFn
if list == nil {
list = m.ListRestorePoints
}
pts, _ := list(stackName)
for _, rp := range pts {
if at, err := time.Parse(time.RFC3339, rp.Time); err == nil {
return UpdateTierPoint{Tier: tier, At: at}, true
}
}
return UpdateTierPoint{}, false
case UpdateTierOffsite:
inv := m.updateOffsiteInvFn
if inv == nil {
if m.settings == nil || !m.OffboxConfigured() {
return UpdateTierPoint{}, false
}
inv = m.OffsiteInventoryList
}
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
defer cancel()
got, err := inv(cctx)
if err != nil {
if !errors.Is(err, errNoOffsiteTarget) {
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
}
return UpdateTierPoint{}, false
}
for _, a := range got.Apps {
if a.App == stackName && !a.LatestAt.IsZero() {
return UpdateTierPoint{Tier: tier, At: a.LatestAt}, true
}
}
}
return UpdateTierPoint{}, false
}
// CanBackUpApp reports whether "back up first" can run for this app at all right now — the second
// half of R-475 Scenario L: an app with no copy anywhere is refused only when this is false too.
// Cheap and read-only; RunAppBackupNow re-checks everything when it actually runs.
func (m *Manager) CanBackUpApp(stackName string) (bool, string) {
if m == nil {
return false, "backup is not enabled on this box"
}
if m.stackProvider == nil {
return false, "stack provider not configured"
}
if m.migrationActive() {
return false, "a data migration is running"
}
drivePath := m.GetAppDrivePath(stackName)
if drivePath == "" || !filepath.IsAbs(drivePath) {
return false, "the app's drive cannot be resolved"
}
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
return false, fmt.Sprintf("the app's drive %s is not available", drivePath)
}
return true, ""
}
// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an
// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer
// sentence is chosen by the caller.
@@ -142,6 +286,7 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName)
start := time.Now()
var nsRoot string
legErr := func() error {
defer m.releaseRunning()
defer m.beginAdmissionRun()()
@@ -156,7 +301,7 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
if !m.admitApp(stackName) {
return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath)
}
nsRoot := m.namespaceRoot(drivePath)
nsRoot = m.namespaceRoot(drivePath)
discover := m.discoverDBs
if discover == nil {
@@ -203,16 +348,35 @@ func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
return legErr
}
m.updatePreBackupTail(stackName, nsRoot, time.Now())
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
return nil
}
// updatePreBackupTail is what "back up first" does after the capture succeeded (R-475).
//
// 1. It marks the app's OWN unit as proven current NOW. CaptureRecoveryUnit leaves the manifest alone
// when nothing changed (the checksum skip), and Tier 1's age is the newest artifact's mtime — so on
// an app with no database and no named volume a fresh "back up first" would leave Tier 1 as old as
// its last definition change, and the update would be refused forever. That is the trap
// ProvenCopyTime documents for Tier 2, one tier down. The capture has just compared the unit with
// the live definition, so "current as of now" is exactly what it established.
// 2. It runs the Tier-2 copy, and a Tier-2 failure is a WARN, not a failure: the update may lean on
// any tier, and the Tier-1 unit it can lean on was just written. Pinned by
// TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh.
func (m *Manager) updatePreBackupTail(stackName, nsRoot string, now time.Time) {
if nsRoot != "" {
if err := os.Chtimes(RecoveryUnitManifestPath(nsRoot, stackName), now, now); err != nil {
m.logger.Printf("[WARN] [backup] update pre-backup for %s: could not mark the recovery unit as proven current (%v) — its own-unit copy may read older than it is", stackName, err)
}
}
runOne := m.perAppTier2
if runOne == nil {
runOne = m.RunTier2
}
if err := runOne(stackName); err != nil {
m.logger.Printf("[ERROR] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v", stackName, err)
return fmt.Errorf("a másodlagos másolat elkészítése sikertelen: %w", err)
m.logger.Printf("[WARN] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v — not fatal: the app's own recovery unit was just captured, and an update may lean on any tier (R-475)", stackName, err)
}
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
return nil
}
// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the
@@ -244,9 +408,18 @@ func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) (
}
// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app,
// the time of the failure, and the PROVEN date of the copy it can be restored from. One named string so
// a test asserts it verbatim instead of retyping Hungarian (R-364).
// the time of the failure, the TIER of the copy it can be restored from (UpdateTierLabel), and that
// copy's PROVEN date. One named string so a test asserts it verbatim instead of retyping Hungarian
// (R-364). Since v0.239.0 (R-475) it names the tier: the copy may be on any of three, and each is
// restored from a different place on the Mentések page.
const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a Mentések oldalon ebből a biztonsági mentésből: %s, %s."
// UpdateHoldLegacyFmt is the v0.237.0–v0.238.1 sentence, kept for a hold written before the tier was
// recorded (CopyTier 0) — every such hold named a Tier-2 copy, but it did not SAY so, and rewriting
// it now would state a fact the record does not hold.
const UpdateHoldLegacyFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
"Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon."
@@ -276,7 +449,7 @@ func fmtHoldTime(rfc3339 string) string {
// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
// failure on the page.
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time) error {
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time, copyTier int) error {
if m == nil || m.settings == nil {
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
}
@@ -287,11 +460,12 @@ func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate
}
if !copyDate.IsZero() {
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
h.CopyTier = copyTier
}
if err := m.settings.SetRestoreHold(h); err != nil {
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
}
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: %s)", stackName, h.CopyDate)
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: tier %d %q, %s)", stackName, h.CopyTier, UpdateTierLabel(h.CopyTier), h.CopyDate)
return nil
}
@@ -0,0 +1,317 @@
package backup
import (
"bytes"
"context"
"errors"
"fmt"
"go/ast"
"go/parser"
"go/token"
"io"
"log"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// R-475 — the backup side of "any backup tier lets an app update" (controller v0.239.0).
var r475T0 = time.Date(2026, 9, 13, 12, 0, 0, 0, time.UTC)
func r475Manager() (*Manager, *bytes.Buffer) {
var buf bytes.Buffer
return &Manager{logger: log.New(&buf, "", 0)}, &buf
}
func tier2At(at time.Time) func(string) (Tier2RestorePoint, error) {
return func(string) (Tier2RestorePoint, error) {
ts := at.UTC().Format(time.RFC3339)
return Tier2RestorePoint{Restorable: true, CopyDateProven: true, CopyLastSuccess: ts, CopyDate: ts}, nil
}
}
func noTier2(string) (Tier2RestorePoint, error) {
return Tier2RestorePoint{}, errors.New("no Tier-2 copy recorded for this app")
}
func tier1At(at time.Time) func(string) ([]RestorePoint, bool) {
return func(string) ([]RestorePoint, bool) {
return []RestorePoint{{Time: at.UTC().Format(time.RFC3339), ShortID: "helyi", Tier: 1}}, true
}
}
func noTier1(string) ([]RestorePoint, bool) { return []RestorePoint{}, true }
func offsiteWith(app string, at time.Time) func(context.Context) (OffsiteInventory, error) {
return func(context.Context) (OffsiteInventory, error) {
return OffsiteInventory{Apps: []OffsiteInventoryApp{{App: app, LatestAt: at}}}, nil
}
}
func noOffsiteTarget(context.Context) (OffsiteInventory, error) {
return OffsiteInventory{}, errNoOffsiteTarget
}
func tiersOf(ps []UpdateTierPoint) []int {
out := make([]int, 0, len(ps))
for _, p := range ps {
out = append(out, p.Tier)
}
return out
}
// G — the order is 2, 1, 3, and the walk STOPS at the first accepted copy (so a box with a fresh
// second-drive copy never reaches the network).
func TestR475_G_TierOrderIsSecondDriveThenOwnUnitThenOffsite(t *testing.T) {
m, _ := r475Manager()
offsiteCalls := 0
m.updateTier2PointFn = tier2At(r475T0.Add(-1 * time.Hour))
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
offsiteCalls++
return offsiteWith("gokapi", r475T0.Add(-3*time.Hour))(ctx)
}
ctx := context.Background()
p, ok, seen := m.UpdateRestorePoints(ctx, "gokapi", nil)
if !ok || p.Tier != UpdateTierSecondDrive || fmt.Sprint(tiersOf(seen)) != "[2]" || offsiteCalls != 0 {
t.Errorf("all three present: want tier 2 and a stop; got %+v ok=%v seen=%v offsite calls=%d", p, ok, tiersOf(seen), offsiteCalls)
}
p, ok, seen = m.UpdateRestorePoints(ctx, "gokapi", func(p UpdateTierPoint) bool { return p.Tier != UpdateTierSecondDrive })
if !ok || p.Tier != UpdateTierLocal || fmt.Sprint(tiersOf(seen)) != "[2 1]" {
t.Errorf("tier 2 refused: want tier 1; got %+v seen=%v", p, tiersOf(seen))
}
p, ok, seen = m.UpdateRestorePoints(ctx, "gokapi", func(p UpdateTierPoint) bool { return p.Tier == UpdateTierOffsite })
if !ok || p.Tier != UpdateTierOffsite || !p.At.Equal(r475T0.Add(-3*time.Hour)) || fmt.Sprint(tiersOf(seen)) != "[2 1 3]" {
t.Errorf("tiers 2 and 1 refused: want tier 3; got %+v seen=%v", p, tiersOf(seen))
}
if _, ok, seen = m.UpdateRestorePoints(ctx, "gokapi", func(UpdateTierPoint) bool { return false }); ok || len(seen) != 3 {
t.Errorf("nothing accepted: want not found with all three seen; ok=%v seen=%v", ok, tiersOf(seen))
}
}
func TestR475_H_OwnUnitOnly(t *testing.T) {
m, _ := r475Manager()
m.updateTier2PointFn = noTier2
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
m.updateOffsiteInvFn = noOffsiteTarget
p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
if !ok || p.Tier != UpdateTierLocal || !p.At.Equal(r475T0.Add(-2*time.Hour)) {
t.Fatalf("got %+v ok=%v", p, ok)
}
// A Tier-2 record that was only ATTEMPTED is not a copy (R-101) — the own unit still wins.
m.updateTier2PointFn = func(string) (Tier2RestorePoint, error) {
return Tier2RestorePoint{Restorable: true, CopyDateProven: false}, nil
}
if p, ok, _ = m.UpdateRestorePoints(context.Background(), "gokapi", nil); !ok || p.Tier != UpdateTierLocal {
t.Errorf("an unproven Tier-2 record must be skipped; got %+v", p)
}
// No unit on disk (ListRestorePoints' empty list) is no copy.
m.updateTier1PointsFn = noTier1
if _, ok, _ = m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok {
t.Error("no copy on any tier must be not found")
}
}
func TestR475_I_OffsiteOnly(t *testing.T) {
m, _ := r475Manager()
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
m.updateOffsiteInvFn = offsiteWith("gokapi", r475T0.Add(-5*time.Hour))
if p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); !ok || p.Tier != UpdateTierOffsite {
t.Fatalf("got %+v ok=%v", p, ok)
}
// Another app's snapshot is not this app's copy.
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "nextcloud", nil); ok {
t.Error("a snapshot tagged for a different app must not count")
}
}
// J — an unreachable off-site repository is ABSENT with a WARN, and it is bounded in time.
func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
m, buf := r475Manager()
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
m.updateOffsiteInvFn = func(context.Context) (OffsiteInventory, error) {
return OffsiteInventory{}, errors.New("ssh: connect to host: connection timed out")
}
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok {
t.Error("an unreachable off-site copy must count as absent")
}
if !strings.Contains(buf.String(), "[WARN]") || !strings.Contains(buf.String(), "counted as ABSENT") {
t.Errorf("an unreachable off-site copy must WARN; log = %q", buf.String())
}
// Control: a box with NO off-site target is plainly absent — that is not a fault, so no WARN.
buf.Reset()
m.updateOffsiteInvFn = noOffsiteTarget
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok || strings.Contains(buf.String(), "WARN") {
t.Errorf("no off-site target: want absent and silent; ok=%v log=%q", ok, buf.String())
}
// The bound: a repository that never answers is given up on at the timeout.
old := updateOffsiteCheckTimeout
updateOffsiteCheckTimeout = 50 * time.Millisecond
defer func() { updateOffsiteCheckTimeout = old }()
buf.Reset()
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
<-ctx.Done()
return OffsiteInventory{}, ctx.Err()
}
start := time.Now()
_, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
if took := time.Since(start); ok || took > 5*time.Second || !strings.Contains(buf.String(), "counted as ABSENT") {
t.Errorf("a hanging off-site check must end at its bound as absent with a WARN; ok=%v took=%s log=%q", ok, took, buf.String())
}
}
// The hold names the tier. The labels are the ruling's exact words.
func TestR475_HoldTextNamesTheTier(t *testing.T) {
at := time.Date(2026, 9, 13, 8, 0, 0, 0, time.UTC)
copyAt := time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)
for tier, label := range map[int]string{
UpdateTierSecondDrive: "második meghajtó",
UpdateTierLocal: "saját meghajtó",
UpdateTierOffsite: "távoli mentés",
} {
sett := slice4Settings(t)
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
if err := m.HoldAfterFailedUpdate("gokapi", at, copyAt, tier); err != nil {
t.Fatal(err)
}
held, why := m.RestoreHoldFor("gokapi")
want := fmt.Sprintf(UpdateHoldFmt, "gokapi", "2026-09-13 10:00", label, "2026-09-13 03:30")
if !held || why != want {
t.Errorf("tier %d: hold text =\n%q\nwant\n%q", tier, why, want)
}
if !strings.HasSuffix(why, "ebből a biztonsági mentésből: "+label+", 2026-09-13 03:30.") {
t.Errorf("tier %d: the sentence must END naming the tier and the date, got %q", tier, why)
}
}
}
func TestR475_AHoldWrittenBeforeTheTierKeepsItsSentence(t *testing.T) {
sett := slice4Settings(t)
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "uptime-kuma", At: "2026-09-13T10:18:02Z",
Reason: settings.HoldReasonUpdateFailed, CopyDate: "2026-09-13T10:09:51Z"}); err != nil {
t.Fatal(err)
}
_, why := m.RestoreHoldFor("uptime-kuma")
// The exact sentence quoted in audits/slice4-2026-09-13 (13-H-refusals.txt), live on v0.238.1.
if want := fmt.Sprintf(UpdateHoldLegacyFmt, "uptime-kuma", "2026-09-13 12:18", "2026-09-13 12:09"); why != want {
t.Errorf("a v0.238.1 hold must keep its sentence:\n%q\nwant\n%q", why, want)
}
}
// RunAppBackupNow's tail: a Tier-2 failure no longer fails "back up first", and the own unit it
// captured reads as fresh even when the capture found nothing to rewrite.
//
// COMPANION RED-PROOF (REPORT.md): aim the Chtimes at a path that does not exist — this test fails on
// the mtime: "back up first" would then leave a quiet app's own unit as old as its last definition change.
func TestR475_PreBackupTail_Tier2FailureIsAWarnAndTheOwnUnitIsFresh(t *testing.T) {
nsRoot := t.TempDir()
mp := RecoveryUnitManifestPath(nsRoot, "gokapi")
if err := os.MkdirAll(filepath.Dir(mp), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(mp, []byte(`{"app_name":"gokapi"}`), 0o644); err != nil {
t.Fatal(err)
}
old := time.Now().Add(-30 * time.Hour)
if err := os.Chtimes(mp, old, old); err != nil {
t.Fatal(err)
}
m, buf := r475Manager()
tier2Ran := false
m.perAppTier2 = func(string) error { tier2Ran = true; return errors.New("no second drive with room") }
now := time.Now()
m.updatePreBackupTail("gokapi", nsRoot, now)
if !tier2Ran {
t.Fatal("the Tier-2 copy must still be attempted")
}
if !strings.Contains(buf.String(), "[WARN]") || !strings.Contains(buf.String(), "Tier 2 copy FAILED") {
t.Errorf("a Tier-2 failure must be logged as a WARN; log = %q", buf.String())
}
fi, err := os.Stat(mp)
if err != nil {
t.Fatal(err)
}
if d := fi.ModTime().Sub(now); d < -time.Second || d > time.Second {
t.Errorf("the own unit must read as proven NOW (mtime %s, now %s)", fi.ModTime(), now)
}
// Control: the same unit through ListRestorePoints' own rule reads fresh — the newest artifact.
buf.Reset()
m.perAppTier2 = func(string) error { return nil }
m.updatePreBackupTail("gokapi", nsRoot, now)
if strings.Contains(buf.String(), "WARN") {
t.Errorf("a successful Tier-2 copy must not WARN; log = %q", buf.String())
}
}
// The tail is really what RunAppBackupNow runs — a helper tested and never called is the "seam built
// but never wired" shape this project has shipped repeatedly.
func TestR475_RunAppBackupNowUsesTheTolerantTail(t *testing.T) {
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, "update_guard.go", nil, 0)
if err != nil {
t.Fatal(err)
}
var calls []string
found := false
for _, d := range f.Decls {
fn, ok := d.(*ast.FuncDecl)
if !ok || fn.Name.Name != "RunAppBackupNow" {
continue
}
found = true
ast.Inspect(fn.Body, func(n ast.Node) bool {
if sel, ok := n.(*ast.SelectorExpr); ok {
calls = append(calls, sel.Sel.Name)
}
return true
})
}
joined := " " + strings.Join(calls, " ") + " "
if !found || !strings.Contains(joined, " updatePreBackupTail ") {
t.Fatalf("RunAppBackupNow must end in updatePreBackupTail; selectors = %s", joined)
}
if strings.Contains(joined, " RunTier2 ") || strings.Contains(joined, " perAppTier2 ") {
t.Error("RunAppBackupNow must not run the Tier-2 copy itself any more — only through the tolerant tail")
}
}
func TestR475_CanBackUpApp(t *testing.T) {
var nilM *Manager
if ok, why := nilM.CanBackUpApp("gokapi"); ok || why == "" {
t.Error("no backup manager: cannot back up, with a reason")
}
m, _ := r475Manager()
if ok, why := m.CanBackUpApp("gokapi"); ok || !strings.Contains(why, "stack provider") {
t.Errorf("no stack provider: cannot back up; got ok=%v why=%q", ok, why)
}
}
// Tier 3's route back is the off-site restore, and it never went through RestoreFromRecoveryUnitAt —
// so it must lift an update hold itself, or a hold naming „távoli mentés" could never be cleared.
//
// COMPANION RED-PROOF (REPORT.md): remove the clearUpdateHoldAfterRestore call from
// ReconstituteFromOffsite — this test fails with the hold still in place.
func TestR475_OffsiteRestoreClearsAnUpdateHold(t *testing.T) {
m, prov, _ := reconFixture(t, "run1", "2026-07-19T06:00:00Z", "")
prov.composePath = writeLiveCompose(t, noDBCompose)
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) { return nil, nil }
if err := m.HoldAfterFailedUpdate("immich", r475T0, r475T0.Add(-time.Hour), UpdateTierOffsite); err != nil {
t.Fatal(err)
}
if held, why := m.RestoreHoldFor("immich"); !held || !strings.Contains(why, "távoli mentés") {
t.Fatalf("fixture: the app must be held naming the off-site copy, got held=%v %q", held, why)
}
if _, err := m.ReconstituteFromOffsite(context.Background(), "immich", false); err != nil {
t.Fatalf("reconstitute: %v", err)
}
if held, _ := m.RestoreHoldFor("immich"); held {
t.Error("a successful off-site restore is the route back the hold names — the hold must be lifted")
}
}