controller v0.240.0: seven defects from the any-tier proof and the first nightly rotation
gates / gates (push) Successful in 13s

R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record,
so the second-drive restore is no longer refused over an intact mirror.
R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps).
R-485: the backup card sizes the recovery unit and the mirror(s).
R-480: a held update's sentence leaves the card once the hold is lifted.
R-477: the update's off-site lookup is one snapshots call, no stats.
R-478: a copy older than this install's deploy does not count.
R-474: "delete backups" deletes the unit, the mirror(s) and the prefs.

Tests and red-proofs per row; evidence in felhom.eu
documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
This commit is contained in:
2026-09-13 19:26:50 +02:00
parent 0e3d831030
commit bdcbd50b42
20 changed files with 673 additions and 82 deletions
+4 -4
View File
@@ -174,11 +174,11 @@ type Manager struct {
updatingCheck func(stackName string) bool
// R-475 update-precondition seams, one per tier. Nil → the real Tier2UnitRestorePoint /
// ListRestorePoints / OffsiteInventoryList. They let a test reach Tier 1 and Tier 3 without a drive
// ListRestorePoints / OffsiteSnapshotTimes. They let a test reach Tier 1 and Tier 3 without a drive
// or a restic repository; see UpdateRestorePoints.
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
updateOffsiteInvFn func(ctx context.Context) (OffsiteInventory, error)
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
updateOffsiteTimesFn func(ctx context.Context) (map[string]time.Time, error)
// R-354 volume-REPLAY seam — the mirror of the F17 DB seams above, so the off-site path's new
// volume leg is unit-testable without Docker. Nil → the real restoreDockerVolumesFrom.
+57 -26
View File
@@ -52,19 +52,21 @@ type OffsiteInventory struct {
Empty bool
}
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
//
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
// app is in there matters more than knowing how big it is, and dropping it would under-report the
// customer's own data.
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
var inv OffsiteInventory
// offsiteNewest is one app tag's newest snapshot.
type offsiteNewest struct {
id string
at time.Time
}
// offsiteNewestPerTag runs ONE `snapshots --json` and returns the newest snapshot per app tag, and
// whether the repository opened cleanly and holds no snapshots at all. Shared by the inventory page
// and the update precondition (R-477), so the two cannot disagree about what is in the repository.
func (m *Manager) offsiteNewestPerTag(ctx context.Context) (map[string]offsiteNewest, bool, error) {
// A box can hold a recovered key and still have no off-site COORDINATES — the pristine rebuilt
// shape, before its target is re-applied. Reading the repository is impossible then, and saying so
// is the honest answer; without this guard offboxBaseArgs nil-derefs on the missing target.
if !m.OffboxConfigured() {
return inv, errNoOffsiteTarget
return nil, false, errNoOffsiteTarget
}
t := m.settings.GetOffboxTarget()
base, env := m.offboxBaseArgs(t)
@@ -72,7 +74,7 @@ func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, e
defer cancel()
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
if err != nil {
return inv, err
return nil, false, err
}
var snaps []struct {
ShortID string `json:"short_id"`
@@ -81,34 +83,63 @@ func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, e
Tags []string `json:"tags"`
}
if uerr := json.Unmarshal(out, &snaps); uerr != nil {
return inv, uerr
return nil, false, uerr
}
if len(snaps) == 0 {
inv.Empty = true
return inv, nil
return nil, true, nil
}
// Newest snapshot per tag. A snapshot may carry several tags; each names an app it belongs to.
newest := map[string]struct {
id string
at time.Time
}{}
for _, s := range snaps {
id := s.ShortID
newest := map[string]offsiteNewest{}
for _, sn := range snaps {
id := sn.ShortID
if id == "" {
id = s.ID
id = sn.ID
}
for _, tag := range s.Tags {
for _, tag := range sn.Tags {
if tag == "" {
continue
}
if cur, ok := newest[tag]; !ok || s.Time.After(cur.at) {
newest[tag] = struct {
id string
at time.Time
}{id: id, at: s.Time}
if cur, ok := newest[tag]; !ok || sn.Time.After(cur.at) {
newest[tag] = offsiteNewest{id: id, at: sn.Time}
}
}
}
return newest, false, nil
}
// OffsiteSnapshotTimes (R-477, v0.240.0) is the newest snapshot time per app — ONE `snapshots --json`,
// no per-app `stats`. It is what the update precondition needs. Measured on demo-hp 2026-09-13: going
// through OffsiteInventoryList instead, the update's check spent its whole 15 s bound on the size calls
// and the bound killed one for an unrelated app (`size of kimai's newest snapshot unknown: signal:
// killed`). Pinned by TestR477_TheUpdateOffsiteLookupRunsNoStats.
func (m *Manager) OffsiteSnapshotTimes(ctx context.Context) (map[string]time.Time, error) {
newest, _, err := m.offsiteNewestPerTag(ctx)
if err != nil {
return nil, err
}
out := make(map[string]time.Time, len(newest))
for tag, n := range newest {
out[tag] = n.at
}
return out, nil
}
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
//
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
// app is in there matters more than knowing how big it is, and dropping it would under-report the
// customer's own data.
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
var inv OffsiteInventory
newest, empty, err := m.offsiteNewestPerTag(ctx)
if err != nil {
return inv, err
}
if empty {
inv.Empty = true
return inv, nil
}
if len(newest) == 0 {
// Snapshots exist but carry no tags — not "empty", and saying so would be a lie. Report an
// empty app list without the Empty flag; the page renders the honest in-between wording.
@@ -57,6 +57,7 @@ var offsiteExempt = map[string]string{
// browsing a page refuse while a backup runs, for no safety gain: it can neither take a lock nor
// remove one.
"OffsiteInventoryList": "restic snapshots --json only; snapshots measured 2026-08-31 not to lock, and it never routes through resticStep",
"offsiteNewestPerTag": "restic snapshots --json only — the one reader behind OffsiteInventoryList and OffsiteSnapshotTimes (R-477, v0.240.0); same measurement, never routes through resticStep",
}
// offsiteReachers are the calls that mean "this function talks to the off-site repository".
@@ -0,0 +1,95 @@
package backup
import (
"fmt"
"os"
"path/filepath"
"strings"
)
// R-474 (v0.240.0) — "delete backups" on removal deletes the app's Tier-2 mirror too.
//
// Measured three times on 2026-09-13: removing an app with `remove_backups:true` deleted only its
// db-dumps directory; the recovery unit, the volume tars and the Tier-2 mirror all survived, and a
// later reinstall of the same app then leaned on the old install's unit as its "fresh" restore point
// (R-478). The unit is inside the app's own backups base and RemoveStack deletes it; the MIRROR lives
// on another drive, outside that base, so it is removed here, with its own path check.
func validMirrorStackName(n string) bool {
return n != "" && n != "." && n != ".." && n != SharesPseudoStack && !strings.ContainsAny(n, `/\`)
}
// tier2MirrorRoots are the namespace roots a Tier-2 mirror of this app can live under: the recorded
// destination, every registered drive, and the system data path (a mirror outlives a changed target).
func (m *Manager) tier2MirrorRoots(stackName string) []string {
seen := map[string]bool{}
var roots []string
add := func(r string) {
if r == "" {
return
}
r = filepath.Clean(r)
if filepath.IsAbs(r) && !seen[r] {
seen[r] = true
roots = append(roots, r)
}
}
if m.settings != nil {
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil {
add(cfg.DestinationPath)
}
for _, sp := range m.settings.GetStoragePaths() {
if sp.Path != "" {
add(NamespaceRootFor(sp.Path, m.systemDataPath))
}
}
}
if m.systemDataPath != "" {
add(NamespaceRootFor(m.systemDataPath, m.systemDataPath))
}
return roots
}
// Tier2MirrorDirsForApp lists the app's Tier-2 mirror directories that exist now. Call it BEFORE the
// removal clears the app's cross-drive record, which is one of the places it looks.
func (m *Manager) Tier2MirrorDirsForApp(stackName string) []string {
if m == nil || !validMirrorStackName(stackName) {
return nil
}
var out []string
for _, root := range m.tier2MirrorRoots(stackName) {
d := filepath.Join(root, "backups", "secondary", stackName)
if fi, err := os.Stat(d); err == nil && fi.IsDir() {
out = append(out, d)
}
}
return out
}
// RemoveTier2Mirrors deletes the given mirror directories, each only if it is exactly
// <root>/backups/secondary/<stackName> for one of the app's mirror roots — never another app's mirror,
// never the shares mirror, never a path that merely cleans to one. Returns "path (size)" per removal.
func (m *Manager) RemoveTier2Mirrors(stackName string, dirs []string) []string {
if m == nil || !validMirrorStackName(stackName) {
return nil
}
allowed := map[string]bool{}
for _, root := range m.tier2MirrorRoots(stackName) {
allowed[filepath.Join(root, "backups", "secondary", stackName)] = true
}
var removed []string
for _, d := range dirs {
if !allowed[d] {
m.logger.Printf("[WARN] [backup] remove %s: refusing to delete %q — not this app's Tier-2 mirror", stackName, d)
continue
}
size := humanizeBytes(dirSizeBytes(d))
if err := os.RemoveAll(d); err != nil {
m.logger.Printf("[ERROR] [backup] remove %s: deleting the Tier-2 mirror %s failed: %v", stackName, d, err)
continue
}
m.logger.Printf("[INFO] [backup] remove %s: Tier-2 mirror deleted: %s (%s)", stackName, d, size)
removed = append(removed, fmt.Sprintf("%s (%s)", d, size))
}
return removed
}
@@ -0,0 +1,110 @@
package backup
import (
"context"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
// R-477 — the update's Tier-3 lookup is ONE snapshots call. The control proves the fake runner does
// see stats calls when something makes them (the inventory page), so a zero is a measurement.
//
// COMPANION RED-PROOF (REPORT.md): make OffsiteSnapshotTimes read OffsiteInventoryList — this fails.
func TestR477_TheUpdateOffsiteLookupRunsNoStats(t *testing.T) {
m, _ := newOffboxManager(t)
var calls []string
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
calls = append(calls, strings.Join(args, " "))
if contains(args, "snapshots") {
return []byte(`[{"short_id":"a1","time":"2026-09-13T01:00:00Z","tags":["gokapi"]},{"short_id":"b2","time":"2026-09-13T02:00:00Z","tags":["kimai"]}]`), nil
}
return []byte(`{"total_size":1024,"total_file_count":1}`), nil
})
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
count := func(word string) int {
n := 0
for _, c := range calls {
if strings.Contains(" "+c+" ", " "+word+" ") {
n++
}
}
return n
}
p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
if !ok || p.Tier != UpdateTierOffsite || !p.At.Equal(time.Date(2026, 9, 13, 1, 0, 0, 0, time.UTC)) {
t.Fatalf("want gokapi's off-site snapshot; got %+v ok=%v", p, ok)
}
if count("snapshots") != 1 || count("stats") != 0 {
t.Errorf("the update lookup must be one snapshots call and no stats; calls=%v", calls)
}
calls = nil
if _, err := m.OffsiteInventoryList(context.Background()); err != nil {
t.Fatal(err)
}
if count("stats") == 0 {
t.Fatalf("control: the inventory DOES run stats, so the fake must see them; calls=%v", calls)
}
}
// R-474 — only the app's own Tier-2 mirror is found and removed.
func TestR474_OnlyTheAppsOwnMirrorIsRemoved(t *testing.T) {
m, sett := newOffboxManager(t)
drive := t.TempDir()
if err := sett.AddStoragePath(settings.StoragePath{Path: drive, Label: "HDD", Schedulable: true}); err != nil {
t.Fatal(err)
}
root := NamespaceRootFor(drive, m.systemDataPath)
sec := filepath.Join(root, "backups", "secondary")
mine, other, shares := filepath.Join(sec, "gokapi"), filepath.Join(sec, "kimai"), filepath.Join(sec, SharesPseudoStack)
for _, d := range []string{mine, other, shares} {
if err := os.MkdirAll(filepath.Join(d, "recovery-unit"), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(d, "recovery-unit", "manifest.json"), []byte("{}"), 0o644); err != nil {
t.Fatal(err)
}
}
dirs := m.Tier2MirrorDirsForApp("gokapi")
if len(dirs) != 1 || dirs[0] != mine {
t.Fatalf("mirror dirs for gokapi = %v, want [%s]", dirs, mine)
}
removed := m.RemoveTier2Mirrors("gokapi", append(dirs, other, shares, filepath.Join(sec, "gokapi", "..", "kimai")))
if len(removed) != 1 {
t.Errorf("exactly the app's own mirror must be removed, got %v", removed)
}
if _, err := os.Stat(mine); !os.IsNotExist(err) {
t.Error("gokapi's mirror must be gone")
}
for _, keep := range []string{other, shares} {
if _, err := os.Stat(keep); err != nil {
t.Errorf("%s must survive: %v", keep, err)
}
}
if m.Tier2MirrorDirsForApp("../kimai") != nil || m.RemoveTier2Mirrors(SharesPseudoStack, []string{shares}) != nil {
t.Error("an unsafe or shares name must find and remove nothing")
}
}
func TestR474_DeleteAppBackupPrefsForgetsOnlyThatApp(t *testing.T) {
_, sett := newOffboxManager(t)
_ = sett.SetAppOffbox("gokapi", true)
_ = sett.SetCrossDriveConfig("gokapi", &settings.CrossDriveBackup{DestinationPath: "/mnt/x"})
_ = sett.SetAppOffbox("kimai", true)
if err := sett.DeleteAppBackupPrefs("gokapi"); err != nil {
t.Fatal(err)
}
if sett.IsAppOffbox("gokapi") || sett.GetCrossDriveConfig("gokapi") != nil {
t.Error("gokapi's prefs must be forgotten")
}
if !sett.IsAppOffbox("kimai") {
t.Error("another app's prefs must stay")
}
if err := sett.DeleteAppBackupPrefs("never-there"); err != nil {
t.Errorf("deleting absent prefs is a no-op, got %v", err)
}
}
+6 -8
View File
@@ -189,26 +189,24 @@ func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier in
}
return UpdateTierPoint{}, false
case UpdateTierOffsite:
inv := m.updateOffsiteInvFn
if inv == nil {
times := m.updateOffsiteTimesFn
if times == nil {
if m.settings == nil || !m.OffboxConfigured() {
return UpdateTierPoint{}, false
}
inv = m.OffsiteInventoryList
times = m.OffsiteSnapshotTimes
}
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
defer cancel()
got, err := inv(cctx)
got, err := times(cctx)
if err != nil {
if !errors.Is(err, errNoOffsiteTarget) {
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
}
return UpdateTierPoint{}, false
}
for _, a := range got.Apps {
if a.App == stackName && !a.LatestAt.IsZero() {
return UpdateTierPoint{Tier: tier, At: a.LatestAt}, true
}
if at, ok := got[stackName]; ok && !at.IsZero() {
return UpdateTierPoint{Tier: tier, At: at}, true
}
}
return UpdateTierPoint{}, false
+13 -13
View File
@@ -43,13 +43,13 @@ func tier1At(at time.Time) func(string) ([]RestorePoint, bool) {
}
}
func noTier1(string) ([]RestorePoint, bool) { return []RestorePoint{}, true }
func offsiteWith(app string, at time.Time) func(context.Context) (OffsiteInventory, error) {
return func(context.Context) (OffsiteInventory, error) {
return OffsiteInventory{Apps: []OffsiteInventoryApp{{App: app, LatestAt: at}}}, nil
func offsiteWith(app string, at time.Time) func(context.Context) (map[string]time.Time, error) {
return func(context.Context) (map[string]time.Time, error) {
return map[string]time.Time{app: at}, nil
}
}
func noOffsiteTarget(context.Context) (OffsiteInventory, error) {
return OffsiteInventory{}, errNoOffsiteTarget
func noOffsiteTarget(context.Context) (map[string]time.Time, error) {
return nil, errNoOffsiteTarget
}
func tiersOf(ps []UpdateTierPoint) []int {
@@ -67,7 +67,7 @@ func TestR475_G_TierOrderIsSecondDriveThenOwnUnitThenOffsite(t *testing.T) {
offsiteCalls := 0
m.updateTier2PointFn = tier2At(r475T0.Add(-1 * time.Hour))
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
m.updateOffsiteTimesFn = func(ctx context.Context) (map[string]time.Time, error) {
offsiteCalls++
return offsiteWith("gokapi", r475T0.Add(-3*time.Hour))(ctx)
}
@@ -94,7 +94,7 @@ func TestR475_H_OwnUnitOnly(t *testing.T) {
m, _ := r475Manager()
m.updateTier2PointFn = noTier2
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
m.updateOffsiteInvFn = noOffsiteTarget
m.updateOffsiteTimesFn = noOffsiteTarget
p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
if !ok || p.Tier != UpdateTierLocal || !p.At.Equal(r475T0.Add(-2*time.Hour)) {
t.Fatalf("got %+v ok=%v", p, ok)
@@ -116,7 +116,7 @@ func TestR475_H_OwnUnitOnly(t *testing.T) {
func TestR475_I_OffsiteOnly(t *testing.T) {
m, _ := r475Manager()
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
m.updateOffsiteInvFn = offsiteWith("gokapi", r475T0.Add(-5*time.Hour))
m.updateOffsiteTimesFn = offsiteWith("gokapi", r475T0.Add(-5*time.Hour))
if p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); !ok || p.Tier != UpdateTierOffsite {
t.Fatalf("got %+v ok=%v", p, ok)
}
@@ -130,8 +130,8 @@ func TestR475_I_OffsiteOnly(t *testing.T) {
func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
m, buf := r475Manager()
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
m.updateOffsiteInvFn = func(context.Context) (OffsiteInventory, error) {
return OffsiteInventory{}, errors.New("ssh: connect to host: connection timed out")
m.updateOffsiteTimesFn = func(context.Context) (map[string]time.Time, error) {
return nil, errors.New("ssh: connect to host: connection timed out")
}
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok {
t.Error("an unreachable off-site copy must count as absent")
@@ -142,7 +142,7 @@ func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
// Control: a box with NO off-site target is plainly absent — that is not a fault, so no WARN.
buf.Reset()
m.updateOffsiteInvFn = noOffsiteTarget
m.updateOffsiteTimesFn = noOffsiteTarget
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok || strings.Contains(buf.String(), "WARN") {
t.Errorf("no off-site target: want absent and silent; ok=%v log=%q", ok, buf.String())
}
@@ -152,9 +152,9 @@ func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
updateOffsiteCheckTimeout = 50 * time.Millisecond
defer func() { updateOffsiteCheckTimeout = old }()
buf.Reset()
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
m.updateOffsiteTimesFn = func(ctx context.Context) (map[string]time.Time, error) {
<-ctx.Done()
return OffsiteInventory{}, ctx.Err()
return nil, ctx.Err()
}
start := time.Now()
_, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)