controller v0.240.0: seven defects from the any-tier proof and the first nightly rotation
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record, so the second-drive restore is no longer refused over an intact mirror. R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps). R-485: the backup card sizes the recovery unit and the mirror(s). R-480: a held update's sentence leaves the card once the hold is lifted. R-477: the update's off-site lookup is one snapshots call, no stats. R-478: a copy older than this install's deploy does not count. R-474: "delete backups" deletes the unit, the mirror(s) and the prefs. Tests and red-proofs per row; evidence in felhom.eu documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
This commit is contained in:
@@ -576,7 +576,7 @@ or a migration running).
|
||||
|
||||
**Any backup tier counts (v0.239.0, R-475).** The update leans on the first FRESH copy in the order
|
||||
second drive (Tier 2), the app's own recovery unit (Tier 1, „helyi"), off-site (Tier 3, looked up with
|
||||
a 15 s bound — unreachable counts as absent, with a WARN). `update.backup_max_age` applies to whichever
|
||||
a 15 s bound — unreachable counts as absent, with a WARN; since v0.240.0 it runs no per-app `stats`). `update.backup_max_age` applies to whichever
|
||||
tier is chosen. An app with no copy anywhere is backed up first. Tier 2 is required nowhere in the
|
||||
update path; the backups page's „Teljes visszaállítás" still reads the Tier-2 predicate alone.
|
||||
|
||||
@@ -588,7 +588,9 @@ every container running for an app with none; bounded by `update.health_timeout`
|
||||
puts the pin back. An app that does not become healthy is stopped and HELD** — the pin stays on the
|
||||
new version, and the hold sentence names the tier and the date of the copy it can be restored from
|
||||
(„második meghajtó" / „saját meghajtó" / „távoli mentés"). A successful unit restore — and, since
|
||||
v0.239.0, a successful off-site restore — lifts an update hold.
|
||||
v0.239.0, a successful off-site restore — lifts an update hold, and since v0.240.0 the held update's
|
||||
sentence leaves the card with it. A copy older than the app's current `deployed_at` does not count
|
||||
(v0.240.0) — it belongs to a previous install. The off-site lookup is one `restic snapshots` call.
|
||||
|
||||
**Config (`controller.yaml`):**
|
||||
|
||||
@@ -3780,7 +3782,7 @@ All daily jobs use Europe/Budapest timezone. Skip-if-running prevents concurrent
|
||||
| GET | `/api/stacks/{name}/logs` | Container logs (`?raw=1` for plain text) |
|
||||
| GET | `/api/stacks/{name}/hdd-data` | HDD data paths + sizes — resolved from the app's OWN `app.yaml` `HDD_PATH` (v0.236.0, R-442), never the global config |
|
||||
| GET | `/api/stacks/{name}/backup-data` | Backup data paths + sizes (DB dumps, cross-drive rsync) |
|
||||
| POST | `/api/stacks/{name}/remove` | Remove deployed stack (revert to "not deployed"). `remove_hdd_data: true` deletes the app's folders under its recorded `HDD_PATH` and lists them; **409 + a Hungarian sentence when the data was asked for but its location cannot be resolved or the drive is absent — nothing is touched, the app is kept** (v0.236.0, R-442). `hdd_paths_removed` is `[]` for an SSD app (never `null`); `hdd_paths_missing`, `hdd_note`, `backup_paths_refused` state what was not found / not removed |
|
||||
| POST | `/api/stacks/{name}/remove` | Remove deployed stack (revert to "not deployed"). `remove_hdd_data: true` deletes the app's folders under its recorded `HDD_PATH` and lists them; **409 + a Hungarian sentence when the data was asked for but its location cannot be resolved or the drive is absent — nothing is touched, the app is kept** (v0.236.0, R-442). `hdd_paths_removed` is `[]` for an SSD app (never `null`); `hdd_paths_missing`, `hdd_note`, `backup_paths_refused` state what was not found / not removed. `remove_backups: true` (v0.240.0, R-474) deletes the app's whole recovery unit, its Tier-2 mirror(s) on any registered drive and its backup preferences (`backup_paths_removed` lists them); **without it the backups AND the Tier-2 record are kept, so the removed app can still be restored from the second drive (R-486)**. Off-site snapshots are never touched by removal |
|
||||
| DELETE | `/api/stacks/{name}` | Delete orphaned stack — same R-442 resolution and refusal shape as `/remove` |
|
||||
| POST | `/api/sync` | Trigger catalog sync |
|
||||
| GET | `/api/system/info` | System info + sync status |
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"go/ast"
|
||||
"go/parser"
|
||||
"go/token"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// R-474 — the remove handler really deletes the whole unit, the Tier-2 mirror and the prefs.
|
||||
func TestR474_RemoveHandlerDeletesUnitMirrorAndPrefs(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var sel []string
|
||||
for _, d := range f.Decls {
|
||||
fn, ok := d.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "removeStack" || fn.Body == nil {
|
||||
continue
|
||||
}
|
||||
ast.Inspect(fn.Body, func(n ast.Node) bool {
|
||||
if s, ok := n.(*ast.SelectorExpr); ok {
|
||||
sel = append(sel, s.Sel.Name)
|
||||
}
|
||||
return true
|
||||
})
|
||||
}
|
||||
joined := " " + strings.Join(sel, " ") + " "
|
||||
for _, want := range []string{"RecoveryUnitPath", "Tier2MirrorDirsForApp", "RemoveTier2Mirrors", "DeleteAppBackupPrefs"} {
|
||||
if !strings.Contains(joined, " "+want+" ") {
|
||||
t.Errorf("removeStack must call %s", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// R-486 — removing an app with its backups KEPT keeps its Tier-2 record. The only call that may
|
||||
// forget the record is DeleteAppBackupPrefs, which the R-474 branch guards on remove_backups.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): put `r.sett.SetCrossDriveConfig(name, nil)` back into
|
||||
// removeStack unconditionally — this fails.
|
||||
func TestR486_RemovalKeepsTheTier2RecordUnlessBackupsGo(t *testing.T) {
|
||||
fset := token.NewFileSet()
|
||||
f, err := parser.ParseFile(fset, "router.go", nil, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, d := range f.Decls {
|
||||
fn, ok := d.(*ast.FuncDecl)
|
||||
if !ok || fn.Name.Name != "removeStack" || fn.Body == nil {
|
||||
continue
|
||||
}
|
||||
ast.Inspect(fn.Body, func(n ast.Node) bool {
|
||||
if s, ok := n.(*ast.SelectorExpr); ok && s.Sel.Name == "SetCrossDriveConfig" {
|
||||
t.Errorf("removeStack calls SetCrossDriveConfig at line %d — a removal with backups kept would forget the mirror it kept (R-486)", fset.Position(s.Pos()).Line)
|
||||
}
|
||||
return true
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -817,11 +817,13 @@ func (r *Router) getStackBackupData(w http.ResponseWriter, _ *http.Request, name
|
||||
// mount IS the namespace; SSD-only: <systemDataPath>/felhom-data). Passing the namespace root
|
||||
// (not the raw drive) keeps GetStackBackupData's paths single-nested under Model A.
|
||||
var nsRoot string
|
||||
var mirrors []string
|
||||
if r.backupMgr != nil {
|
||||
nsRoot = r.backupMgr.AppNamespaceRoot(name)
|
||||
mirrors = r.backupMgr.Tier2MirrorDirsForApp(name) // R-485: the real second-drive copy, not a dead `rsync` path
|
||||
}
|
||||
|
||||
resp, err := r.stackMgr.GetStackBackupData(name, nsRoot)
|
||||
resp, err := r.stackMgr.GetStackBackupData(name, nsRoot, mirrors)
|
||||
if err != nil {
|
||||
writeJSON(w, http.StatusNotFound, apiResponse{OK: false, Error: err.Error()})
|
||||
return
|
||||
@@ -850,12 +852,18 @@ func (r *Router) removeStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
|
||||
// Compute backup paths to remove if requested. Disk-tier (cross-drive rsync)
|
||||
// backup has moved to the host agent; only the app-data DB-dump path is removed here.
|
||||
var backupPaths []string
|
||||
//
|
||||
// R-474 (v0.240.0): "delete backups" deletes the app's WHOLE recovery unit (definition, db-dumps,
|
||||
// volume tars) and its Tier-2 mirror, not only db-dumps. The mirror is on another drive, outside
|
||||
// RemoveStack's per-app base, so the backup manager removes it — and its location is read NOW,
|
||||
// before the cross-drive record is cleared below. Off-site snapshots are not touched by this path.
|
||||
var backupPaths, mirrorDirs []string
|
||||
if body.RemoveBackups && r.backupMgr != nil {
|
||||
nsRoot := r.backupMgr.AppNamespaceRoot(name)
|
||||
if nsRoot != "" {
|
||||
backupPaths = append(backupPaths, backup.AppDBDumpPath(nsRoot, name))
|
||||
backupPaths = append(backupPaths, backup.RecoveryUnitPath(nsRoot, name), backup.AppDBDumpPath(nsRoot, name))
|
||||
}
|
||||
mirrorDirs = r.backupMgr.Tier2MirrorDirsForApp(name)
|
||||
}
|
||||
|
||||
resp, err := r.stackMgr.RemoveStack(name, body.RemoveHDDData, backupPaths)
|
||||
@@ -882,10 +890,19 @@ func (r *Router) removeStack(w http.ResponseWriter, req *http.Request, name stri
|
||||
return
|
||||
}
|
||||
|
||||
// Clean up cross-drive backup config for this stack
|
||||
if r.sett != nil {
|
||||
if err := r.sett.SetCrossDriveConfig(name, nil); err != nil {
|
||||
r.logger.Printf("[WARN] [api] Failed to clean cross-drive config for %s: %v", name, err)
|
||||
if body.RemoveBackups && r.backupMgr != nil && len(mirrorDirs) > 0 {
|
||||
resp.BackupPathsRemoved = append(resp.BackupPathsRemoved, r.backupMgr.RemoveTier2Mirrors(name, mirrorDirs)...)
|
||||
}
|
||||
|
||||
// R-486 (v0.240.0): the app's backup preferences — and with them the Tier-2 RECORD that
|
||||
// tier2RecordedCopyDir needs — are forgotten ONLY when the customer asked for the backups to be
|
||||
// deleted. Until v0.239.0 the cross-drive record was cleared on every removal, so an app removed
|
||||
// with its backups kept could not be restored from its intact second-drive mirror (measured on
|
||||
// demo-hp 2026-09-13: „nincs másodlagos fájlmásolat" over a 236 MB mirror). Pinned by
|
||||
// TestR486_RemovalKeepsTheTier2RecordUnlessBackupsGo.
|
||||
if r.sett != nil && body.RemoveBackups {
|
||||
if err := r.sett.DeleteAppBackupPrefs(name); err != nil {
|
||||
r.logger.Printf("[WARN] [api] Failed to forget the backup preferences of %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -30,7 +30,10 @@ import (
|
||||
func dbTypeForImage(image string) (DBType, bool) {
|
||||
img := strings.ToLower(image)
|
||||
switch {
|
||||
case strings.Contains(img, "postgres"):
|
||||
// R-484 (v0.240.0): derived Postgres images whose name does not carry "postgres". Measured on
|
||||
// demo-hp 2026-09-13: adventurelog's `postgis/postgis:16-3.5-alpine` was "not a database", so
|
||||
// the app got no logical dump anywhere. Same pg_dump, same PG* credentials.
|
||||
case strings.Contains(img, "postgres"), strings.Contains(img, "postgis"), strings.Contains(img, "pgvector"), strings.Contains(img, "timescaledb"):
|
||||
return DBTypePostgres, true
|
||||
case strings.Contains(img, "mariadb"), strings.Contains(img, "mysql"):
|
||||
return DBTypeMariaDB, true
|
||||
|
||||
@@ -41,6 +41,10 @@ func TestDBTypeForImage(t *testing.T) {
|
||||
// immich's real pin — a vector-extended postgres whose REPO segment carries the substring.
|
||||
{"ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0", DBTypePostgres, true},
|
||||
{"postgres", DBTypePostgres, true},
|
||||
// R-484: adventurelog's real pin — no "postgres" substring anywhere in the reference.
|
||||
{"postgis/postgis:16-3.5-alpine", DBTypePostgres, true},
|
||||
{"pgvector/pgvector:pg16", DBTypePostgres, true},
|
||||
{"timescale/timescaledb:2.17.0-pg16", DBTypePostgres, true},
|
||||
{"POSTGRES:16", DBTypePostgres, true}, // the discovery path lowercases; so does this
|
||||
{"mariadb:11", DBTypeMariaDB, true},
|
||||
{"mysql:8.4", DBTypeMariaDB, true},
|
||||
|
||||
@@ -174,11 +174,11 @@ type Manager struct {
|
||||
updatingCheck func(stackName string) bool
|
||||
|
||||
// R-475 update-precondition seams, one per tier. Nil → the real Tier2UnitRestorePoint /
|
||||
// ListRestorePoints / OffsiteInventoryList. They let a test reach Tier 1 and Tier 3 without a drive
|
||||
// ListRestorePoints / OffsiteSnapshotTimes. They let a test reach Tier 1 and Tier 3 without a drive
|
||||
// or a restic repository; see UpdateRestorePoints.
|
||||
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
|
||||
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
|
||||
updateOffsiteInvFn func(ctx context.Context) (OffsiteInventory, error)
|
||||
updateTier2PointFn func(stackName string) (Tier2RestorePoint, error)
|
||||
updateTier1PointsFn func(stackName string) ([]RestorePoint, bool)
|
||||
updateOffsiteTimesFn func(ctx context.Context) (map[string]time.Time, error)
|
||||
|
||||
// R-354 volume-REPLAY seam — the mirror of the F17 DB seams above, so the off-site path's new
|
||||
// volume leg is unit-testable without Docker. Nil → the real restoreDockerVolumesFrom.
|
||||
|
||||
@@ -52,19 +52,21 @@ type OffsiteInventory struct {
|
||||
Empty bool
|
||||
}
|
||||
|
||||
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
|
||||
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
|
||||
//
|
||||
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
|
||||
// app is in there matters more than knowing how big it is, and dropping it would under-report the
|
||||
// customer's own data.
|
||||
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
|
||||
var inv OffsiteInventory
|
||||
// offsiteNewest is one app tag's newest snapshot.
|
||||
type offsiteNewest struct {
|
||||
id string
|
||||
at time.Time
|
||||
}
|
||||
|
||||
// offsiteNewestPerTag runs ONE `snapshots --json` and returns the newest snapshot per app tag, and
|
||||
// whether the repository opened cleanly and holds no snapshots at all. Shared by the inventory page
|
||||
// and the update precondition (R-477), so the two cannot disagree about what is in the repository.
|
||||
func (m *Manager) offsiteNewestPerTag(ctx context.Context) (map[string]offsiteNewest, bool, error) {
|
||||
// A box can hold a recovered key and still have no off-site COORDINATES — the pristine rebuilt
|
||||
// shape, before its target is re-applied. Reading the repository is impossible then, and saying so
|
||||
// is the honest answer; without this guard offboxBaseArgs nil-derefs on the missing target.
|
||||
if !m.OffboxConfigured() {
|
||||
return inv, errNoOffsiteTarget
|
||||
return nil, false, errNoOffsiteTarget
|
||||
}
|
||||
t := m.settings.GetOffboxTarget()
|
||||
base, env := m.offboxBaseArgs(t)
|
||||
@@ -72,7 +74,7 @@ func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, e
|
||||
defer cancel()
|
||||
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
||||
if err != nil {
|
||||
return inv, err
|
||||
return nil, false, err
|
||||
}
|
||||
var snaps []struct {
|
||||
ShortID string `json:"short_id"`
|
||||
@@ -81,34 +83,63 @@ func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, e
|
||||
Tags []string `json:"tags"`
|
||||
}
|
||||
if uerr := json.Unmarshal(out, &snaps); uerr != nil {
|
||||
return inv, uerr
|
||||
return nil, false, uerr
|
||||
}
|
||||
if len(snaps) == 0 {
|
||||
inv.Empty = true
|
||||
return inv, nil
|
||||
return nil, true, nil
|
||||
}
|
||||
// Newest snapshot per tag. A snapshot may carry several tags; each names an app it belongs to.
|
||||
newest := map[string]struct {
|
||||
id string
|
||||
at time.Time
|
||||
}{}
|
||||
for _, s := range snaps {
|
||||
id := s.ShortID
|
||||
newest := map[string]offsiteNewest{}
|
||||
for _, sn := range snaps {
|
||||
id := sn.ShortID
|
||||
if id == "" {
|
||||
id = s.ID
|
||||
id = sn.ID
|
||||
}
|
||||
for _, tag := range s.Tags {
|
||||
for _, tag := range sn.Tags {
|
||||
if tag == "" {
|
||||
continue
|
||||
}
|
||||
if cur, ok := newest[tag]; !ok || s.Time.After(cur.at) {
|
||||
newest[tag] = struct {
|
||||
id string
|
||||
at time.Time
|
||||
}{id: id, at: s.Time}
|
||||
if cur, ok := newest[tag]; !ok || sn.Time.After(cur.at) {
|
||||
newest[tag] = offsiteNewest{id: id, at: sn.Time}
|
||||
}
|
||||
}
|
||||
}
|
||||
return newest, false, nil
|
||||
}
|
||||
|
||||
// OffsiteSnapshotTimes (R-477, v0.240.0) is the newest snapshot time per app — ONE `snapshots --json`,
|
||||
// no per-app `stats`. It is what the update precondition needs. Measured on demo-hp 2026-09-13: going
|
||||
// through OffsiteInventoryList instead, the update's check spent its whole 15 s bound on the size calls
|
||||
// and the bound killed one for an unrelated app (`size of kimai's newest snapshot unknown: signal:
|
||||
// killed`). Pinned by TestR477_TheUpdateOffsiteLookupRunsNoStats.
|
||||
func (m *Manager) OffsiteSnapshotTimes(ctx context.Context) (map[string]time.Time, error) {
|
||||
newest, _, err := m.offsiteNewestPerTag(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make(map[string]time.Time, len(newest))
|
||||
for tag, n := range newest {
|
||||
out[tag] = n.at
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
|
||||
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
|
||||
//
|
||||
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
|
||||
// app is in there matters more than knowing how big it is, and dropping it would under-report the
|
||||
// customer's own data.
|
||||
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
|
||||
var inv OffsiteInventory
|
||||
newest, empty, err := m.offsiteNewestPerTag(ctx)
|
||||
if err != nil {
|
||||
return inv, err
|
||||
}
|
||||
if empty {
|
||||
inv.Empty = true
|
||||
return inv, nil
|
||||
}
|
||||
if len(newest) == 0 {
|
||||
// Snapshots exist but carry no tags — not "empty", and saying so would be a lie. Report an
|
||||
// empty app list without the Empty flag; the page renders the honest in-between wording.
|
||||
|
||||
@@ -57,6 +57,7 @@ var offsiteExempt = map[string]string{
|
||||
// browsing a page refuse while a backup runs, for no safety gain: it can neither take a lock nor
|
||||
// remove one.
|
||||
"OffsiteInventoryList": "restic snapshots --json only; snapshots measured 2026-08-31 not to lock, and it never routes through resticStep",
|
||||
"offsiteNewestPerTag": "restic snapshots --json only — the one reader behind OffsiteInventoryList and OffsiteSnapshotTimes (R-477, v0.240.0); same measurement, never routes through resticStep",
|
||||
}
|
||||
|
||||
// offsiteReachers are the calls that mean "this function talks to the off-site repository".
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// R-474 (v0.240.0) — "delete backups" on removal deletes the app's Tier-2 mirror too.
|
||||
//
|
||||
// Measured three times on 2026-09-13: removing an app with `remove_backups:true` deleted only its
|
||||
// db-dumps directory; the recovery unit, the volume tars and the Tier-2 mirror all survived, and a
|
||||
// later reinstall of the same app then leaned on the old install's unit as its "fresh" restore point
|
||||
// (R-478). The unit is inside the app's own backups base and RemoveStack deletes it; the MIRROR lives
|
||||
// on another drive, outside that base, so it is removed here, with its own path check.
|
||||
|
||||
func validMirrorStackName(n string) bool {
|
||||
return n != "" && n != "." && n != ".." && n != SharesPseudoStack && !strings.ContainsAny(n, `/\`)
|
||||
}
|
||||
|
||||
// tier2MirrorRoots are the namespace roots a Tier-2 mirror of this app can live under: the recorded
|
||||
// destination, every registered drive, and the system data path (a mirror outlives a changed target).
|
||||
func (m *Manager) tier2MirrorRoots(stackName string) []string {
|
||||
seen := map[string]bool{}
|
||||
var roots []string
|
||||
add := func(r string) {
|
||||
if r == "" {
|
||||
return
|
||||
}
|
||||
r = filepath.Clean(r)
|
||||
if filepath.IsAbs(r) && !seen[r] {
|
||||
seen[r] = true
|
||||
roots = append(roots, r)
|
||||
}
|
||||
}
|
||||
if m.settings != nil {
|
||||
if cfg := m.settings.GetCrossDriveConfig(stackName); cfg != nil {
|
||||
add(cfg.DestinationPath)
|
||||
}
|
||||
for _, sp := range m.settings.GetStoragePaths() {
|
||||
if sp.Path != "" {
|
||||
add(NamespaceRootFor(sp.Path, m.systemDataPath))
|
||||
}
|
||||
}
|
||||
}
|
||||
if m.systemDataPath != "" {
|
||||
add(NamespaceRootFor(m.systemDataPath, m.systemDataPath))
|
||||
}
|
||||
return roots
|
||||
}
|
||||
|
||||
// Tier2MirrorDirsForApp lists the app's Tier-2 mirror directories that exist now. Call it BEFORE the
|
||||
// removal clears the app's cross-drive record, which is one of the places it looks.
|
||||
func (m *Manager) Tier2MirrorDirsForApp(stackName string) []string {
|
||||
if m == nil || !validMirrorStackName(stackName) {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
for _, root := range m.tier2MirrorRoots(stackName) {
|
||||
d := filepath.Join(root, "backups", "secondary", stackName)
|
||||
if fi, err := os.Stat(d); err == nil && fi.IsDir() {
|
||||
out = append(out, d)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// RemoveTier2Mirrors deletes the given mirror directories, each only if it is exactly
|
||||
// <root>/backups/secondary/<stackName> for one of the app's mirror roots — never another app's mirror,
|
||||
// never the shares mirror, never a path that merely cleans to one. Returns "path (size)" per removal.
|
||||
func (m *Manager) RemoveTier2Mirrors(stackName string, dirs []string) []string {
|
||||
if m == nil || !validMirrorStackName(stackName) {
|
||||
return nil
|
||||
}
|
||||
allowed := map[string]bool{}
|
||||
for _, root := range m.tier2MirrorRoots(stackName) {
|
||||
allowed[filepath.Join(root, "backups", "secondary", stackName)] = true
|
||||
}
|
||||
var removed []string
|
||||
for _, d := range dirs {
|
||||
if !allowed[d] {
|
||||
m.logger.Printf("[WARN] [backup] remove %s: refusing to delete %q — not this app's Tier-2 mirror", stackName, d)
|
||||
continue
|
||||
}
|
||||
size := humanizeBytes(dirSizeBytes(d))
|
||||
if err := os.RemoveAll(d); err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] remove %s: deleting the Tier-2 mirror %s failed: %v", stackName, d, err)
|
||||
continue
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] remove %s: Tier-2 mirror deleted: %s (%s)", stackName, d, size)
|
||||
removed = append(removed, fmt.Sprintf("%s (%s)", d, size))
|
||||
}
|
||||
return removed
|
||||
}
|
||||
@@ -0,0 +1,110 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// R-477 — the update's Tier-3 lookup is ONE snapshots call. The control proves the fake runner does
|
||||
// see stats calls when something makes them (the inventory page), so a zero is a measurement.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): make OffsiteSnapshotTimes read OffsiteInventoryList — this fails.
|
||||
func TestR477_TheUpdateOffsiteLookupRunsNoStats(t *testing.T) {
|
||||
m, _ := newOffboxManager(t)
|
||||
var calls []string
|
||||
m.SetOffboxRunner(func(_ context.Context, _ []string, args ...string) ([]byte, error) {
|
||||
calls = append(calls, strings.Join(args, " "))
|
||||
if contains(args, "snapshots") {
|
||||
return []byte(`[{"short_id":"a1","time":"2026-09-13T01:00:00Z","tags":["gokapi"]},{"short_id":"b2","time":"2026-09-13T02:00:00Z","tags":["kimai"]}]`), nil
|
||||
}
|
||||
return []byte(`{"total_size":1024,"total_file_count":1}`), nil
|
||||
})
|
||||
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
|
||||
count := func(word string) int {
|
||||
n := 0
|
||||
for _, c := range calls {
|
||||
if strings.Contains(" "+c+" ", " "+word+" ") {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
|
||||
if !ok || p.Tier != UpdateTierOffsite || !p.At.Equal(time.Date(2026, 9, 13, 1, 0, 0, 0, time.UTC)) {
|
||||
t.Fatalf("want gokapi's off-site snapshot; got %+v ok=%v", p, ok)
|
||||
}
|
||||
if count("snapshots") != 1 || count("stats") != 0 {
|
||||
t.Errorf("the update lookup must be one snapshots call and no stats; calls=%v", calls)
|
||||
}
|
||||
calls = nil
|
||||
if _, err := m.OffsiteInventoryList(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if count("stats") == 0 {
|
||||
t.Fatalf("control: the inventory DOES run stats, so the fake must see them; calls=%v", calls)
|
||||
}
|
||||
}
|
||||
|
||||
// R-474 — only the app's own Tier-2 mirror is found and removed.
|
||||
func TestR474_OnlyTheAppsOwnMirrorIsRemoved(t *testing.T) {
|
||||
m, sett := newOffboxManager(t)
|
||||
drive := t.TempDir()
|
||||
if err := sett.AddStoragePath(settings.StoragePath{Path: drive, Label: "HDD", Schedulable: true}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
root := NamespaceRootFor(drive, m.systemDataPath)
|
||||
sec := filepath.Join(root, "backups", "secondary")
|
||||
mine, other, shares := filepath.Join(sec, "gokapi"), filepath.Join(sec, "kimai"), filepath.Join(sec, SharesPseudoStack)
|
||||
for _, d := range []string{mine, other, shares} {
|
||||
if err := os.MkdirAll(filepath.Join(d, "recovery-unit"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(d, "recovery-unit", "manifest.json"), []byte("{}"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
dirs := m.Tier2MirrorDirsForApp("gokapi")
|
||||
if len(dirs) != 1 || dirs[0] != mine {
|
||||
t.Fatalf("mirror dirs for gokapi = %v, want [%s]", dirs, mine)
|
||||
}
|
||||
removed := m.RemoveTier2Mirrors("gokapi", append(dirs, other, shares, filepath.Join(sec, "gokapi", "..", "kimai")))
|
||||
if len(removed) != 1 {
|
||||
t.Errorf("exactly the app's own mirror must be removed, got %v", removed)
|
||||
}
|
||||
if _, err := os.Stat(mine); !os.IsNotExist(err) {
|
||||
t.Error("gokapi's mirror must be gone")
|
||||
}
|
||||
for _, keep := range []string{other, shares} {
|
||||
if _, err := os.Stat(keep); err != nil {
|
||||
t.Errorf("%s must survive: %v", keep, err)
|
||||
}
|
||||
}
|
||||
if m.Tier2MirrorDirsForApp("../kimai") != nil || m.RemoveTier2Mirrors(SharesPseudoStack, []string{shares}) != nil {
|
||||
t.Error("an unsafe or shares name must find and remove nothing")
|
||||
}
|
||||
}
|
||||
|
||||
func TestR474_DeleteAppBackupPrefsForgetsOnlyThatApp(t *testing.T) {
|
||||
_, sett := newOffboxManager(t)
|
||||
_ = sett.SetAppOffbox("gokapi", true)
|
||||
_ = sett.SetCrossDriveConfig("gokapi", &settings.CrossDriveBackup{DestinationPath: "/mnt/x"})
|
||||
_ = sett.SetAppOffbox("kimai", true)
|
||||
if err := sett.DeleteAppBackupPrefs("gokapi"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if sett.IsAppOffbox("gokapi") || sett.GetCrossDriveConfig("gokapi") != nil {
|
||||
t.Error("gokapi's prefs must be forgotten")
|
||||
}
|
||||
if !sett.IsAppOffbox("kimai") {
|
||||
t.Error("another app's prefs must stay")
|
||||
}
|
||||
if err := sett.DeleteAppBackupPrefs("never-there"); err != nil {
|
||||
t.Errorf("deleting absent prefs is a no-op, got %v", err)
|
||||
}
|
||||
}
|
||||
@@ -189,26 +189,24 @@ func (m *Manager) updateTierPoint(ctx context.Context, stackName string, tier in
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
case UpdateTierOffsite:
|
||||
inv := m.updateOffsiteInvFn
|
||||
if inv == nil {
|
||||
times := m.updateOffsiteTimesFn
|
||||
if times == nil {
|
||||
if m.settings == nil || !m.OffboxConfigured() {
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
inv = m.OffsiteInventoryList
|
||||
times = m.OffsiteSnapshotTimes
|
||||
}
|
||||
cctx, cancel := context.WithTimeout(ctx, updateOffsiteCheckTimeout)
|
||||
defer cancel()
|
||||
got, err := inv(cctx)
|
||||
got, err := times(cctx)
|
||||
if err != nil {
|
||||
if !errors.Is(err, errNoOffsiteTarget) {
|
||||
m.logger.Printf("[WARN] [backup] update precondition for %s: the off-site copy could not be checked within %s (%v) — counted as ABSENT", stackName, updateOffsiteCheckTimeout, err)
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
}
|
||||
for _, a := range got.Apps {
|
||||
if a.App == stackName && !a.LatestAt.IsZero() {
|
||||
return UpdateTierPoint{Tier: tier, At: a.LatestAt}, true
|
||||
}
|
||||
if at, ok := got[stackName]; ok && !at.IsZero() {
|
||||
return UpdateTierPoint{Tier: tier, At: at}, true
|
||||
}
|
||||
}
|
||||
return UpdateTierPoint{}, false
|
||||
|
||||
@@ -43,13 +43,13 @@ func tier1At(at time.Time) func(string) ([]RestorePoint, bool) {
|
||||
}
|
||||
}
|
||||
func noTier1(string) ([]RestorePoint, bool) { return []RestorePoint{}, true }
|
||||
func offsiteWith(app string, at time.Time) func(context.Context) (OffsiteInventory, error) {
|
||||
return func(context.Context) (OffsiteInventory, error) {
|
||||
return OffsiteInventory{Apps: []OffsiteInventoryApp{{App: app, LatestAt: at}}}, nil
|
||||
func offsiteWith(app string, at time.Time) func(context.Context) (map[string]time.Time, error) {
|
||||
return func(context.Context) (map[string]time.Time, error) {
|
||||
return map[string]time.Time{app: at}, nil
|
||||
}
|
||||
}
|
||||
func noOffsiteTarget(context.Context) (OffsiteInventory, error) {
|
||||
return OffsiteInventory{}, errNoOffsiteTarget
|
||||
func noOffsiteTarget(context.Context) (map[string]time.Time, error) {
|
||||
return nil, errNoOffsiteTarget
|
||||
}
|
||||
|
||||
func tiersOf(ps []UpdateTierPoint) []int {
|
||||
@@ -67,7 +67,7 @@ func TestR475_G_TierOrderIsSecondDriveThenOwnUnitThenOffsite(t *testing.T) {
|
||||
offsiteCalls := 0
|
||||
m.updateTier2PointFn = tier2At(r475T0.Add(-1 * time.Hour))
|
||||
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
|
||||
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
|
||||
m.updateOffsiteTimesFn = func(ctx context.Context) (map[string]time.Time, error) {
|
||||
offsiteCalls++
|
||||
return offsiteWith("gokapi", r475T0.Add(-3*time.Hour))(ctx)
|
||||
}
|
||||
@@ -94,7 +94,7 @@ func TestR475_H_OwnUnitOnly(t *testing.T) {
|
||||
m, _ := r475Manager()
|
||||
m.updateTier2PointFn = noTier2
|
||||
m.updateTier1PointsFn = tier1At(r475T0.Add(-2 * time.Hour))
|
||||
m.updateOffsiteInvFn = noOffsiteTarget
|
||||
m.updateOffsiteTimesFn = noOffsiteTarget
|
||||
p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
|
||||
if !ok || p.Tier != UpdateTierLocal || !p.At.Equal(r475T0.Add(-2*time.Hour)) {
|
||||
t.Fatalf("got %+v ok=%v", p, ok)
|
||||
@@ -116,7 +116,7 @@ func TestR475_H_OwnUnitOnly(t *testing.T) {
|
||||
func TestR475_I_OffsiteOnly(t *testing.T) {
|
||||
m, _ := r475Manager()
|
||||
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
|
||||
m.updateOffsiteInvFn = offsiteWith("gokapi", r475T0.Add(-5*time.Hour))
|
||||
m.updateOffsiteTimesFn = offsiteWith("gokapi", r475T0.Add(-5*time.Hour))
|
||||
if p, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); !ok || p.Tier != UpdateTierOffsite {
|
||||
t.Fatalf("got %+v ok=%v", p, ok)
|
||||
}
|
||||
@@ -130,8 +130,8 @@ func TestR475_I_OffsiteOnly(t *testing.T) {
|
||||
func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
|
||||
m, buf := r475Manager()
|
||||
m.updateTier2PointFn, m.updateTier1PointsFn = noTier2, noTier1
|
||||
m.updateOffsiteInvFn = func(context.Context) (OffsiteInventory, error) {
|
||||
return OffsiteInventory{}, errors.New("ssh: connect to host: connection timed out")
|
||||
m.updateOffsiteTimesFn = func(context.Context) (map[string]time.Time, error) {
|
||||
return nil, errors.New("ssh: connect to host: connection timed out")
|
||||
}
|
||||
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok {
|
||||
t.Error("an unreachable off-site copy must count as absent")
|
||||
@@ -142,7 +142,7 @@ func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
|
||||
|
||||
// Control: a box with NO off-site target is plainly absent — that is not a fault, so no WARN.
|
||||
buf.Reset()
|
||||
m.updateOffsiteInvFn = noOffsiteTarget
|
||||
m.updateOffsiteTimesFn = noOffsiteTarget
|
||||
if _, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil); ok || strings.Contains(buf.String(), "WARN") {
|
||||
t.Errorf("no off-site target: want absent and silent; ok=%v log=%q", ok, buf.String())
|
||||
}
|
||||
@@ -152,9 +152,9 @@ func TestR475_J_OffsiteUnreachableIsAbsentWithAWarn(t *testing.T) {
|
||||
updateOffsiteCheckTimeout = 50 * time.Millisecond
|
||||
defer func() { updateOffsiteCheckTimeout = old }()
|
||||
buf.Reset()
|
||||
m.updateOffsiteInvFn = func(ctx context.Context) (OffsiteInventory, error) {
|
||||
m.updateOffsiteTimesFn = func(ctx context.Context) (map[string]time.Time, error) {
|
||||
<-ctx.Done()
|
||||
return OffsiteInventory{}, ctx.Err()
|
||||
return nil, ctx.Err()
|
||||
}
|
||||
start := time.Now()
|
||||
_, ok, _ := m.UpdateRestorePoints(context.Background(), "gokapi", nil)
|
||||
|
||||
@@ -1142,6 +1142,19 @@ func (s *Settings) SetAppOffbox(stackName string, on bool) error {
|
||||
return s.save()
|
||||
}
|
||||
|
||||
// DeleteAppBackupPrefs (R-474, v0.240.0) forgets every per-app backup preference of an app whose
|
||||
// backups the customer asked to delete on removal — the off-site toggle and the cross-drive record
|
||||
// together. Before it, "delete backups" left `app_backup[<app>]` behind, and a reinstall inherited it.
|
||||
func (s *Settings) DeleteAppBackupPrefs(stackName string) error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
if _, ok := s.AppBackup[stackName]; !ok {
|
||||
return nil
|
||||
}
|
||||
delete(s.AppBackup, stackName)
|
||||
return s.save()
|
||||
}
|
||||
|
||||
// GetOffboxApps returns the stack names toggled for off-box backup.
|
||||
func (s *Settings) GetOffboxApps() []string {
|
||||
s.mu.RLock()
|
||||
|
||||
@@ -598,9 +598,15 @@ func (m *Manager) RemoveStack(name string, removeHDDData bool, backupPathsToRemo
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// GetStackBackupData returns information about backup data for a stack.
|
||||
// drivePath is the app's home drive (HDD or system data path).
|
||||
func (m *Manager) GetStackBackupData(name string, drivePath string) (*BackupDataResponse, error) {
|
||||
// GetStackBackupData returns information about backup data for a stack — what the remove dialog
|
||||
// sizes "delete backups" from. drivePath is the app's namespace root; mirrorDirs are the app's
|
||||
// Tier-2 mirror directories (backup.Manager.Tier2MirrorDirsForApp — stacks cannot import backup).
|
||||
//
|
||||
// R-485 (v0.240.0): until v0.239.0 this read `<ns>/backups/primary/<app>/db-dumps` and a pre-v2
|
||||
// `<ns>/backups/secondary/<app>/rsync` path — both dead for an app whose backups are the recovery
|
||||
// unit and a v2 mirror — and answered `has_backups:false` over 484 MB (measured on demo-hp
|
||||
// 2026-09-13). It now sizes the whole recovery unit and every mirror.
|
||||
func (m *Manager) GetStackBackupData(name string, drivePath string, mirrorDirs []string) (*BackupDataResponse, error) {
|
||||
_, ok := m.GetStack(name)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("stack %q not found", name)
|
||||
@@ -617,14 +623,12 @@ func (m *Manager) GetStackBackupData(name string, drivePath string) (*BackupData
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// Check DB dump directory. drivePath is the felhom-data namespace ROOT (Model A: the in-guest
|
||||
// drive mount itself), so backups/ sits directly under it: <nsRoot>/backups/primary/<stack>/db-dumps
|
||||
dbDumpPath := filepath.Join(drivePath, "backups", "primary", name, "db-dumps")
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(dbDumpPath))
|
||||
|
||||
// Check cross-drive rsync directory: <nsRoot>/backups/secondary/<stack>/rsync
|
||||
rsyncPath := filepath.Join(drivePath, "backups", "secondary", name, "rsync")
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(rsyncPath))
|
||||
// The recovery unit — definition, db-dumps and volume-dumps together. drivePath is the felhom-data
|
||||
// namespace ROOT (Model A: the in-guest drive mount itself), so backups/ sits directly under it.
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(appbackup.RecoveryUnitPath(drivePath, name)))
|
||||
for _, d := range mirrorDirs {
|
||||
resp.BackupPaths = append(resp.BackupPaths, buildPathInfo(d))
|
||||
}
|
||||
|
||||
if m.isDebug() {
|
||||
for _, p := range resp.BackupPaths {
|
||||
|
||||
@@ -149,6 +149,9 @@ type Stack struct {
|
||||
UpdatePhase string `json:"update_phase,omitempty"`
|
||||
UpdatePhaseLabel string `json:"update_phase_label,omitempty"`
|
||||
UpdateError string `json:"update_error,omitempty"`
|
||||
// updateHeld (R-480, v0.240.0) — the last update ended HELD, so UpdateError is the hold's own
|
||||
// sentence („… leállítva marad"). It is shown only while that hold is in force; fillHoldReason.
|
||||
updateHeld bool
|
||||
// HoldReason is the customer sentence of a hold in force on this app (a failed update or a failed
|
||||
// restore), "" when none. Filled on every read from the ONE hold store, never cached, so the page
|
||||
// and the API cannot show a hold the gate has already lifted — or miss one it enforces.
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-480 — the failed update's sentence goes when the hold it announced is lifted, or the app is gone.
|
||||
|
||||
func heldFailedUpdate(t *testing.T) (*Manager, *fakeGuards) {
|
||||
t.Helper()
|
||||
m, _, g, _ := newSlice4Manager(t)
|
||||
m.updateHealthFn = func(context.Context, string, time.Duration) (bool, string) { return false, "crash loop" }
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdatePhase != UpdatePhaseFailed || st.UpdateError != "HELD-SENTENCE" {
|
||||
t.Fatalf("fixture: a held failure must show the hold sentence while held; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
return m, g
|
||||
}
|
||||
|
||||
func TestR480_TheFailureSentenceGoesWhenItsHoldIsLifted(t *testing.T) {
|
||||
m, g := heldFailedUpdate(t)
|
||||
g.mu.Lock()
|
||||
g.held, g.holdWhy = false, "" // a successful restore cleared the hold
|
||||
g.mu.Unlock()
|
||||
if st, _ := m.GetStack("nextcloud"); st.UpdateError != "" || st.UpdatePhase != "" {
|
||||
t.Errorf("GetStack after the hold is lifted: phase=%q err=%q — the card would say a running app is stopped", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
for _, st := range m.GetStacks() {
|
||||
if st.Name == "nextcloud" && st.UpdateError != "" {
|
||||
t.Errorf("GetStacks after the hold is lifted still carries %q", st.UpdateError)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestR480_ARemovedAppShowsNoUpdateOutcome(t *testing.T) {
|
||||
m, _ := heldFailedUpdate(t)
|
||||
m.mu.Lock()
|
||||
m.stacks["nextcloud"].Deployed = false
|
||||
m.mu.Unlock()
|
||||
if st, _ := m.GetStack("nextcloud"); st.UpdateError != "" {
|
||||
t.Errorf("a removed app must not carry the held update's sentence, got %q", st.UpdateError)
|
||||
}
|
||||
}
|
||||
|
||||
func TestR480_AFailureThatHeldNothingKeepsItsSentence(t *testing.T) {
|
||||
m, _, _, c := newSlice4Manager(t)
|
||||
c.fail["pull"] = errors.New("manifest unknown")
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
st := waitUpdateDone(t, m, "nextcloud")
|
||||
if st.UpdateError != MsgUpdatePullFailed || st.UpdatePhase != UpdatePhaseFailed {
|
||||
t.Errorf("a pull failure held nothing and its sentence is still true; phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
}
|
||||
|
||||
// R-478 — a copy older than THIS install's deploy does not carry the update.
|
||||
|
||||
func runWithDeployedAt(t *testing.T, deployedAgo time.Duration) *fakeGuards {
|
||||
t.Helper()
|
||||
m, _, g, _ := newSlice4Manager(t)
|
||||
m.mu.Lock()
|
||||
if m.stacks["nextcloud"].AppConfig == nil {
|
||||
m.stacks["nextcloud"].AppConfig = &AppConfig{}
|
||||
}
|
||||
m.stacks["nextcloud"].AppConfig.DeployedAt = slice4T0.Add(-deployedAgo).Format(time.RFC3339)
|
||||
m.mu.Unlock()
|
||||
g.points = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-2 * time.Hour)}}
|
||||
g.pointsAfterBackup = []UpdateRestorePoint{{Tier: UpdateTierLocal, ProvenAt: slice4T0.Add(-time.Minute)}}
|
||||
if err := m.StartGuardedUpdate("nextcloud"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st := waitUpdateDone(t, m, "nextcloud"); st.UpdatePhase != UpdatePhaseDone {
|
||||
t.Fatalf("phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
func TestR478_ACopyOlderThanThisInstallDoesNotCount(t *testing.T) {
|
||||
if g := runWithDeployedAt(t, 10*time.Minute); !hasGuardCall(g, "BackupNow") {
|
||||
t.Errorf("a 2-hour-old unit under a 10-minute-old install belongs to the previous install — back up first; calls=%v", g.callList())
|
||||
}
|
||||
// Control: the same unit under a 3-hour-old install is this install's own copy.
|
||||
if g := runWithDeployedAt(t, 3*time.Hour); hasGuardCall(g, "BackupNow") {
|
||||
t.Errorf("control: a copy newer than the deploy must carry the update with no backup; calls=%v", g.callList())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
package stacks
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||||
)
|
||||
|
||||
// R-485 — the backup card sizes the recovery unit and the mirror, not two dead paths.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): put the old `db-dumps` + `secondary/<app>/rsync` paths back —
|
||||
// this fails with has_backups=false over a unit that exists.
|
||||
func TestR485_BackupCardSeesTheUnitAndTheMirror(t *testing.T) {
|
||||
m, _ := newPinManager(t, pinTplOld, pinTplNew, "deployed: true\nenv: {}\n")
|
||||
ns := t.TempDir()
|
||||
unit := appbackup.RecoveryUnitPath(ns, "nextcloud")
|
||||
if err := os.MkdirAll(filepath.Join(unit, "volume-dumps"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(unit, "volume-dumps", "x.tar"), make([]byte, 4096), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mirror := filepath.Join(t.TempDir(), "backups", "secondary", "nextcloud")
|
||||
if err := os.MkdirAll(mirror, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(mirror, "manifest.json"), []byte("{}"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
resp, err := m.GetStackBackupData("nextcloud", ns, []string{mirror})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !resp.HasBackups {
|
||||
t.Fatal("a recovery unit on disk IS a backup — has_backups must be true")
|
||||
}
|
||||
var seen []string
|
||||
for _, p := range resp.BackupPaths {
|
||||
if p.Exists {
|
||||
seen = append(seen, p.Path)
|
||||
}
|
||||
}
|
||||
if len(seen) != 2 || seen[0] != unit || seen[1] != mirror {
|
||||
t.Errorf("want the unit and the mirror, got %v", seen)
|
||||
}
|
||||
if resp.BackupPaths[0].SizeBytes < 4096 {
|
||||
t.Errorf("the unit's size must include its volume dumps, got %d", resp.BackupPaths[0].SizeBytes)
|
||||
}
|
||||
// No unit, no mirror: nothing to delete, honestly.
|
||||
if resp, _ := m.GetStackBackupData("nextcloud", t.TempDir(), nil); resp.HasBackups {
|
||||
t.Error("nothing on disk must be has_backups=false")
|
||||
}
|
||||
}
|
||||
@@ -148,6 +148,50 @@ func freshRestorePoint(now time.Time, maxAge time.Duration) func(UpdateRestorePo
|
||||
}
|
||||
}
|
||||
|
||||
// usableRestorePoint is freshRestorePoint plus R-478 (v0.240.0): a copy older than THIS install's
|
||||
// deploy does not count — it belongs to a previous install of the same app. Measured on demo-hp
|
||||
// 2026-09-13: a reinstalled gokapi leaned on a unit left by the removed install (06:59Z) for an update
|
||||
// at 15:31Z. A zero deployedAt (an app.yaml without deployed_at) applies no such rule. A restore also
|
||||
// rewrites deployed_at, so the update after a restore backs up first — slower, never less safe.
|
||||
// COMPANION RED-PROOF (REPORT.md): drop the deployedAt check — TestR478_… fails.
|
||||
func usableRestorePoint(now time.Time, maxAge time.Duration, deployedAt time.Time) func(UpdateRestorePoint) bool {
|
||||
fresh := freshRestorePoint(now, maxAge)
|
||||
return func(p UpdateRestorePoint) bool {
|
||||
if !deployedAt.IsZero() && p.ProvenAt.Before(deployedAt) {
|
||||
return false
|
||||
}
|
||||
return fresh(p)
|
||||
}
|
||||
}
|
||||
|
||||
// currentDeployTime is the app's recorded deployed_at, zero when absent or unreadable.
|
||||
func (m *Manager) currentDeployTime(name string) time.Time {
|
||||
st, ok := m.GetStack(name)
|
||||
if !ok || st.AppConfig == nil || st.AppConfig.DeployedAt == "" {
|
||||
return time.Time{}
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339, st.AppConfig.DeployedAt)
|
||||
if err != nil {
|
||||
return time.Time{}
|
||||
}
|
||||
return t
|
||||
}
|
||||
|
||||
func (m *Manager) markUpdateHeld(name string) {
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.updateHeld = true
|
||||
}
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
func fmtDeployTime(t time.Time) string {
|
||||
if t.IsZero() {
|
||||
return "unknown"
|
||||
}
|
||||
return t.UTC().Format(time.RFC3339)
|
||||
}
|
||||
|
||||
func describeRestorePoints(now time.Time, pts []UpdateRestorePoint) string {
|
||||
if len(pts) == 0 {
|
||||
return "none"
|
||||
@@ -189,11 +233,22 @@ func (m *Manager) guards() UpdateGuards {
|
||||
}
|
||||
|
||||
func fillHoldReason(g UpdateGuards, st *Stack) {
|
||||
if g == nil || st == nil || !st.Deployed {
|
||||
if st == nil {
|
||||
return
|
||||
}
|
||||
if held, why := g.HoldFor(st.Name); held {
|
||||
st.HoldReason = why
|
||||
held := false
|
||||
if g != nil && st.Deployed {
|
||||
if h, why := g.HoldFor(st.Name); h {
|
||||
st.HoldReason, held = why, true
|
||||
}
|
||||
}
|
||||
// R-480: an update that ended HELD carries the hold's sentence as its UpdateError. Once that hold
|
||||
// is lifted — a successful restore — or the app is removed, the sentence says a running (or absent)
|
||||
// app „leállítva marad", which is false. Measured on demo-hp 2026-09-13 after the „helyi" restore.
|
||||
// A failure that held nothing (a pull failure) keeps its sentence: it is still true.
|
||||
// COMPANION RED-PROOF (REPORT.md): delete this block — TestR480_… fails.
|
||||
if st.updateHeld && !st.Updating && (!st.Deployed || (g != nil && !held)) {
|
||||
st.UpdatePhase, st.UpdatePhaseLabel, st.UpdateError = "", "", ""
|
||||
}
|
||||
}
|
||||
|
||||
@@ -329,7 +384,7 @@ func (m *Manager) StartGuardedUpdate(name string) error {
|
||||
m.mu.Unlock()
|
||||
return m.refuseUpdate(name, "updating", fmt.Sprintf(MsgUpdateAlreadyFmt, name), "lost the race for the Updating flag")
|
||||
}
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseChecking, UpdatePhaseLabel(UpdatePhaseChecking)
|
||||
m.mu.Unlock()
|
||||
|
||||
@@ -445,11 +500,12 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
// applies to whichever tier is chosen. The first FRESH copy wins — not merely the first copy — so a
|
||||
// stale second-drive mirror never forces a backup while the app's own unit is minutes old.
|
||||
maxAge := m.backupMaxAge()
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, freshRestorePoint(start, maxAge))
|
||||
deployedAt := m.currentDeployTime(name)
|
||||
rp, ok, seen := g.RestorePoints(ctx, name, usableRestorePoint(start, maxAge, deployedAt))
|
||||
if ok {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: precondition met — %s copy from %s (%s old, limit %s)", name, updateTierName(rp.Tier), rp.ProvenAt.UTC().Format(time.RFC3339), start.Sub(rp.ProvenAt).Round(time.Minute), maxAge)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no copy younger than %s on any tier (found: %s) — backing up first", name, maxAge, describeRestorePoints(start, seen))
|
||||
m.logger.Printf("[INFO] [stacks] update %s: no usable copy on any tier — younger than %s and not older than this install's deploy (%s) (found: %s) — backing up first", name, maxAge, fmtDeployTime(deployedAt), describeRestorePoints(start, seen))
|
||||
if !m.enterUpdatePhase(name, &entry, UpdatePhaseBackingUp) {
|
||||
fail(MsgUpdateJournalFailed, "journal write failed")
|
||||
return
|
||||
@@ -459,7 +515,7 @@ func (m *Manager) runGuardedUpdate(ctx context.Context, name string) {
|
||||
return
|
||||
}
|
||||
now := m.now()
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, freshRestorePoint(now, maxAge))
|
||||
rp, ok, seen = g.RestorePoints(ctx, name, usableRestorePoint(now, maxAge, deployedAt))
|
||||
if !ok {
|
||||
fail(MsgUpdateBackupNoUnit, fmt.Sprintf("after the backup there is still no copy younger than %s on any tier (found: %s)", maxAge, describeRestorePoints(now, seen)))
|
||||
return
|
||||
@@ -577,6 +633,7 @@ func (m *Manager) failAndHold(ctx context.Context, name, dir string, env []strin
|
||||
m.logger.Printf("[ERROR] [stacks] update %s: %v", name, err)
|
||||
} else if _, why := g.HoldFor(name); why != "" {
|
||||
msg = why
|
||||
m.markUpdateHeld(name)
|
||||
}
|
||||
_ = m.RefreshStatus()
|
||||
m.clearJournal(name)
|
||||
@@ -814,7 +871,7 @@ func (m *Manager) RecoverUpdates() []string {
|
||||
m.logger.Printf("[WARN] [stacks] update recovery: %s was interrupted in %s (started %s) — the new version may have run; marking it Updating and RESUMING the health wait", name, e.Phase, e.StartedAt.Format(time.RFC3339))
|
||||
m.mu.Lock()
|
||||
if s, ok := m.stacks[name]; ok {
|
||||
s.Updating, s.UpdateError = true, ""
|
||||
s.Updating, s.UpdateError, s.updateHeld = true, "", false
|
||||
s.UpdatePhase, s.UpdatePhaseLabel = UpdatePhaseVerifying, UpdatePhaseLabel(UpdatePhaseVerifying)
|
||||
}
|
||||
m.updateResume = append(m.updateResume, name)
|
||||
|
||||
Reference in New Issue
Block a user