bdcbd50b42
gates / gates (push) Successful in 13s
R-486 (P1): removing an app with its backups KEPT keeps its Tier-2 record, so the second-drive restore is no longer refused over an intact mirror. R-484: postgis/pgvector/timescaledb images are Postgres (logical dumps). R-485: the backup card sizes the recovery unit and the mirror(s). R-480: a held update's sentence leaves the card once the hold is lifted. R-477: the update's off-site lookup is one snapshots call, no stats. R-478: a copy older than this install's deploy does not count. R-474: "delete backups" deletes the unit, the mirror(s) and the prefs. Tests and red-proofs per row; evidence in felhom.eu documentation/audits/v0240-2026-09-13/ and nightly-2026-09-13-adventurelog/.
209 lines
8.6 KiB
Go
209 lines
8.6 KiB
Go
package backup
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"sort"
|
|
"sync"
|
|
"time"
|
|
)
|
|
|
|
// R-193 Part 3 — WHAT IS IN THERE. After a successful unlock the customer is shown the contents of the
|
|
// repository they just opened: which apps, from when, how big.
|
|
//
|
|
// READ-ONLY, AND THAT IS THE POINT. This restores nothing, puts nothing back, and compares nothing
|
|
// against live data. Unlocking and restoring are separate (operator ruling, 2026-08-05): restore is
|
|
// already per-app and already lives in the backups area, and a screen that unlocks and then offers to
|
|
// overwrite is two decisions wearing one button.
|
|
//
|
|
// WHY A LISTING AT ALL, rather than a success message: "unlocked" with nothing shown is
|
|
// indistinguishable from having unlocked an EMPTY store, and the customer has no way to tell whether
|
|
// what came back is the right thing. Seeing their own app names and dates is how they know.
|
|
|
|
// errNoOffsiteTarget is returned when the repository cannot even be addressed — no off-site target is
|
|
// configured on this box yet. Distinguished from a read failure because the remedy differs: this one
|
|
// resolves by itself once the tier is re-applied.
|
|
var errNoOffsiteTarget = errors.New("no off-site target is configured on this box yet")
|
|
|
|
// ErrNoOffsiteTarget reports whether err is the not-yet-configured case, so a caller can say the right
|
|
// thing rather than showing a generic failure.
|
|
func ErrNoOffsiteTarget(err error) bool { return errors.Is(err, errNoOffsiteTarget) }
|
|
|
|
// ErrNoOffsiteTargetSentinel exposes the sentinel itself so other packages — and their tests — can
|
|
// construct the not-yet-configured case. Added for R-237, whose restore list must distinguish
|
|
// "no target yet" (resolves by itself) from "could not read" (does not), and must be able to pin
|
|
// both in a table test.
|
|
func ErrNoOffsiteTargetSentinel() error { return errNoOffsiteTarget }
|
|
|
|
// OffsiteInventoryApp is one app's presence in the opened repository. Non-secret throughout.
|
|
type OffsiteInventoryApp struct {
|
|
App string // the restic tag == the stack name
|
|
LatestAt time.Time // the newest snapshot's time for this app
|
|
SizeBytes int64 // restore size of that newest snapshot (0 = could not be determined)
|
|
}
|
|
|
|
// OffsiteInventory is the whole answer, including the EMPTY case stated explicitly.
|
|
type OffsiteInventory struct {
|
|
Apps []OffsiteInventoryApp
|
|
// Empty is true when the repository opened cleanly and holds no snapshots. It is a real and
|
|
// confusing outcome — a bare list there reads as a broken page — so it is named rather than
|
|
// inferred from len(Apps)==0, which is also what a failed read looks like.
|
|
Empty bool
|
|
}
|
|
|
|
// offsiteNewest is one app tag's newest snapshot.
|
|
type offsiteNewest struct {
|
|
id string
|
|
at time.Time
|
|
}
|
|
|
|
// offsiteNewestPerTag runs ONE `snapshots --json` and returns the newest snapshot per app tag, and
|
|
// whether the repository opened cleanly and holds no snapshots at all. Shared by the inventory page
|
|
// and the update precondition (R-477), so the two cannot disagree about what is in the repository.
|
|
func (m *Manager) offsiteNewestPerTag(ctx context.Context) (map[string]offsiteNewest, bool, error) {
|
|
// A box can hold a recovered key and still have no off-site COORDINATES — the pristine rebuilt
|
|
// shape, before its target is re-applied. Reading the repository is impossible then, and saying so
|
|
// is the honest answer; without this guard offboxBaseArgs nil-derefs on the missing target.
|
|
if !m.OffboxConfigured() {
|
|
return nil, false, errNoOffsiteTarget
|
|
}
|
|
t := m.settings.GetOffboxTarget()
|
|
base, env := m.offboxBaseArgs(t)
|
|
sctx, cancel := context.WithTimeout(ctx, offboxProbeTimeout)
|
|
defer cancel()
|
|
out, err := m.runner()(sctx, env, append(append([]string{}, base...), "snapshots", "--json")...)
|
|
if err != nil {
|
|
return nil, false, err
|
|
}
|
|
var snaps []struct {
|
|
ShortID string `json:"short_id"`
|
|
ID string `json:"id"`
|
|
Time time.Time `json:"time"`
|
|
Tags []string `json:"tags"`
|
|
}
|
|
if uerr := json.Unmarshal(out, &snaps); uerr != nil {
|
|
return nil, false, uerr
|
|
}
|
|
if len(snaps) == 0 {
|
|
return nil, true, nil
|
|
}
|
|
// Newest snapshot per tag. A snapshot may carry several tags; each names an app it belongs to.
|
|
newest := map[string]offsiteNewest{}
|
|
for _, sn := range snaps {
|
|
id := sn.ShortID
|
|
if id == "" {
|
|
id = sn.ID
|
|
}
|
|
for _, tag := range sn.Tags {
|
|
if tag == "" {
|
|
continue
|
|
}
|
|
if cur, ok := newest[tag]; !ok || sn.Time.After(cur.at) {
|
|
newest[tag] = offsiteNewest{id: id, at: sn.Time}
|
|
}
|
|
}
|
|
}
|
|
return newest, false, nil
|
|
}
|
|
|
|
// OffsiteSnapshotTimes (R-477, v0.240.0) is the newest snapshot time per app — ONE `snapshots --json`,
|
|
// no per-app `stats`. It is what the update precondition needs. Measured on demo-hp 2026-09-13: going
|
|
// through OffsiteInventoryList instead, the update's check spent its whole 15 s bound on the size calls
|
|
// and the bound killed one for an unrelated app (`size of kimai's newest snapshot unknown: signal:
|
|
// killed`). Pinned by TestR477_TheUpdateOffsiteLookupRunsNoStats.
|
|
func (m *Manager) OffsiteSnapshotTimes(ctx context.Context) (map[string]time.Time, error) {
|
|
newest, _, err := m.offsiteNewestPerTag(ctx)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := make(map[string]time.Time, len(newest))
|
|
for tag, n := range newest {
|
|
out[tag] = n.at
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// OffsiteInventoryList opens the repository and reports what is in it, grouped per app. One
|
|
// `snapshots --json` call for the whole repo, then one `stats` per app for the newest snapshot's size.
|
|
//
|
|
// A per-app size failure is NOT fatal: the app is still listed, with SizeBytes 0, because knowing an
|
|
// app is in there matters more than knowing how big it is, and dropping it would under-report the
|
|
// customer's own data.
|
|
func (m *Manager) OffsiteInventoryList(ctx context.Context) (OffsiteInventory, error) {
|
|
var inv OffsiteInventory
|
|
newest, empty, err := m.offsiteNewestPerTag(ctx)
|
|
if err != nil {
|
|
return inv, err
|
|
}
|
|
if empty {
|
|
inv.Empty = true
|
|
return inv, nil
|
|
}
|
|
if len(newest) == 0 {
|
|
// Snapshots exist but carry no tags — not "empty", and saying so would be a lie. Report an
|
|
// empty app list without the Empty flag; the page renders the honest in-between wording.
|
|
return inv, nil
|
|
}
|
|
// R-351 Part 4 — THE SIZE CALLS RUN CONCURRENTLY, BOUNDED.
|
|
//
|
|
// MEASURED before changing anything, on demo-hp against the live off-site target
|
|
// (u629488-sub3.your-storagebox.de:23), 2026-08-21:
|
|
//
|
|
// restic snapshots --json (once, whole repo) 2605 ms
|
|
// restic stats --mode restore-size (per app) 2697 ms each, 5 app tags, SEQUENTIAL
|
|
// => 2605 + 5*2697 = ~16.1 s
|
|
//
|
|
// which is the ten-to-fifteen seconds the page was reported to take. The cause is the shape
|
|
// already on file — one network call per app, one after another — so the fix is the same one:
|
|
// run them at once. Each call is an independent SSH round-trip to the repository and `stats` is
|
|
// a READ (restic takes a shared lock), so they do not contend.
|
|
//
|
|
// WHY BOUNDED, and why the bound is small: the target is a Hetzner Storage Box, which caps
|
|
// concurrent SSH sessions. Unbounded fan-out over a large app list would trade a slow page for
|
|
// refused connections — and a refused size call degrades to SizeBytes 0, i.e. it would quietly
|
|
// UNDER-REPORT the customer's own data rather than fail loudly. Four keeps well clear of the cap
|
|
// and still collapses the common case to a single wave.
|
|
const inventorySizeConcurrency = 4
|
|
|
|
type sized struct {
|
|
app OffsiteInventoryApp
|
|
err error
|
|
}
|
|
results := make([]sized, 0, len(newest))
|
|
var mu sync.Mutex
|
|
var wg sync.WaitGroup
|
|
sem := make(chan struct{}, inventorySizeConcurrency)
|
|
for tag, n := range newest {
|
|
wg.Add(1)
|
|
go func(tag, id string, at time.Time) {
|
|
defer wg.Done()
|
|
sem <- struct{}{}
|
|
defer func() { <-sem }()
|
|
app := OffsiteInventoryApp{App: tag, LatestAt: at}
|
|
size, serr := m.offboxSnapshotSize(ctx, id)
|
|
if serr == nil {
|
|
app.SizeBytes = size
|
|
}
|
|
mu.Lock()
|
|
results = append(results, sized{app: app, err: serr})
|
|
mu.Unlock()
|
|
}(tag, n.id, n.at)
|
|
}
|
|
wg.Wait()
|
|
// Logging happens on the caller's goroutine, after the fan-out: m.logger is shared and the
|
|
// per-app WARN is the only thing that tells an operator a size is missing rather than zero.
|
|
for _, r := range results {
|
|
if r.err != nil {
|
|
m.logger.Printf("[WARN] [offbox] inventory: size of %s's newest snapshot unknown: %v (listing it anyway)", r.app.App, r.err)
|
|
}
|
|
inv.Apps = append(inv.Apps, r.app)
|
|
}
|
|
sort.Slice(inv.Apps, func(i, j int) bool { return inv.Apps[i].App < inv.Apps[j].App })
|
|
return inv, nil
|
|
}
|
|
|
|
// HumanizeBytes exposes the shared byte formatter to the web layer so the recovery page renders sizes
|
|
// the same way every other surface does.
|
|
func HumanizeBytes(n int64) string { return humanizeBytes(n) }
|