hub v0.136.0: nightly VACUUM INTO snapshot, keep 2 (R-173 decision A); hub PVC 2Gi + Longhorn default group; 09 rulings 125-127; 05 §16.3
gates / gates (push) Failing after 14m0s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 14:28:30 +02:00
parent 0826e41b31
commit d4be9f6ff1
11 changed files with 544 additions and 3 deletions
+150
View File
@@ -0,0 +1,150 @@
// Package dbsnap makes the hub's nightly database snapshot (R-173, hub v0.136.0, `05` §16.3,
// runbooks/RUNBOOK-hub-db-offsite-backup.md Step 3).
//
// Every night the hub writes `<data>/snapshots/hub-<UTC stamp>.db` with SQLite's VACUUM INTO — one consistent point
// in time, WAL-aware — then keeps the newest Keep (2). DooPlex picks up the newest one, checks it, encrypts it and
// pushes it to ep0 (Step 4). The hub never ships anything itself: it has no ep0 credential.
//
// Written as `<name>.tmp` and renamed, so a reader never sees half a file; a run never overlaps another (a second call
// returns ErrBusy). Pinned by dbsnap_test.go.
package dbsnap
import (
"context"
"errors"
"fmt"
"log"
"os"
"path/filepath"
"sort"
"strings"
"sync"
"time"
)
// Keep is how many snapshots stay on the volume (the newest). Decided by CC — operator may reverse (the volume was
// grown to 2 GiB for it, operator choice 2026-10-05).
const Keep = 2
// Prefix and Suffix make a snapshot's file name: hub-20261005T020000Z.db.
const (
Prefix = "hub-"
Suffix = ".db"
)
// ErrBusy is returned when a snapshot is already being written.
var ErrBusy = errors.New("dbsnap: a snapshot is already being written")
// Snapshotter is the store's VACUUM INTO.
type Snapshotter interface {
SnapshotInto(ctx context.Context, path string) error
}
// Maker writes and prunes snapshots in Dir.
type Maker struct {
Store Snapshotter
Dir string
Logger *log.Logger
Now func() time.Time
mu sync.Mutex
running bool
}
// Result is one snapshot written.
type Result struct {
Name string `json:"name"`
Bytes int64 `json:"bytes"`
Duration time.Duration `json:"duration_ns"`
Pruned []string `json:"pruned,omitempty"`
}
// Make writes one snapshot and prunes to Keep. Safe to call from the nightly job and the operator button at once.
func (m *Maker) Make(ctx context.Context) (Result, error) {
m.mu.Lock()
if m.running {
m.mu.Unlock()
return Result{}, ErrBusy
}
m.running = true
m.mu.Unlock()
defer func() { m.mu.Lock(); m.running = false; m.mu.Unlock() }()
now := time.Now
if m.Now != nil {
now = m.Now
}
if err := os.MkdirAll(m.Dir, 0o700); err != nil {
return Result{}, fmt.Errorf("dbsnap: %w", err)
}
name := Prefix + now().UTC().Format("20060102T150405Z") + Suffix
final := filepath.Join(m.Dir, name)
tmp := final + ".tmp"
_ = os.Remove(tmp) // a leftover from a crash mid-write
start := time.Now()
if err := m.Store.SnapshotInto(ctx, tmp); err != nil {
_ = os.Remove(tmp)
return Result{}, err
}
if err := os.Chmod(tmp, 0o600); err != nil {
_ = os.Remove(tmp)
return Result{}, fmt.Errorf("dbsnap: chmod: %w", err)
}
if err := os.Rename(tmp, final); err != nil {
_ = os.Remove(tmp)
return Result{}, fmt.Errorf("dbsnap: rename: %w", err)
}
fi, err := os.Stat(final)
if err != nil {
return Result{}, fmt.Errorf("dbsnap: stat: %w", err)
}
res := Result{Name: name, Bytes: fi.Size(), Duration: time.Since(start)}
res.Pruned = m.prune()
if m.Logger != nil {
m.Logger.Printf("[INFO] db snapshot written: %s (%d bytes, %s); pruned %d, keeping %d",
name, res.Bytes, res.Duration.Round(time.Millisecond), len(res.Pruned), Keep)
}
return res, nil
}
// List returns the snapshot names in Dir, oldest first (the stamp sorts as text).
func List(dir string) []string {
entries, _ := os.ReadDir(dir)
var out []string
for _, e := range entries {
n := e.Name()
if !e.IsDir() && strings.HasPrefix(n, Prefix) && strings.HasSuffix(n, Suffix) {
out = append(out, n)
}
}
sort.Strings(out)
return out
}
func (m *Maker) prune() []string {
names := List(m.Dir)
var pruned []string
for len(names) > Keep {
if err := os.Remove(filepath.Join(m.Dir, names[0])); err != nil && m.Logger != nil {
m.Logger.Printf("[WARN] db snapshot prune %s: %v", names[0], err)
}
pruned = append(pruned, names[0])
names = names[1:]
}
return pruned
}
// NeedsCatchUp reports whether the newest snapshot in dir is missing or older than maxAge — the hub runs one at
// start-up then, so a pod that was down at 02:00 does not leave DooPlex pushing a two-day-old copy.
func NeedsCatchUp(dir string, now time.Time, maxAge time.Duration) bool {
names := List(dir)
if len(names) == 0 {
return true
}
stamp := strings.TrimSuffix(strings.TrimPrefix(names[len(names)-1], Prefix), Suffix)
t, err := time.Parse("20060102T150405Z", stamp)
if err != nil {
return true
}
return now.Sub(t) > maxAge
}
+208
View File
@@ -0,0 +1,208 @@
package dbsnap
import (
"context"
"database/sql"
"errors"
"fmt"
"io"
"log"
"os"
"path/filepath"
"strings"
"sync"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
)
func newStore(t *testing.T) (*store.Store, string) {
t.Helper()
p := filepath.Join(t.TempDir(), "hub.db")
s, err := store.New(p, log.New(io.Discard, "", 0))
if err != nil {
t.Fatalf("store.New: %v", err)
}
t.Cleanup(func() { s.Close() })
return s, p
}
func openRO(t *testing.T, p string) *sql.DB {
t.Helper()
db, err := sql.Open("sqlite", "file:"+p+"?mode=ro")
if err != nil {
t.Fatalf("open %s: %v", p, err)
}
t.Cleanup(func() { db.Close() })
return db
}
// rowCounts returns count(*) of every table.
func rowCounts(t *testing.T, db *sql.DB) map[string]int {
t.Helper()
rows, err := db.Query(`SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%' ORDER BY name`)
if err != nil {
t.Fatalf("list tables: %v", err)
}
var names []string
for rows.Next() {
var n string
_ = rows.Scan(&n)
names = append(names, n)
}
rows.Close()
out := map[string]int{}
for _, n := range names {
var c int
if err := db.QueryRow(fmt.Sprintf(`SELECT count(*) FROM "%s"`, n)).Scan(&c); err != nil {
t.Fatalf("count %s: %v", n, err)
}
out[n] = c
}
return out
}
// TestSnapshot_ConsistentWithLiveDBIncludingWAL is the consequence test: the snapshot passes integrity_check and has
// the live database's row count in EVERY table — including rows that so far live only in hub.db-wal. Precondition
// asserted in-test: copying hub.db alone (the naive file backup) LOSES those rows, so this test can tell the two apart.
func TestSnapshot_ConsistentWithLiveDBIncludingWAL(t *testing.T) {
s, live := newStore(t)
for i := 0; i < 150; i++ {
if err := s.UpsertHost(&store.Host{HostID: fmt.Sprintf("h%03d", i), CustomerID: "c1", APIKey: fmt.Sprintf("k%03d", i)}); err != nil {
t.Fatalf("UpsertHost: %v", err)
}
}
if fi, err := os.Stat(live + "-wal"); err != nil || fi.Size() == 0 {
t.Fatalf("precondition: want a non-empty WAL, got %v", err)
}
// Precondition: a plain copy of hub.db does not hold the WAL rows.
naive := filepath.Join(t.TempDir(), "naive.db")
b, _ := os.ReadFile(live)
_ = os.WriteFile(naive, b, 0o600)
if c := rowCounts(t, openRO(t, naive))["hosts"]; c == 150 {
t.Fatalf("precondition: a plain file copy already holds all 150 hosts — the test cannot tell a WAL-blind copy apart")
}
dir := filepath.Join(t.TempDir(), "snapshots")
m := &Maker{Store: s, Dir: dir}
res, err := m.Make(context.Background())
if err != nil {
t.Fatalf("Make: %v", err)
}
snap := openRO(t, filepath.Join(dir, res.Name))
var ic string
if err := snap.QueryRow(`PRAGMA integrity_check`).Scan(&ic); err != nil || ic != "ok" {
t.Fatalf("integrity_check = %q, %v", ic, err)
}
want, got := rowCounts(t, openRO(t, live)), rowCounts(t, snap)
if len(want) < 10 || want["hosts"] != 150 {
t.Fatalf("live counts look wrong: %d tables, hosts=%d", len(want), want["hosts"])
}
for tbl, n := range want {
if got[tbl] != n {
t.Errorf("table %s: snapshot %d rows, live %d", tbl, got[tbl], n)
}
}
if fi, _ := os.Stat(filepath.Join(dir, res.Name)); fi == nil || fi.Mode().Perm() != 0o600 || res.Bytes != fi.Size() {
t.Errorf("snapshot file mode/size wrong: %+v res.Bytes=%d", fi, res.Bytes)
}
if _, err := os.Stat(filepath.Join(dir, res.Name+".tmp")); !os.IsNotExist(err) {
t.Errorf("the .tmp file is left behind")
}
}
// TestSnapshot_KeepsNewestTwo pins Keep: three runs leave the two newest.
func TestSnapshot_KeepsNewestTwo(t *testing.T) {
s, _ := newStore(t)
dir := filepath.Join(t.TempDir(), "snapshots")
base := time.Date(2026, 10, 5, 0, 0, 0, 0, time.UTC)
i := 0
m := &Maker{Store: s, Dir: dir, Now: func() time.Time { i++; return base.Add(time.Duration(i) * 24 * time.Hour) }}
for k := 0; k < 3; k++ {
if _, err := m.Make(context.Background()); err != nil {
t.Fatalf("Make %d: %v", k, err)
}
}
got := List(dir)
want := []string{"hub-20261007T000000Z.db", "hub-20261008T000000Z.db"}
if strings.Join(got, ",") != strings.Join(want, ",") {
t.Fatalf("kept %v, want %v", got, want)
}
}
type blockingStore struct {
in, release chan struct{}
once *sync.Once
}
func (b blockingStore) SnapshotInto(ctx context.Context, path string) error {
b.once.Do(func() { close(b.in) })
<-b.release
return os.WriteFile(path, []byte("x"), 0o600)
}
// TestSnapshot_NeverTwoAtOnce: a second Make while the first runs returns ErrBusy and writes nothing.
func TestSnapshot_NeverTwoAtOnce(t *testing.T) {
bs := blockingStore{in: make(chan struct{}), release: make(chan struct{}), once: &sync.Once{}}
dir := t.TempDir()
m := &Maker{Store: bs, Dir: dir}
done := make(chan error, 1)
go func() { _, err := m.Make(context.Background()); done <- err }()
<-bs.in
second := make(chan error, 1)
go func() { _, err := m.Make(context.Background()); second <- err }()
select {
case err := <-second:
if !errors.Is(err, ErrBusy) {
t.Fatalf("second Make = %v, want ErrBusy", err)
}
case <-time.After(2 * time.Second):
close(bs.release)
t.Fatalf("second Make did not return ErrBusy at once: it ran alongside the first")
}
close(bs.release)
if err := <-done; err != nil {
t.Fatalf("first Make: %v", err)
}
if n := len(List(dir)); n != 1 {
t.Fatalf("%d snapshots, want 1", n)
}
m.Now = func() time.Time { return time.Now().Add(time.Hour) } // a distinct name
if _, err := m.Make(context.Background()); err != nil {
t.Fatalf("Make after the first ended: %v", err)
}
}
type failStore struct{}
func (failStore) SnapshotInto(ctx context.Context, path string) error {
_ = os.WriteFile(path, []byte("half"), 0o600)
return errors.New("disk full")
}
// TestSnapshot_FailureLeavesNoFile: a failed VACUUM INTO leaves neither a snapshot nor a .tmp for DooPlex to pick up.
func TestSnapshot_FailureLeavesNoFile(t *testing.T) {
dir := t.TempDir()
if _, err := (&Maker{Store: failStore{}, Dir: dir}).Make(context.Background()); err == nil {
t.Fatal("Make succeeded over a failing store")
}
if e, _ := os.ReadDir(dir); len(e) != 0 {
t.Fatalf("left %d file(s) behind", len(e))
}
}
func TestNeedsCatchUp(t *testing.T) {
dir := t.TempDir()
now := time.Date(2026, 10, 5, 12, 0, 0, 0, time.UTC)
if !NeedsCatchUp(dir, now, 24*time.Hour) {
t.Error("empty dir: want catch-up")
}
_ = os.WriteFile(filepath.Join(dir, "hub-20261005T000000Z.db"), nil, 0o600)
if NeedsCatchUp(dir, now, 24*time.Hour) {
t.Error("12 h old: want no catch-up")
}
if !NeedsCatchUp(dir, now.Add(13*time.Hour), 24*time.Hour) {
t.Error("25 h old: want catch-up")
}
}
+17
View File
@@ -0,0 +1,17 @@
package store
import (
"context"
"fmt"
)
// SnapshotInto writes a consistent, compacted copy of the whole database to path with SQLite's `VACUUM INTO`
// (R-173, hub v0.136.0). One statement inside one read transaction: it sees one point in time and includes every
// committed write still sitting in hub.db-wal — unlike copying hub.db (+ -wal) as files while the hub writes. The
// target must not exist. Pinned by internal/dbsnap (TestSnapshot_*).
func (s *Store) SnapshotInto(ctx context.Context, path string) error {
if _, err := s.db.ExecContext(ctx, `VACUUM INTO ?`, path); err != nil {
return fmt.Errorf("store: VACUUM INTO %s: %w", path, err)
}
return nil
}