controller v0.296.0: a backup run cut off by a power cut or restart is said on the backup pages (R-519); the whole-system backup text states today's measurement (R-518)
gates / gates (push) Successful in 30s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-05 12:14:54 +02:00
parent 477e2548db
commit ff69074514
21 changed files with 2700 additions and 21 deletions
+24 -5
View File
@@ -299,6 +299,9 @@ type Manager struct {
// R-550: persistence of the above (restore_record.go) and the per-app interrupted notices.
opRecordPath string
opInterrupted map[string]RestoreOpResult
// R-519: the app-data run record (run_record.go) and the start of the last run a stop cut off (zero = none).
runRecordPath string
runInterrupted time.Time
// Cached status for page rendering (refreshed periodically)
cachedStatus *FullBackupStatus
@@ -311,9 +314,13 @@ type FullBackupStatus struct {
Running bool
// DB Dumps
LastDBDump *DBDumpStatus
DumpFiles []DumpFileInfo
DiscoveredDBs []DiscoveredDB
LastDBDump *DBDumpStatus
// InterruptedRunAt (R-519): the start of the last app-data run a power cut or a restart cut off; zero = the
// newest run finished. Shown on /backups and /backups/apps until a run ends with every step OK.
InterruptedRunAt time.Time
Interrupted bool // InterruptedRunAt is set (the template gate: a time.Time is always "true" to `if`)
DumpFiles []DumpFileInfo
DiscoveredDBs []DiscoveredDB
// Schedule
DBDumpSchedule string
@@ -548,6 +555,11 @@ func (m *Manager) beginOffsiteRunStamp(runID string) func() {
func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
start := time.Now()
m.logger.Printf("[INFO] [backup] Starting database dump run")
// R-519: the run is on record while it runs; a stop before markRunEnded leaves it "running", which the next
// start turns into the page's interrupted-run notice (run_record.go). Every return path ends the record.
m.markRunStarted(start)
runOK := false
defer func() { m.markRunEnded(runOK) }()
// R-181: open the per-run admission scope HERE, because this function is the single orchestrator
// of all three write legs. Each app's reserve verdict is taken at its first write of this run and
@@ -688,6 +700,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
// residue, invisible in the snapshot list. Guarded (deployed + different current drive only).
m.pruneStalePrimaryDirs()
runOK = allOK
// No silent partials: a DB-dump or volume-dump failure fails the whole run.
if !allOK {
return fmt.Errorf("some backup steps failed: %s", strings.Join(failedSummaryLines(summary), "; "))
@@ -1210,6 +1223,7 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
m.mu.Lock()
status.Running = m.running
status.LastDBDump = m.lastDBDump
status.InterruptedRunAt, status.Interrupted = m.runInterrupted, !m.runInterrupted.IsZero()
// Cross-check lastDBDump results inside lock to prevent torn writes.
if m.lastDBDump != nil && len(files) > 0 {
@@ -1252,7 +1266,8 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
// Update dynamic fields that don't need subprocess calls
status.Running = m.running
status.NextDBDump = nextDBDump
status.SingleCopyWarning = m.singleCopyWarning() // F6: honest single-drive signal
status.SingleCopyWarning = m.singleCopyWarning() // F6: honest single-drive signal
status.InterruptedRunAt, status.Interrupted = m.runInterrupted, !m.runInterrupted.IsZero() // R-519
// Deep-copy lastDBDump so callers cannot mutate shared state.
if m.lastDBDump != nil {
copyDump := *m.lastDBDump
@@ -1278,10 +1293,12 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
latestTime = f.ModTime
}
}
// R-519: after a restart the run's result is not in memory, so it is SYNTHESISED from the files. A run
// the stop cut off left a fresh .sql beside last night's tars — "Success" must not be read off that.
status.LastDBDump = &DBDumpStatus{
LastRun: latestTime,
Results: results,
Success: true,
Success: m.runInterrupted.IsZero(),
}
}
@@ -1295,6 +1312,8 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
DBDumpSchedule: m.cfg.Backup.DBDumpSchedule,
NextDBDump: nextDBDump,
SingleCopyWarning: m.singleCopyWarning(), // F6
InterruptedRunAt: m.runInterrupted, // R-519
Interrupted: !m.runInterrupted.IsZero(),
}
if m.lastDBDump != nil {
copyDump := *m.lastDBDump
+113
View File
@@ -0,0 +1,113 @@
package backup
import (
"encoding/json"
"os"
"time"
)
// R-519 (controller v0.296.0) — an app-data backup run cut off by a power cut or a restart is SAID on the page.
//
// MEASURED 2026-09-14 (BIGNIGHT F2): the power was cut during a volume dump; afterwards /backups and /backups/apps
// said nothing of an interruption, and every app read „Utolsó: 5 perce". The controller knew only when apps had
// been stopped (the app-stop marker); a cut during the DATABASE leg, with every app running, left no trace at all.
//
// Shape (the restore record's, R-550): one JSON file in DataDir, written atomically at BOTH ends of a run. At
// startup a file still marked running is, by construction, a run nothing is running any more: it becomes the
// INTERRUPTED notice, kept until the next app-data run that ends with every step OK. The restore points of a torn
// run already carry the time of their OLDEST part (the data block, v0.275.0) — this adds the sentence.
// No path set (a bare Manager in a test, a box with backup disabled) = no persistence, silently.
//
// Pinned by run_record_test.go (TestRunRecord_*).
type runRecordFile struct {
Running bool `json:"running"`
StartedAt time.Time `json:"started_at,omitempty"`
Interrupted time.Time `json:"interrupted,omitempty"` // the start of the run that was cut off; zero = none
}
// SetRunRecordPath wires persistence. Call before LoadRunRecord and before any backup runs.
func (m *Manager) SetRunRecordPath(path string) {
m.mu.Lock()
defer m.mu.Unlock()
m.runRecordPath = path
}
// LoadRunRecord reads the record at startup. It returns the start time of a run the stop cut off — non-zero
// exactly once per interruption (the conversion is written back), so the caller logs it once.
func (m *Manager) LoadRunRecord() time.Time {
m.mu.Lock()
defer m.mu.Unlock()
if m.runRecordPath == "" {
return time.Time{}
}
rec, ok := m.readRunRecordLocked()
if !ok {
return time.Time{}
}
m.runInterrupted = rec.Interrupted
if !rec.Running {
return time.Time{}
}
rec.Running = false
rec.Interrupted = rec.StartedAt
m.runInterrupted = rec.StartedAt
m.writeRunRecordLocked(rec)
return rec.StartedAt
}
// InterruptedRunAt is the start of the last app-data run that was cut off, or zero when the newest run finished.
func (m *Manager) InterruptedRunAt() time.Time {
m.mu.Lock()
defer m.mu.Unlock()
return m.runInterrupted
}
// markRunStarted is called at the start of an app-data run (runDBDumpsInternal).
func (m *Manager) markRunStarted(at time.Time) {
m.mu.Lock()
defer m.mu.Unlock()
if m.runRecordPath == "" {
return
}
m.writeRunRecordLocked(runRecordFile{Running: true, StartedAt: at, Interrupted: m.runInterrupted})
}
// markRunEnded is called on EVERY return of the run. allOK clears the interrupted notice: the household's copies
// are whole again. A run that ended with a failed step keeps it (that run has its own failure line too).
func (m *Manager) markRunEnded(allOK bool) {
m.mu.Lock()
defer m.mu.Unlock()
if allOK {
m.runInterrupted = time.Time{}
}
if m.runRecordPath == "" {
return
}
m.writeRunRecordLocked(runRecordFile{Running: false, Interrupted: m.runInterrupted})
}
func (m *Manager) readRunRecordLocked() (runRecordFile, bool) {
var rec runRecordFile
b, err := os.ReadFile(m.runRecordPath)
if err != nil {
if !os.IsNotExist(err) && m.logger != nil {
m.logger.Printf("[WARN] [backup] run record unreadable (%v) — no interrupted-run notice", err)
}
return rec, false
}
if err := json.Unmarshal(b, &rec); err != nil {
if m.logger != nil {
m.logger.Printf("[WARN] [backup] run record is not JSON (%v) — no interrupted-run notice", err)
}
return rec, false
}
return rec, true
}
func (m *Manager) writeRunRecordLocked(rec runRecordFile) {
b, _ := json.Marshal(rec)
if err := atomicWrite(m.runRecordPath, b, 0o600); err != nil && m.logger != nil {
m.logger.Printf("[WARN] [backup] could not persist the run record to %s: %v", m.runRecordPath, err)
}
}
@@ -0,0 +1,99 @@
package backup
import (
"context"
"encoding/json"
"errors"
"os"
"path/filepath"
"testing"
"time"
)
// R-519 (controller v0.296.0). RED-PROOFS recorded in felhom.eu/documentation/audits/hub-safety-2026-10-05/partE/:
// 1. delete the m.markRunStarted call in runDBDumpsInternal → TestRunRecord_TheRealRunIsOnRecordWhileItRuns fails;
// 2. synthesise LastDBDump with `Success: true` again → TestRunRecord_SynthesisedStatusIsNotOKAfterACut fails.
// A run the stop cut off is said — once on load, then on the page until a run ends with every step OK. A run that
// ended with a failed step does not clear it.
func TestRunRecord_CutRunIsSaidUntilACompleteOne(t *testing.T) {
path := filepath.Join(t.TempDir(), "appdata-run.json")
t0 := time.Date(2026, 9, 14, 19, 40, 3, 0, time.UTC)
before, _ := newTestManager(t, "/srv/sys")
before.SetRunRecordPath(path)
before.markRunStarted(t0) // the power goes here: no markRunEnded
after, _ := newTestManager(t, "/srv/sys")
after.SetRunRecordPath(path)
if got := after.LoadRunRecord(); !got.Equal(t0) {
t.Fatalf("the cut run was not found at startup: %v", got)
}
if !after.InterruptedRunAt().Equal(t0) || !after.GetFullStatus(time.Time{}).Interrupted {
t.Fatal("the page status does not carry the cut run")
}
again, _ := newTestManager(t, "/srv/sys")
again.SetRunRecordPath(path)
if got := again.LoadRunRecord(); !got.IsZero() {
t.Fatalf("the same cut was reported twice: %v", got)
}
if !again.InterruptedRunAt().Equal(t0) {
t.Fatal("the notice must survive a second restart until a complete run")
}
again.markRunStarted(t0.Add(time.Hour))
again.markRunEnded(false) // a run with a failed step
if !again.InterruptedRunAt().Equal(t0) {
t.Fatal("a run with a failed step cleared the notice")
}
again.markRunStarted(t0.Add(2 * time.Hour))
again.markRunEnded(true)
if !again.InterruptedRunAt().IsZero() || again.GetFullStatus(time.Time{}).Interrupted {
t.Fatal("a complete run did not clear the notice")
}
last, _ := newTestManager(t, "/srv/sys")
last.SetRunRecordPath(path)
if got := last.LoadRunRecord(); !got.IsZero() || !last.InterruptedRunAt().IsZero() {
t.Fatal("the cleared notice came back after a restart")
}
}
// The production path: runDBDumpsInternal itself puts the run on record (running=true) before its first step,
// and takes it off on an early error return.
func TestRunRecord_TheRealRunIsOnRecordWhileItRuns(t *testing.T) {
path := filepath.Join(t.TempDir(), "appdata-run.json")
m, _ := newTestManager(t, t.TempDir())
m.SetRunRecordPath(path)
sawRunning := false
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
var rec runRecordFile
if b, err := os.ReadFile(path); err == nil && json.Unmarshal(b, &rec) == nil && rec.Running {
sawRunning = true
}
return nil, errors.New("docker is not reachable")
}
_ = m.runDBDumpsInternal(context.Background())
if !sawRunning {
t.Fatal("the run was not on record while it ran — a cut here would go unnoticed")
}
var rec runRecordFile
b, _ := os.ReadFile(path)
if json.Unmarshal(b, &rec) != nil || rec.Running {
t.Fatalf("an ended run is still on record as running: %s", b)
}
}
// After a restart the page's "last database backup" is synthesised from the files. After a cut it must not read OK
// (BIGNIGHT F2: „Utolsó adatbázis mentés … OK" over a torn run).
func TestRunRecord_SynthesisedStatusIsNotOKAfterACut(t *testing.T) {
m, _ := newTestManager(t, "/srv/sys")
m.cachedStatus = &FullBackupStatus{DumpFiles: []DumpFileInfo{{StackName: "adventurelog", FileName: "adventurelog-postgres.sql", ModTime: time.Now()}}}
if st := m.GetFullStatus(time.Time{}); st.LastDBDump == nil || !st.LastDBDump.Success {
t.Fatal("control: with no cut the synthesised status reads OK")
}
m.runInterrupted = time.Now().Add(-time.Minute)
if st := m.GetFullStatus(time.Time{}); st.LastDBDump == nil || st.LastDBDump.Success || !st.Interrupted {
t.Fatalf("after a cut the synthesised status still reads OK: %+v", st.LastDBDump)
}
}