controller v0.296.0: a backup run cut off by a power cut or restart is said on the backup pages (R-519); the whole-system backup text states today's measurement (R-518)
gates / gates (push) Successful in 30s
gates / gates (push) Successful in 30s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -299,6 +299,9 @@ type Manager struct {
|
||||
// R-550: persistence of the above (restore_record.go) and the per-app interrupted notices.
|
||||
opRecordPath string
|
||||
opInterrupted map[string]RestoreOpResult
|
||||
// R-519: the app-data run record (run_record.go) and the start of the last run a stop cut off (zero = none).
|
||||
runRecordPath string
|
||||
runInterrupted time.Time
|
||||
|
||||
// Cached status for page rendering (refreshed periodically)
|
||||
cachedStatus *FullBackupStatus
|
||||
@@ -311,9 +314,13 @@ type FullBackupStatus struct {
|
||||
Running bool
|
||||
|
||||
// DB Dumps
|
||||
LastDBDump *DBDumpStatus
|
||||
DumpFiles []DumpFileInfo
|
||||
DiscoveredDBs []DiscoveredDB
|
||||
LastDBDump *DBDumpStatus
|
||||
// InterruptedRunAt (R-519): the start of the last app-data run a power cut or a restart cut off; zero = the
|
||||
// newest run finished. Shown on /backups and /backups/apps until a run ends with every step OK.
|
||||
InterruptedRunAt time.Time
|
||||
Interrupted bool // InterruptedRunAt is set (the template gate: a time.Time is always "true" to `if`)
|
||||
DumpFiles []DumpFileInfo
|
||||
DiscoveredDBs []DiscoveredDB
|
||||
|
||||
// Schedule
|
||||
DBDumpSchedule string
|
||||
@@ -548,6 +555,11 @@ func (m *Manager) beginOffsiteRunStamp(runID string) func() {
|
||||
func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
start := time.Now()
|
||||
m.logger.Printf("[INFO] [backup] Starting database dump run")
|
||||
// R-519: the run is on record while it runs; a stop before markRunEnded leaves it "running", which the next
|
||||
// start turns into the page's interrupted-run notice (run_record.go). Every return path ends the record.
|
||||
m.markRunStarted(start)
|
||||
runOK := false
|
||||
defer func() { m.markRunEnded(runOK) }()
|
||||
|
||||
// R-181: open the per-run admission scope HERE, because this function is the single orchestrator
|
||||
// of all three write legs. Each app's reserve verdict is taken at its first write of this run and
|
||||
@@ -688,6 +700,7 @@ func (m *Manager) runDBDumpsInternal(ctx context.Context) error {
|
||||
// residue, invisible in the snapshot list. Guarded (deployed + different current drive only).
|
||||
m.pruneStalePrimaryDirs()
|
||||
|
||||
runOK = allOK
|
||||
// No silent partials: a DB-dump or volume-dump failure fails the whole run.
|
||||
if !allOK {
|
||||
return fmt.Errorf("some backup steps failed: %s", strings.Join(failedSummaryLines(summary), "; "))
|
||||
@@ -1210,6 +1223,7 @@ func (m *Manager) RefreshCache(nextDBDump time.Time) {
|
||||
m.mu.Lock()
|
||||
status.Running = m.running
|
||||
status.LastDBDump = m.lastDBDump
|
||||
status.InterruptedRunAt, status.Interrupted = m.runInterrupted, !m.runInterrupted.IsZero()
|
||||
|
||||
// Cross-check lastDBDump results inside lock to prevent torn writes.
|
||||
if m.lastDBDump != nil && len(files) > 0 {
|
||||
@@ -1252,7 +1266,8 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
// Update dynamic fields that don't need subprocess calls
|
||||
status.Running = m.running
|
||||
status.NextDBDump = nextDBDump
|
||||
status.SingleCopyWarning = m.singleCopyWarning() // F6: honest single-drive signal
|
||||
status.SingleCopyWarning = m.singleCopyWarning() // F6: honest single-drive signal
|
||||
status.InterruptedRunAt, status.Interrupted = m.runInterrupted, !m.runInterrupted.IsZero() // R-519
|
||||
// Deep-copy lastDBDump so callers cannot mutate shared state.
|
||||
if m.lastDBDump != nil {
|
||||
copyDump := *m.lastDBDump
|
||||
@@ -1278,10 +1293,12 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
latestTime = f.ModTime
|
||||
}
|
||||
}
|
||||
// R-519: after a restart the run's result is not in memory, so it is SYNTHESISED from the files. A run
|
||||
// the stop cut off left a fresh .sql beside last night's tars — "Success" must not be read off that.
|
||||
status.LastDBDump = &DBDumpStatus{
|
||||
LastRun: latestTime,
|
||||
Results: results,
|
||||
Success: true,
|
||||
Success: m.runInterrupted.IsZero(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1295,6 +1312,8 @@ func (m *Manager) GetFullStatus(nextDBDump time.Time) *FullBackupStatus {
|
||||
DBDumpSchedule: m.cfg.Backup.DBDumpSchedule,
|
||||
NextDBDump: nextDBDump,
|
||||
SingleCopyWarning: m.singleCopyWarning(), // F6
|
||||
InterruptedRunAt: m.runInterrupted, // R-519
|
||||
Interrupted: !m.runInterrupted.IsZero(),
|
||||
}
|
||||
if m.lastDBDump != nil {
|
||||
copyDump := *m.lastDBDump
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"os"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-519 (controller v0.296.0) — an app-data backup run cut off by a power cut or a restart is SAID on the page.
|
||||
//
|
||||
// MEASURED 2026-09-14 (BIGNIGHT F2): the power was cut during a volume dump; afterwards /backups and /backups/apps
|
||||
// said nothing of an interruption, and every app read „Utolsó: 5 perce". The controller knew only when apps had
|
||||
// been stopped (the app-stop marker); a cut during the DATABASE leg, with every app running, left no trace at all.
|
||||
//
|
||||
// Shape (the restore record's, R-550): one JSON file in DataDir, written atomically at BOTH ends of a run. At
|
||||
// startup a file still marked running is, by construction, a run nothing is running any more: it becomes the
|
||||
// INTERRUPTED notice, kept until the next app-data run that ends with every step OK. The restore points of a torn
|
||||
// run already carry the time of their OLDEST part (the data block, v0.275.0) — this adds the sentence.
|
||||
// No path set (a bare Manager in a test, a box with backup disabled) = no persistence, silently.
|
||||
//
|
||||
// Pinned by run_record_test.go (TestRunRecord_*).
|
||||
|
||||
type runRecordFile struct {
|
||||
Running bool `json:"running"`
|
||||
StartedAt time.Time `json:"started_at,omitempty"`
|
||||
Interrupted time.Time `json:"interrupted,omitempty"` // the start of the run that was cut off; zero = none
|
||||
}
|
||||
|
||||
// SetRunRecordPath wires persistence. Call before LoadRunRecord and before any backup runs.
|
||||
func (m *Manager) SetRunRecordPath(path string) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.runRecordPath = path
|
||||
}
|
||||
|
||||
// LoadRunRecord reads the record at startup. It returns the start time of a run the stop cut off — non-zero
|
||||
// exactly once per interruption (the conversion is written back), so the caller logs it once.
|
||||
func (m *Manager) LoadRunRecord() time.Time {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.runRecordPath == "" {
|
||||
return time.Time{}
|
||||
}
|
||||
rec, ok := m.readRunRecordLocked()
|
||||
if !ok {
|
||||
return time.Time{}
|
||||
}
|
||||
m.runInterrupted = rec.Interrupted
|
||||
if !rec.Running {
|
||||
return time.Time{}
|
||||
}
|
||||
rec.Running = false
|
||||
rec.Interrupted = rec.StartedAt
|
||||
m.runInterrupted = rec.StartedAt
|
||||
m.writeRunRecordLocked(rec)
|
||||
return rec.StartedAt
|
||||
}
|
||||
|
||||
// InterruptedRunAt is the start of the last app-data run that was cut off, or zero when the newest run finished.
|
||||
func (m *Manager) InterruptedRunAt() time.Time {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
return m.runInterrupted
|
||||
}
|
||||
|
||||
// markRunStarted is called at the start of an app-data run (runDBDumpsInternal).
|
||||
func (m *Manager) markRunStarted(at time.Time) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.runRecordPath == "" {
|
||||
return
|
||||
}
|
||||
m.writeRunRecordLocked(runRecordFile{Running: true, StartedAt: at, Interrupted: m.runInterrupted})
|
||||
}
|
||||
|
||||
// markRunEnded is called on EVERY return of the run. allOK clears the interrupted notice: the household's copies
|
||||
// are whole again. A run that ended with a failed step keeps it (that run has its own failure line too).
|
||||
func (m *Manager) markRunEnded(allOK bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if allOK {
|
||||
m.runInterrupted = time.Time{}
|
||||
}
|
||||
if m.runRecordPath == "" {
|
||||
return
|
||||
}
|
||||
m.writeRunRecordLocked(runRecordFile{Running: false, Interrupted: m.runInterrupted})
|
||||
}
|
||||
|
||||
func (m *Manager) readRunRecordLocked() (runRecordFile, bool) {
|
||||
var rec runRecordFile
|
||||
b, err := os.ReadFile(m.runRecordPath)
|
||||
if err != nil {
|
||||
if !os.IsNotExist(err) && m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] run record unreadable (%v) — no interrupted-run notice", err)
|
||||
}
|
||||
return rec, false
|
||||
}
|
||||
if err := json.Unmarshal(b, &rec); err != nil {
|
||||
if m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] run record is not JSON (%v) — no interrupted-run notice", err)
|
||||
}
|
||||
return rec, false
|
||||
}
|
||||
return rec, true
|
||||
}
|
||||
|
||||
func (m *Manager) writeRunRecordLocked(rec runRecordFile) {
|
||||
b, _ := json.Marshal(rec)
|
||||
if err := atomicWrite(m.runRecordPath, b, 0o600); err != nil && m.logger != nil {
|
||||
m.logger.Printf("[WARN] [backup] could not persist the run record to %s: %v", m.runRecordPath, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// R-519 (controller v0.296.0). RED-PROOFS recorded in felhom.eu/documentation/audits/hub-safety-2026-10-05/partE/:
|
||||
// 1. delete the m.markRunStarted call in runDBDumpsInternal → TestRunRecord_TheRealRunIsOnRecordWhileItRuns fails;
|
||||
// 2. synthesise LastDBDump with `Success: true` again → TestRunRecord_SynthesisedStatusIsNotOKAfterACut fails.
|
||||
|
||||
// A run the stop cut off is said — once on load, then on the page until a run ends with every step OK. A run that
|
||||
// ended with a failed step does not clear it.
|
||||
func TestRunRecord_CutRunIsSaidUntilACompleteOne(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "appdata-run.json")
|
||||
t0 := time.Date(2026, 9, 14, 19, 40, 3, 0, time.UTC)
|
||||
|
||||
before, _ := newTestManager(t, "/srv/sys")
|
||||
before.SetRunRecordPath(path)
|
||||
before.markRunStarted(t0) // the power goes here: no markRunEnded
|
||||
|
||||
after, _ := newTestManager(t, "/srv/sys")
|
||||
after.SetRunRecordPath(path)
|
||||
if got := after.LoadRunRecord(); !got.Equal(t0) {
|
||||
t.Fatalf("the cut run was not found at startup: %v", got)
|
||||
}
|
||||
if !after.InterruptedRunAt().Equal(t0) || !after.GetFullStatus(time.Time{}).Interrupted {
|
||||
t.Fatal("the page status does not carry the cut run")
|
||||
}
|
||||
|
||||
again, _ := newTestManager(t, "/srv/sys")
|
||||
again.SetRunRecordPath(path)
|
||||
if got := again.LoadRunRecord(); !got.IsZero() {
|
||||
t.Fatalf("the same cut was reported twice: %v", got)
|
||||
}
|
||||
if !again.InterruptedRunAt().Equal(t0) {
|
||||
t.Fatal("the notice must survive a second restart until a complete run")
|
||||
}
|
||||
|
||||
again.markRunStarted(t0.Add(time.Hour))
|
||||
again.markRunEnded(false) // a run with a failed step
|
||||
if !again.InterruptedRunAt().Equal(t0) {
|
||||
t.Fatal("a run with a failed step cleared the notice")
|
||||
}
|
||||
again.markRunStarted(t0.Add(2 * time.Hour))
|
||||
again.markRunEnded(true)
|
||||
if !again.InterruptedRunAt().IsZero() || again.GetFullStatus(time.Time{}).Interrupted {
|
||||
t.Fatal("a complete run did not clear the notice")
|
||||
}
|
||||
last, _ := newTestManager(t, "/srv/sys")
|
||||
last.SetRunRecordPath(path)
|
||||
if got := last.LoadRunRecord(); !got.IsZero() || !last.InterruptedRunAt().IsZero() {
|
||||
t.Fatal("the cleared notice came back after a restart")
|
||||
}
|
||||
}
|
||||
|
||||
// The production path: runDBDumpsInternal itself puts the run on record (running=true) before its first step,
|
||||
// and takes it off on an early error return.
|
||||
func TestRunRecord_TheRealRunIsOnRecordWhileItRuns(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "appdata-run.json")
|
||||
m, _ := newTestManager(t, t.TempDir())
|
||||
m.SetRunRecordPath(path)
|
||||
sawRunning := false
|
||||
m.discoverDBs = func(context.Context) ([]DiscoveredDB, error) {
|
||||
var rec runRecordFile
|
||||
if b, err := os.ReadFile(path); err == nil && json.Unmarshal(b, &rec) == nil && rec.Running {
|
||||
sawRunning = true
|
||||
}
|
||||
return nil, errors.New("docker is not reachable")
|
||||
}
|
||||
_ = m.runDBDumpsInternal(context.Background())
|
||||
if !sawRunning {
|
||||
t.Fatal("the run was not on record while it ran — a cut here would go unnoticed")
|
||||
}
|
||||
var rec runRecordFile
|
||||
b, _ := os.ReadFile(path)
|
||||
if json.Unmarshal(b, &rec) != nil || rec.Running {
|
||||
t.Fatalf("an ended run is still on record as running: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
// After a restart the page's "last database backup" is synthesised from the files. After a cut it must not read OK
|
||||
// (BIGNIGHT F2: „Utolsó adatbázis mentés … OK" over a torn run).
|
||||
func TestRunRecord_SynthesisedStatusIsNotOKAfterACut(t *testing.T) {
|
||||
m, _ := newTestManager(t, "/srv/sys")
|
||||
m.cachedStatus = &FullBackupStatus{DumpFiles: []DumpFileInfo{{StackName: "adventurelog", FileName: "adventurelog-postgres.sql", ModTime: time.Now()}}}
|
||||
if st := m.GetFullStatus(time.Time{}); st.LastDBDump == nil || !st.LastDBDump.Success {
|
||||
t.Fatal("control: with no cut the synthesised status reads OK")
|
||||
}
|
||||
m.runInterrupted = time.Now().Add(-time.Minute)
|
||||
if st := m.GetFullStatus(time.Time{}); st.LastDBDump == nil || st.LastDBDump.Success || !st.Interrupted {
|
||||
t.Fatalf("after a cut the synthesised status still reads OK: %+v", st.LastDBDump)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user