v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.
The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.
Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.
Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.
Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -680,6 +680,15 @@ func (m *Manager) runVolumeDumps() (summary []string, dumped int, allOK bool) {
|
||||
if m.cfg != nil && m.cfg.IsProtectedStack(stack.Name) {
|
||||
continue
|
||||
}
|
||||
// Slice 4 / R-379: a HELD app is deliberately stopped, and DumpAppVolumesSafe ends in
|
||||
// StartStack — so without this line the nightly backup restarted every held app, every night.
|
||||
// "A hold that only one path honours is not a hold." Checked BEFORE the volume check so a held
|
||||
// app is never stopped or started by this leg at all.
|
||||
if m.isHeld(stack.Name) {
|
||||
m.logger.Printf("[WARN] [backup] Skipping volume dump for %s — the app is HELD stopped (its restore point is preserved)", stack.Name)
|
||||
summary = append(summary, fmt.Sprintf("SKIP %s volumes (held)", stack.Name))
|
||||
continue
|
||||
}
|
||||
// Volume check FIRST — a volume-less stack must not be stopped at all (see gate-order note).
|
||||
if len(m.stackProvider.GetDockerVolumes(stack.Name)) == 0 {
|
||||
if m.isDebug() {
|
||||
|
||||
@@ -337,6 +337,15 @@ func (m *Manager) RestoreHoldFor(stack string) (bool, string) {
|
||||
if !ok {
|
||||
return false, ""
|
||||
}
|
||||
// Slice 4: one storage, two reasons. An update hold names the copy it can be restored from; a
|
||||
// restore hold names nothing, because the restore it refers to already consumed the copy.
|
||||
if h.Reason == settings.HoldReasonUpdateFailed {
|
||||
copyDate := "legutóbbi"
|
||||
if h.CopyDate != "" {
|
||||
copyDate = fmtHoldTime(h.CopyDate)
|
||||
}
|
||||
return true, fmt.Sprintf(UpdateHoldFmt, stack, fmtHoldTime(h.At), copyDate)
|
||||
}
|
||||
when := h.At
|
||||
if t, err := time.Parse(time.RFC3339, h.At); err == nil {
|
||||
when = t.Format("2006-01-02 15:04")
|
||||
|
||||
@@ -385,6 +385,14 @@ func (m *Manager) captureAllRecoveryUnits() {
|
||||
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
|
||||
continue // drive not writable — skip, the existing unit stays as-is
|
||||
}
|
||||
// Slice 4: a HELD app's unit is its RESTORE POINT, and the hold text names that copy's date.
|
||||
// Re-capturing it would write the definition the app is held ON (after a failed update: the new
|
||||
// version that would not start) into the unit, and the next Tier-2 run would mirror it over the
|
||||
// copy the customer was told to restore from. The unit stays exactly as it was until the hold is
|
||||
// lifted.
|
||||
if m.isHeld(stack.Name) {
|
||||
continue
|
||||
}
|
||||
m.noteAttempted(stack.Name)
|
||||
// The reserve, checked BEFORE anything is written. Per app, and the loop continues.
|
||||
if !m.admitApp(stack.Name) {
|
||||
|
||||
@@ -390,5 +390,8 @@ func (m *Manager) RestoreFromRecoveryUnitAt(stackName, unitDir string) (UnitRest
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] Restore-from-unit completed: %s — %d volume(s) of %d listed, %d database(s) of %d listed",
|
||||
stackName, res.VolumesReplayed, res.ManifestVolumes, res.DBsReplayed, res.ManifestDBs)
|
||||
// Slice 4: the app is back on its unit's definition and data and was started — the route back an
|
||||
// update hold names. Lift that hold (and only that kind; see clearUpdateHoldAfterRestore).
|
||||
m.clearUpdateHoldAfterRestore(stackName)
|
||||
return res, nil
|
||||
}
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// Update arc slice 4 — the backup side of the guarded update.
|
||||
|
||||
// The measured case (demo-hp 2026-09-13, bookstack): the mirror's manifest said 2026-09-12T02:15:29Z
|
||||
// while its database dump was written 2026-09-13T00:30Z and the Tier-2 run succeeded at 01:30Z. The
|
||||
// update's age must be the proven COPY time; the manifest date would call a fresh copy stale forever.
|
||||
func TestSlice4_ProvenCopyTime_IsTheLastSuccessNotTheManifestDate(t *testing.T) {
|
||||
rp := restorePointFromCoverage(Tier2Coverage{
|
||||
UnitRestorable: true, UnitPackageDate: "2026-09-12T02:15:29Z",
|
||||
CopyLastRun: "2026-09-13T01:30:00Z", CopyLastSuccess: "2026-09-13T01:30:00Z",
|
||||
})
|
||||
at, ok := rp.ProvenCopyTime()
|
||||
if !ok || !at.Equal(time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)) {
|
||||
t.Fatalf("proven copy time = %v ok=%v, want the last successful copy", at, ok)
|
||||
}
|
||||
// The page still names the PACKAGE date (R-403) — the extraction changes nothing it shows.
|
||||
if rp.CopyDate != "2026-09-12T02:15:29Z" || !rp.CopyDateProven || rp.PackagePreserved {
|
||||
t.Errorf("restore point = %+v", rp)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_ProvenCopyTime_PreservedPackageUsesThePackageDate(t *testing.T) {
|
||||
rp := restorePointFromCoverage(Tier2Coverage{
|
||||
UnitRestorable: true, UnitPackageDate: "2026-09-01T02:00:00Z", UnitLegPreserved: true,
|
||||
CopyLastSuccess: "2026-09-13T01:30:00Z",
|
||||
})
|
||||
if at, ok := rp.ProvenCopyTime(); !ok || !at.Equal(time.Date(2026, 9, 1, 2, 0, 0, 0, time.UTC)) {
|
||||
t.Errorf("a PRESERVED package is as old as the package, got %v ok=%v", at, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_ProvenCopyTime_NoProvenOrNoUnitIsNoRestorePoint(t *testing.T) {
|
||||
for _, cov := range []Tier2Coverage{
|
||||
{UnitRestorable: true, CopyLastRun: "2026-09-13T01:30:00Z"}, // attempt, never a success (R-101)
|
||||
{UnitRestorable: false, CopyLastSuccess: "2026-09-13T01:30:00Z"}, // a copy with no openable unit
|
||||
{UnitRestorable: true, CopyLastSuccess: "not-a-date"}, // unparseable is unknown, never "now"
|
||||
} {
|
||||
if _, ok := restorePointFromCoverage(cov).ProvenCopyTime(); ok {
|
||||
t.Errorf("%+v must not yield a proven copy time", cov)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func slice4Settings(t *testing.T) *settings.Settings {
|
||||
t.Helper()
|
||||
s, err := settings.Load(filepath.Join(t.TempDir(), "settings.json"), log.New(io.Discard, "", 0))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func TestSlice4_UpdateHoldTextNamesTheTimeAndTheCopy(t *testing.T) {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
at := time.Date(2026, 9, 13, 8, 0, 0, 0, time.UTC)
|
||||
copyAt := time.Date(2026, 9, 13, 1, 30, 0, 0, time.UTC)
|
||||
if err := m.HoldAfterFailedUpdate("bookstack", at, copyAt); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h, ok := sett.GetRestoreHold("bookstack")
|
||||
if !ok || h.Reason != settings.HoldReasonUpdateFailed || h.CopyDate != "2026-09-13T01:30:00Z" {
|
||||
t.Fatalf("hold = %+v ok=%v", h, ok)
|
||||
}
|
||||
held, why := m.RestoreHoldFor("bookstack")
|
||||
// Budapest is UTC+2 in September: 08:00Z → 10:00, 01:30Z → 03:30.
|
||||
want := fmt.Sprintf(UpdateHoldFmt, "bookstack", "2026-09-13 10:00", "2026-09-13 03:30")
|
||||
if !held || why != want {
|
||||
t.Errorf("hold text =\n%q\nwant\n%q", why, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_RestoreHoldTextIsUnchanged(t *testing.T) {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "docmost", At: "2026-08-22T14:00:00Z"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, why := m.RestoreHoldFor("docmost")
|
||||
if !strings.Contains(why, "visszaállítása") || !strings.Contains(why, "Vedd fel velünk a kapcsolatot") || strings.Contains(why, "frissítése") {
|
||||
t.Errorf("an R-379 restore hold must keep its own sentence, got %q", why)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_ASuccessfulRestoreClearsOnlyAnUpdateHold(t *testing.T) {
|
||||
sett := slice4Settings(t)
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0), settings: sett}
|
||||
_ = m.HoldAfterFailedUpdate("upd", time.Now(), time.Now())
|
||||
_ = sett.SetRestoreHold(settings.RestoreHold{Stack: "rst", At: "2026-08-22T14:00:00Z"})
|
||||
m.clearUpdateHoldAfterRestore("upd")
|
||||
m.clearUpdateHoldAfterRestore("rst")
|
||||
if _, ok := sett.GetRestoreHold("upd"); ok {
|
||||
t.Error("a restore is the route back from a failed update — its hold must be lifted")
|
||||
}
|
||||
if _, ok := sett.GetRestoreHold("rst"); !ok {
|
||||
t.Error("an R-379 restore hold stays operator-cleared")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_UpdateBusy(t *testing.T) {
|
||||
m := &Manager{logger: log.New(io.Discard, "", 0)}
|
||||
if busy, _ := m.UpdateBusy("app"); busy {
|
||||
t.Fatal("an idle manager is not busy")
|
||||
}
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if busy, _ := m.UpdateBusy("app"); !busy {
|
||||
t.Error("a running backup/restore must make an update wait")
|
||||
}
|
||||
m.releaseRunning()
|
||||
m.BeginRestoreOp("tier2-unit-restore", "other")
|
||||
if busy, _ := m.UpdateBusy("app"); !busy {
|
||||
t.Error("a restore op in flight must make an update wait")
|
||||
}
|
||||
}
|
||||
|
||||
// "A hold that only one path honours is not a hold." The nightly legs are unattended start paths
|
||||
// (DumpAppVolumesSafe ends in StartStack) and writers of the restore point the hold text names.
|
||||
//
|
||||
// COMPANION RED-PROOF (REPORT.md): delete the isHeld skip from runVolumeDumps — the held app is then
|
||||
// stopped (and restarted) by the nightly backup, and this test fails.
|
||||
func TestSlice4_NightlyLegsLeaveAHeldAppAlone(t *testing.T) {
|
||||
h := newAdmissionHarness(t, "held", "free")
|
||||
h.m.settings = slice4Settings(t)
|
||||
if err := h.m.HoldAfterFailedUpdate("held", time.Now(), time.Now()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h.m.runVolumeDumps()
|
||||
for _, n := range append(append([]string{}, h.volDumped...), h.prov.stopped...) {
|
||||
if n == "held" {
|
||||
t.Fatalf("the nightly volume dump touched a HELD app (dumped=%v stopped=%v)", h.volDumped, h.prov.stopped)
|
||||
}
|
||||
}
|
||||
if len(h.volDumped) != 1 || h.volDumped[0] != "free" {
|
||||
t.Errorf("positive control: the unheld app must still be dumped, got %v", h.volDumped)
|
||||
}
|
||||
h.m.captureAllRecoveryUnits()
|
||||
for _, n := range h.prov.infoHits {
|
||||
if n == "held" {
|
||||
t.Error("the capture must not rewrite a HELD app's restore point")
|
||||
}
|
||||
}
|
||||
var mirrored []string
|
||||
h.m.perAppTier2 = func(name string) error { mirrored = append(mirrored, name); return nil }
|
||||
h.m.RunAllTier2()
|
||||
for _, n := range mirrored {
|
||||
if n == "held" {
|
||||
t.Error("Tier 2 must not mirror over a HELD app's copy")
|
||||
}
|
||||
}
|
||||
if len(mirrored) != 1 {
|
||||
t.Errorf("positive control: the unheld app must still be mirrored, got %v", mirrored)
|
||||
}
|
||||
}
|
||||
@@ -457,6 +457,12 @@ func (m *Manager) RunAllTier2() {
|
||||
m.settings.IsDecommissioned(m.GetAppDrivePath(stack.Name))) {
|
||||
continue
|
||||
}
|
||||
// Slice 4: never mirror over a HELD app's copy — it is the restore point the hold text names.
|
||||
// See the matching skip in captureAllRecoveryUnits.
|
||||
if m.isHeld(stack.Name) {
|
||||
m.logger.Printf("[WARN] [backup] Tier 2 skipped for %s — the app is HELD; its copy is the restore point and is preserved", stack.Name)
|
||||
continue
|
||||
}
|
||||
runOne := m.perAppTier2
|
||||
if runOne == nil {
|
||||
runOne = m.RunTier2
|
||||
|
||||
@@ -0,0 +1,325 @@
|
||||
package backup
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
)
|
||||
|
||||
// ── The backup side of the guarded update (update arc slice 4, controller v0.237.0) ─────────────
|
||||
//
|
||||
// 09-update-architecture.md §3 decision 1 (operator ruling 2026-09-02): the safety copy for an update
|
||||
// is a VERIFIED RECENT BACKUP as a PRECONDITION — not a new copy mechanism invented for the update
|
||||
// path. So everything in this file composes machinery that already exists and is proven live:
|
||||
// the Tier-2 unit restore's own predicate (R-102/R-103), the nightly legs (DB dump, volume dump,
|
||||
// unit capture, Tier-2 mirror), the pre-restore safety dump (R-361) and the R-379 hold.
|
||||
//
|
||||
// The stacks package cannot import this one, so the update job reaches all of it through the
|
||||
// stacks.UpdateGuards interface, implemented by an adapter in cmd/controller/main.go.
|
||||
|
||||
// Tier2RestorePoint is the answer to "could this app be restored from its Tier-2 copy, and from
|
||||
// when?" — the predicate the destructive „Teljes visszaállítás" action is gated on.
|
||||
//
|
||||
// EXTRACTED, NOT DUPLICATED (slice 4). Until v0.237.0 this computation lived inline in the backups
|
||||
// page handler (buildAppBackupRows). The update path needs exactly the same question answered, and
|
||||
// a second copy of a predicate is how this project's two copies of `namespaceRoot` came to differ
|
||||
// (R-203). So there is one function, and the page and the update both call it.
|
||||
type Tier2RestorePoint struct {
|
||||
// Restorable — the copy holds an OPENABLE recovery unit (Tier2Coverage.CanRestoreUnit).
|
||||
Restorable bool
|
||||
// CopyDate — the date the unit restore NAMES: the package's own manifest date, falling back to the
|
||||
// copy date (Tier2Coverage.UnitRestoreDate, R-403). RFC3339 as recorded, "" when unknown.
|
||||
CopyDate string
|
||||
// CopyDateProven — a copy actually SUCCEEDED (LastSuccess is set), never merely an attempt (R-101).
|
||||
CopyDateProven bool
|
||||
// PackagePreserved — the newest run PRESERVED an older package instead of refreshing it (R-403).
|
||||
PackagePreserved bool
|
||||
// CopyLastSuccess — the RFC3339 time of the last Tier-2 copy that succeeded.
|
||||
CopyLastSuccess string
|
||||
}
|
||||
|
||||
// restorePointFromCoverage is the pure half of the predicate.
|
||||
func restorePointFromCoverage(cov Tier2Coverage) Tier2RestorePoint {
|
||||
pkgDate, preserved := cov.UnitRestoreDate()
|
||||
return Tier2RestorePoint{
|
||||
Restorable: cov.CanRestoreUnit(),
|
||||
CopyDate: pkgDate,
|
||||
CopyDateProven: cov.CopyLastSuccess != "",
|
||||
PackagePreserved: preserved,
|
||||
CopyLastSuccess: cov.CopyLastSuccess,
|
||||
}
|
||||
}
|
||||
|
||||
// Tier2UnitRestorePoint resolves the app's recorded Tier-2 copy and returns the restore point. The
|
||||
// error is the same refusal Tier2RestoreCoverage raises (no copy, drive gone, pre-v2 layout).
|
||||
func (m *Manager) Tier2UnitRestorePoint(stackName string) (Tier2RestorePoint, error) {
|
||||
cov, err := m.Tier2RestoreCoverage(stackName)
|
||||
if err != nil {
|
||||
return Tier2RestorePoint{}, err
|
||||
}
|
||||
return restorePointFromCoverage(cov), nil
|
||||
}
|
||||
|
||||
// ProvenCopyTime returns WHEN the data this copy would restore was last proven copied, and false when
|
||||
// there is no proven, restorable copy at all.
|
||||
//
|
||||
// WHY NOT CopyDate, measured rather than assumed. CopyDate is the unit MANIFEST's created_at, and a
|
||||
// capture rewrites the manifest only when the app's DEFINITION changes (compose, app.yaml, controller
|
||||
// version) — a nightly DB dump keeps the same file name, so it does not move it. Measured on demo-hp
|
||||
// 2026-09-13: bookstack's Tier-2 mirror held `bookstack-mariadb.sql` written 2026-09-13T00:30Z while
|
||||
// its manifest still read 2026-09-12T02:15:29Z. Judging "recent" by that date would call a fresh copy
|
||||
// stale — and, worse, a "back up first" run would not move it either on a quiet app, so the update
|
||||
// would be refused forever.
|
||||
//
|
||||
// So the age is the last SUCCESSFUL copy (LastSuccess), which the Tier-2 run records only when it
|
||||
// actually mirrored the unit — EXCEPT when the run preserved an older package (R-403), in which case
|
||||
// the package date is the honest one, because that is what the copy really holds.
|
||||
func (p Tier2RestorePoint) ProvenCopyTime() (time.Time, bool) {
|
||||
if !p.Restorable || !p.CopyDateProven {
|
||||
return time.Time{}, false
|
||||
}
|
||||
src := p.CopyLastSuccess
|
||||
if p.PackagePreserved {
|
||||
src = p.CopyDate
|
||||
}
|
||||
t, err := time.Parse(time.RFC3339, src)
|
||||
if err != nil {
|
||||
return time.Time{}, false
|
||||
}
|
||||
return t, true
|
||||
}
|
||||
|
||||
// UpdateBusy reports whether something else is ALREADY touching this app's data, which refuses an
|
||||
// update before anything moves (slice 4 Scenario D). The reason is operator-English; the customer
|
||||
// sentence is chosen by the caller.
|
||||
//
|
||||
// IsRunning is box-wide, deliberately: the backup/restore single-flight is box-wide, and an update's
|
||||
// "back up first" leg needs that same flag — an update started beside a running backup would either
|
||||
// wait on it invisibly or fail half-way.
|
||||
func (m *Manager) UpdateBusy(stackName string) (bool, string) {
|
||||
if m == nil {
|
||||
return false, ""
|
||||
}
|
||||
if m.IsRunning() {
|
||||
return true, "a backup or restore is running (single-flight held)"
|
||||
}
|
||||
if st := m.RestoreStatus(); st.Running {
|
||||
return true, fmt.Sprintf("restore op %q is running for %q", st.Op, st.Stack)
|
||||
}
|
||||
for _, held := range m.appStop.HeldStacks() {
|
||||
if held == stackName {
|
||||
return true, "an app-data operation (volume dump / export / reconstitute) is holding it"
|
||||
}
|
||||
}
|
||||
return false, ""
|
||||
}
|
||||
|
||||
// ErrUpdateBackupNoUnit is returned when a "back up first" run completed but the app still has no
|
||||
// openable Tier-2 unit — typically Tier 2 is switched off for the app or has no second target.
|
||||
var ErrUpdateBackupNoUnit = errors.New("a frissítés előtti mentés lefutott, de nem jött létre visszaállítható másolat")
|
||||
|
||||
// RunAppBackupNow runs THIS app's backup legs now, in the nightly order, and then its Tier-2 copy:
|
||||
// database dump(s) → volume dump (if the app has named volumes) → recovery-unit capture → Tier-2
|
||||
// mirror. It is the "back up first" of slice 4 Scenario B.
|
||||
//
|
||||
// Composed, not reinvented: every leg is the one runDBDumpsInternal and RunAllTier2 already run,
|
||||
// including the R-181 reserve (admitApp) before the first write and the R-166 app-stop marker inside
|
||||
// DumpAppVolumesSafe. What differs is only the scope — one app instead of all of them — because an
|
||||
// update must not bounce every other app on the box to back up one.
|
||||
func (m *Manager) RunAppBackupNow(ctx context.Context, stackName string) error {
|
||||
if m.stackProvider == nil {
|
||||
return fmt.Errorf("stack provider not configured")
|
||||
}
|
||||
if m.migrationActive() {
|
||||
return fmt.Errorf("adatáthelyezés folyamatban — a mentés most nem indítható")
|
||||
}
|
||||
if err := m.acquireRunning(); err != nil {
|
||||
return err
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: starting (DB dump → volume dump → unit capture → Tier 2)", stackName)
|
||||
start := time.Now()
|
||||
legErr := func() error {
|
||||
defer m.releaseRunning()
|
||||
defer m.beginAdmissionRun()()
|
||||
|
||||
drivePath := m.GetAppDrivePath(stackName)
|
||||
if drivePath == "" || !filepath.IsAbs(drivePath) {
|
||||
return fmt.Errorf("az alkalmazás meghajtója nem határozható meg")
|
||||
}
|
||||
if m.settings != nil && (m.settings.IsDisconnected(drivePath) || m.settings.IsDecommissioned(drivePath)) {
|
||||
return fmt.Errorf("az alkalmazás meghajtója nem elérhető (%s)", drivePath)
|
||||
}
|
||||
if !m.admitApp(stackName) {
|
||||
return fmt.Errorf("nincs elég szabad hely a mentéshez a(z) %s meghajtón", drivePath)
|
||||
}
|
||||
nsRoot := m.namespaceRoot(drivePath)
|
||||
|
||||
discover := m.discoverDBs
|
||||
if discover == nil {
|
||||
discover = func(ctx context.Context) ([]DiscoveredDB, error) {
|
||||
return DiscoverDatabases(ctx, m.logger, m.isDebug(), m.knownStackNames())
|
||||
}
|
||||
}
|
||||
dbs, err := discover(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("adatbázis-felderítés sikertelen: %w", err)
|
||||
}
|
||||
dumped := 0
|
||||
for _, db := range dbs {
|
||||
if db.StackName != stackName {
|
||||
continue
|
||||
}
|
||||
res := DumpOne(ctx, db, AppDBDumpPath(nsRoot, stackName), m.logger, m.isDebug())
|
||||
if res.Error != nil {
|
||||
return fmt.Errorf("adatbázis-mentés sikertelen (%s): %w", db.ContainerName, res.Error)
|
||||
}
|
||||
dumped++
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: database dump OK (%s, %s)", stackName, db.ContainerName, humanizeBytes(res.Size))
|
||||
}
|
||||
|
||||
if len(m.stackProvider.GetDockerVolumes(stackName)) > 0 {
|
||||
dump := m.dumpVolumesSafe
|
||||
if dump == nil {
|
||||
dump = m.DumpAppVolumesSafe
|
||||
}
|
||||
if err := dump(stackName); err != nil {
|
||||
return fmt.Errorf("kötetmentés sikertelen: %w", err)
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: volume dump OK", stackName)
|
||||
}
|
||||
|
||||
if err := m.CaptureRecoveryUnit(stackName); err != nil {
|
||||
return fmt.Errorf("a mentési egység rögzítése sikertelen: %w", err)
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: recovery unit captured (%d database dump(s))", stackName, dumped)
|
||||
return nil
|
||||
}()
|
||||
if legErr != nil {
|
||||
m.logger.Printf("[ERROR] [backup] update pre-backup for %s FAILED after %s: %v", stackName, time.Since(start).Round(time.Millisecond), legErr)
|
||||
return legErr
|
||||
}
|
||||
|
||||
runOne := m.perAppTier2
|
||||
if runOne == nil {
|
||||
runOne = m.RunTier2
|
||||
}
|
||||
if err := runOne(stackName); err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] update pre-backup for %s: Tier 2 copy FAILED: %v", stackName, err)
|
||||
return fmt.Errorf("a másodlagos másolat elkészítése sikertelen: %w", err)
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] update pre-backup for %s: complete in %s", stackName, time.Since(start).Round(time.Millisecond))
|
||||
return nil
|
||||
}
|
||||
|
||||
// WriteUpdateSafetyDump takes the last-minute database copy an update makes just before it moves the
|
||||
// pin: "the state the customer was in a minute ago". It is writeSafetyDump (R-361) unchanged — the
|
||||
// same `pre-restore-` undo naming, the same pruning to three, the same never-the-canonical-name rule
|
||||
// — so it is also picked up by the same exclusions (it never enters a manifest's db_dumps).
|
||||
//
|
||||
// Returns the paths written; an app with no database returns (nil, nil), which is a no-op and never a
|
||||
// failure (measured in writeSafetyDump: `len(mine) == 0` returns an empty set).
|
||||
func (m *Manager) WriteUpdateSafetyDump(ctx context.Context, stackName string) ([]string, error) {
|
||||
nsRoot := m.AppNamespaceRoot(stackName)
|
||||
if nsRoot == "" {
|
||||
return nil, fmt.Errorf("az alkalmazás mentési helye nem határozható meg")
|
||||
}
|
||||
set, err := m.writeSafetyDump(ctx, stackName, nsRoot)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var paths []string
|
||||
for _, f := range set.Files {
|
||||
paths = append(paths, f.Path)
|
||||
}
|
||||
if len(paths) == 0 {
|
||||
m.logger.Printf("[INFO] [backup] update safety dump for %s: the app has no database — nothing to copy (no-op)", stackName)
|
||||
} else {
|
||||
m.logger.Printf("[INFO] [backup] update safety dump for %s: %d file(s) %v", stackName, len(paths), paths)
|
||||
}
|
||||
return paths, nil
|
||||
}
|
||||
|
||||
// UpdateHoldFmt is the customer sentence for an app held after a failed update. Arguments: the app,
|
||||
// the time of the failure, and the PROVEN date of the copy it can be restored from. One named string so
|
||||
// a test asserts it verbatim instead of retyping Hungarian (R-364).
|
||||
const UpdateHoldFmt = "A(z) %s frissítése %s-kor nem sikerült, és az alkalmazás nem indult el az új verzióval. " +
|
||||
"Az alkalmazás biztonsági okból leállítva marad, hogy az adatai ne sérüljenek. " +
|
||||
"Visszaállítható a(z) %s-i biztonsági mentésből a Mentések oldalon."
|
||||
|
||||
// holdTimeZone is where the customer-facing hold sentence renders its times. The same zone the web
|
||||
// layer renders the Mentések page's copy dates in (web.getTimezone), so the date in the hold text and
|
||||
// the date on the page it points at are the same string.
|
||||
func holdTimeZone() *time.Location {
|
||||
if loc, err := time.LoadLocation("Europe/Budapest"); err == nil {
|
||||
return loc
|
||||
}
|
||||
return time.UTC
|
||||
}
|
||||
|
||||
func fmtHoldTime(rfc3339 string) string {
|
||||
t, err := time.Parse(time.RFC3339, rfc3339)
|
||||
if err != nil {
|
||||
return rfc3339
|
||||
}
|
||||
return t.In(holdTimeZone()).Format("2006-01-02 15:04")
|
||||
}
|
||||
|
||||
// HoldAfterFailedUpdate records that an app is held stopped because its new version did not come up
|
||||
// healthy. Same storage and same gate as the R-379 hold — every start path that already refuses a
|
||||
// restore hold refuses this one without being touched.
|
||||
//
|
||||
// It returns the error rather than only logging it, unlike holdAppAfterFailedRollback: the update job
|
||||
// has just stopped the app on the strength of this record, and an unrecorded hold is a stopped app
|
||||
// that the next restart button will quietly start again. The caller logs it at ERROR and keeps the
|
||||
// failure on the page.
|
||||
func (m *Manager) HoldAfterFailedUpdate(stackName string, at time.Time, copyDate time.Time) error {
|
||||
if m == nil || m.settings == nil {
|
||||
return fmt.Errorf("no settings wired — the update hold for %s cannot be persisted", stackName)
|
||||
}
|
||||
h := settings.RestoreHold{
|
||||
Stack: stackName,
|
||||
At: at.UTC().Format(time.RFC3339),
|
||||
Reason: settings.HoldReasonUpdateFailed,
|
||||
}
|
||||
if !copyDate.IsZero() {
|
||||
h.CopyDate = copyDate.UTC().Format(time.RFC3339)
|
||||
}
|
||||
if err := m.settings.SetRestoreHold(h); err != nil {
|
||||
return fmt.Errorf("persisting the update hold for %s: %w", stackName, err)
|
||||
}
|
||||
m.logger.Printf("[WARN] [backup] %s is HELD STOPPED after a failed update (restore point: %s)", stackName, h.CopyDate)
|
||||
return nil
|
||||
}
|
||||
|
||||
// isHeld reports whether an app carries ANY hold. Used by the nightly legs to leave a held app alone.
|
||||
func (m *Manager) isHeld(stackName string) bool {
|
||||
held, _ := m.RestoreHoldFor(stackName)
|
||||
return held
|
||||
}
|
||||
|
||||
// clearUpdateHoldAfterRestore lifts an UPDATE hold once a person has restored the app successfully.
|
||||
//
|
||||
// "A person clears it by restoring" — slice 4 Part 3. The restore just put the app back on the
|
||||
// definition and data of its recovery unit and started it, which is the exact route back the hold
|
||||
// text names; leaving the hold in place would refuse the next restart of an app that is now fine.
|
||||
//
|
||||
// A RESTORE hold (R-379) is deliberately NOT cleared here: that hold means a previous restore already
|
||||
// left the database in an unknown state, and it stays operator-cleared (`-clear-restore-hold`).
|
||||
func (m *Manager) clearUpdateHoldAfterRestore(stackName string) {
|
||||
if m.settings == nil {
|
||||
return
|
||||
}
|
||||
h, ok := m.settings.GetRestoreHold(stackName)
|
||||
if !ok || h.Reason != settings.HoldReasonUpdateFailed {
|
||||
return
|
||||
}
|
||||
if _, err := m.settings.ClearRestoreHold(stackName); err != nil {
|
||||
m.logger.Printf("[ERROR] [backup] %s was restored, but its update hold could not be cleared: %v", stackName, err)
|
||||
return
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] %s: restore completed — the update hold (set %s) is CLEARED", stackName, h.At)
|
||||
}
|
||||
Reference in New Issue
Block a user