v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.
The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.
Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.
Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.
Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -579,7 +579,11 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
// R-379/R-380: an app held after a failed restore + failed rollback must not start from the
|
||||
// customer's button either. Checked BEFORE the drive gate because it applies to driveless apps,
|
||||
// which is the class the hold exists for.
|
||||
if action == "start" || action == "restart" {
|
||||
//
|
||||
// R-439 (slice 4): `update` is in this list. It was not until v0.237.0, so a held app could be
|
||||
// updated — the one action most likely to make a held app's data worse. Pinned by
|
||||
// TestR439_UpdateOfAHeldAppIsRefused.
|
||||
if action == "start" || action == "restart" || action == "update" {
|
||||
if held, why := r.restoreHoldFor(name); held {
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false, Error: why})
|
||||
return
|
||||
@@ -618,6 +622,20 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
}
|
||||
}
|
||||
|
||||
// Slice 4: every cheap refusal of an update — busy, already updating, deploying, memory, disk, and
|
||||
// "no backup to return to" — BEFORE the intent below is recorded, so an update that was never going
|
||||
// to happen records nothing (§8.2). Each is a 409 with the Hungarian sentence.
|
||||
if action == "update" {
|
||||
if ref := r.stackMgr.UpdatePreflight(name); ref != nil {
|
||||
status := http.StatusConflict
|
||||
if ref.Reason == "not_found" {
|
||||
status = http.StatusNotFound
|
||||
}
|
||||
writeJSON(w, status, apiResponse{OK: false, Error: ref.Message})
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// R-166: THE CUSTOMER-INTENT POINT. This switch is where a human's decision about whether their
|
||||
// app should be running enters the system, and until v0.189.0 that decision was recorded nowhere
|
||||
// — so the box had to infer it from container counts, and inferred wrong for a power cut and for
|
||||
@@ -647,12 +665,18 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
case "restart":
|
||||
err = r.stackMgr.RestartStack(name)
|
||||
case "update":
|
||||
err = r.stackMgr.UpdateStack(name)
|
||||
// Slice 4: the GUARDED update. It returns as soon as the job has started; the result is only
|
||||
// ever known from GET /api/stacks/{name} (updating / update_phase / update_error).
|
||||
err = r.stackMgr.StartGuardedUpdate(name)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
r.logger.Printf("[ERROR] [api] %s failed for %s: %v", action, name, err)
|
||||
status := http.StatusInternalServerError
|
||||
var ref *stacks.UpdateRefusal
|
||||
if errors.As(err, &ref) {
|
||||
status = http.StatusConflict
|
||||
}
|
||||
if strings.Contains(err.Error(), "protected") {
|
||||
status = http.StatusForbidden
|
||||
}
|
||||
@@ -663,6 +687,15 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
return
|
||||
}
|
||||
|
||||
// R-443 (slice 4): an update is NEVER reported completed here. The spike measured this line saying
|
||||
// "update completed" over an app that was already crash-looping. It answers 202 — accepted, not
|
||||
// finished — and "completed" exists only as update_phase=done on GET /api/stacks/{name}, which is
|
||||
// written after the app's health is known. Pinned by TestR443_UpdateIsNeverReportedCompleteSynchronously.
|
||||
if action == "update" {
|
||||
writeJSON(w, http.StatusAccepted, apiResponse{OK: true, Message: "Frissítés elindult – az állapot a kártyán követhető",
|
||||
Data: map[string]interface{}{"accepted": true, "completed": false}})
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Stack " + name + " " + action + " completed"})
|
||||
|
||||
// Trigger integration lifecycle hooks after successful action
|
||||
|
||||
Reference in New Issue
Block a user