v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.
The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.
Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.
Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.
Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -579,7 +579,11 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
// R-379/R-380: an app held after a failed restore + failed rollback must not start from the
|
||||
// customer's button either. Checked BEFORE the drive gate because it applies to driveless apps,
|
||||
// which is the class the hold exists for.
|
||||
if action == "start" || action == "restart" {
|
||||
//
|
||||
// R-439 (slice 4): `update` is in this list. It was not until v0.237.0, so a held app could be
|
||||
// updated — the one action most likely to make a held app's data worse. Pinned by
|
||||
// TestR439_UpdateOfAHeldAppIsRefused.
|
||||
if action == "start" || action == "restart" || action == "update" {
|
||||
if held, why := r.restoreHoldFor(name); held {
|
||||
writeJSON(w, http.StatusConflict, apiResponse{OK: false, Error: why})
|
||||
return
|
||||
@@ -618,6 +622,20 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
}
|
||||
}
|
||||
|
||||
// Slice 4: every cheap refusal of an update — busy, already updating, deploying, memory, disk, and
|
||||
// "no backup to return to" — BEFORE the intent below is recorded, so an update that was never going
|
||||
// to happen records nothing (§8.2). Each is a 409 with the Hungarian sentence.
|
||||
if action == "update" {
|
||||
if ref := r.stackMgr.UpdatePreflight(name); ref != nil {
|
||||
status := http.StatusConflict
|
||||
if ref.Reason == "not_found" {
|
||||
status = http.StatusNotFound
|
||||
}
|
||||
writeJSON(w, status, apiResponse{OK: false, Error: ref.Message})
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// R-166: THE CUSTOMER-INTENT POINT. This switch is where a human's decision about whether their
|
||||
// app should be running enters the system, and until v0.189.0 that decision was recorded nowhere
|
||||
// — so the box had to infer it from container counts, and inferred wrong for a power cut and for
|
||||
@@ -647,12 +665,18 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
case "restart":
|
||||
err = r.stackMgr.RestartStack(name)
|
||||
case "update":
|
||||
err = r.stackMgr.UpdateStack(name)
|
||||
// Slice 4: the GUARDED update. It returns as soon as the job has started; the result is only
|
||||
// ever known from GET /api/stacks/{name} (updating / update_phase / update_error).
|
||||
err = r.stackMgr.StartGuardedUpdate(name)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
r.logger.Printf("[ERROR] [api] %s failed for %s: %v", action, name, err)
|
||||
status := http.StatusInternalServerError
|
||||
var ref *stacks.UpdateRefusal
|
||||
if errors.As(err, &ref) {
|
||||
status = http.StatusConflict
|
||||
}
|
||||
if strings.Contains(err.Error(), "protected") {
|
||||
status = http.StatusForbidden
|
||||
}
|
||||
@@ -663,6 +687,15 @@ func (r *Router) actionStack(w http.ResponseWriter, action, name string) {
|
||||
return
|
||||
}
|
||||
|
||||
// R-443 (slice 4): an update is NEVER reported completed here. The spike measured this line saying
|
||||
// "update completed" over an app that was already crash-looping. It answers 202 — accepted, not
|
||||
// finished — and "completed" exists only as update_phase=done on GET /api/stacks/{name}, which is
|
||||
// written after the app's health is known. Pinned by TestR443_UpdateIsNeverReportedCompleteSynchronously.
|
||||
if action == "update" {
|
||||
writeJSON(w, http.StatusAccepted, apiResponse{OK: true, Message: "Frissítés elindult – az állapot a kártyán követhető",
|
||||
Data: map[string]interface{}{"accepted": true, "completed": false}})
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, apiResponse{OK: true, Message: "Stack " + name + " " + action + " completed"})
|
||||
|
||||
// Trigger integration lifecycle hooks after successful action
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/backup"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||||
)
|
||||
|
||||
// Update arc slice 4 through the PRODUCTION handler: actionStack → the real stacks.Manager
|
||||
// (NewManager + ScanStacks) and the real backup.Manager over real settings. No docker is reached:
|
||||
// every path here refuses, or fails at the pin (the app's catalog template is absent on purpose).
|
||||
|
||||
type apiFakeGuards struct {
|
||||
b *backup.Manager
|
||||
rp stacks.UpdateRestorePoint
|
||||
// blindToHolds makes the manager-side preflight NOT see holds, so a test can prove the ROUTER's
|
||||
// own hold check refuses — the two layers are each pinned separately (the preflight's by
|
||||
// TestSlice4_D_CheapRefusals/held). Without it, removing either layer passes inertly, because the
|
||||
// other refuses with the same sentence (observed on the first run of red-proof 4, 2026-09-13).
|
||||
blindToHolds bool
|
||||
}
|
||||
|
||||
func (g *apiFakeGuards) HoldFor(n string) (bool, string) {
|
||||
if g.blindToHolds {
|
||||
return false, ""
|
||||
}
|
||||
return g.b.RestoreHoldFor(n)
|
||||
}
|
||||
func (g *apiFakeGuards) Busy(string) (bool, string) { return false, "" }
|
||||
func (g *apiFakeGuards) RestorePoint(string) (stacks.UpdateRestorePoint, error) {
|
||||
return g.rp, nil
|
||||
}
|
||||
func (g *apiFakeGuards) BackupNow(context.Context, string) error { return nil }
|
||||
func (g *apiFakeGuards) SafetyDump(context.Context, string) ([]string, error) {
|
||||
return nil, nil
|
||||
}
|
||||
func (g *apiFakeGuards) HoldAfterFailedUpdate(string, time.Time, time.Time) error { return nil }
|
||||
|
||||
const slice4AppYAML = "deployed: true\nenv: {}\npinned_images:\n app: nginx:1.27\n"
|
||||
|
||||
func newSlice4Router(t *testing.T) (*Router, *settings.Settings, *apiFakeGuards, string) {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
dir := filepath.Join(root, "stacks", "app")
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "docker-compose.yml"), []byte("services:\n app:\n image: nginx:1.27\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "app.yaml"), []byte(slice4AppYAML), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.Paths.StacksDir = filepath.Join(root, "stacks")
|
||||
cfg.Paths.DataDir = filepath.Join(root, "data")
|
||||
cfg.Paths.SystemDataPath = filepath.Join(root, "sys")
|
||||
cfg.Stacks.ComposeCommand = "docker compose"
|
||||
lg := log.New(io.Discard, "", 0)
|
||||
m, err := stacks.NewManager(cfg, lg)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.ScanStacks(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sett, err := settings.Load(filepath.Join(root, "settings.json"), lg)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
b := backup.NewManager(cfg, sett, lg)
|
||||
g := &apiFakeGuards{b: b, rp: stacks.UpdateRestorePoint{Restorable: true, Proven: true, ProvenAt: time.Now().Add(-time.Hour)}}
|
||||
m.SetUpdateGuards(g)
|
||||
return &Router{cfg: cfg, stackMgr: m, backupMgr: b, logger: lg}, sett, g, dir
|
||||
}
|
||||
|
||||
func postUpdate(t *testing.T, r *Router) (int, apiResponse) {
|
||||
t.Helper()
|
||||
w := httptest.NewRecorder()
|
||||
r.actionStack(w, "update", "app")
|
||||
var resp apiResponse
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
|
||||
t.Fatalf("non-JSON body %q: %v", w.Body.String(), err)
|
||||
}
|
||||
return w.Code, resp
|
||||
}
|
||||
|
||||
// TestR439_UpdateOfAHeldAppIsRefused — R-439 closed.
|
||||
//
|
||||
// COMPANION RED-PROOF 4 (REPORT.md): remove `|| action == "update"` from actionStack's hold check. The
|
||||
// preflight then refuses on its own grounds with a DIFFERENT sentence, and this test fails on the
|
||||
// message — which is what proves the router line is the one doing it.
|
||||
func TestR439_UpdateOfAHeldAppIsRefused(t *testing.T) {
|
||||
r, sett, g, dir := newSlice4Router(t)
|
||||
g.rp = stacks.UpdateRestorePoint{} // the preflight's own refusal would say "no backup" — not the hold
|
||||
g.blindToHolds = true // only the router's line can produce the hold's sentence
|
||||
if err := sett.SetRestoreHold(settings.RestoreHold{Stack: "app", At: "2026-09-13T08:00:00Z", Reason: settings.HoldReasonUpdateFailed, CopyDate: "2026-09-13T01:30:00Z"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before, _ := os.ReadFile(filepath.Join(dir, "app.yaml"))
|
||||
code, resp := postUpdate(t, r)
|
||||
_, holdText := r.backupMgr.RestoreHoldFor("app")
|
||||
if code != http.StatusConflict || resp.OK || resp.Error != holdText {
|
||||
t.Fatalf("a HELD app's update must be refused with the hold's own sentence: code=%d ok=%v error=%q", code, resp.OK, resp.Error)
|
||||
}
|
||||
after, _ := os.ReadFile(filepath.Join(dir, "app.yaml"))
|
||||
if string(before) != string(after) {
|
||||
t.Error("a refused update must record no intent — app.yaml changed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlice4_Router_NoBackupIs409AndRecordsNothing(t *testing.T) {
|
||||
r, _, g, dir := newSlice4Router(t)
|
||||
g.rp = stacks.UpdateRestorePoint{Restorable: false}
|
||||
before, _ := os.ReadFile(filepath.Join(dir, "app.yaml"))
|
||||
code, resp := postUpdate(t, r)
|
||||
if code != http.StatusConflict || resp.Error != fmt.Sprintf(stacks.MsgUpdateNoBackupFmt, "app") {
|
||||
t.Fatalf("code=%d error=%q", code, resp.Error)
|
||||
}
|
||||
if after, _ := os.ReadFile(filepath.Join(dir, "app.yaml")); string(before) != string(after) {
|
||||
t.Error("the preflight refusal must come BEFORE the intent write")
|
||||
}
|
||||
}
|
||||
|
||||
// TestR443_UpdateIsNeverReportedCompleteSynchronously — R-443 closed. The handler answers 202 with
|
||||
// completed:false; the job then runs (and here fails at the pin, the catalog being absent), and the
|
||||
// outcome exists ONLY on GET /api/stacks/{name}.
|
||||
func TestR443_UpdateIsNeverReportedCompleteSynchronously(t *testing.T) {
|
||||
r, _, _, _ := newSlice4Router(t)
|
||||
code, resp := postUpdate(t, r)
|
||||
if code != http.StatusAccepted {
|
||||
t.Fatalf("an accepted update must answer 202, got %d (%+v)", code, resp)
|
||||
}
|
||||
data, _ := resp.Data.(map[string]interface{})
|
||||
if data["completed"] != false || data["accepted"] != true {
|
||||
t.Errorf("the body must say accepted and NOT completed, got %v", resp.Data)
|
||||
}
|
||||
if resp.Message == "Stack app update completed" {
|
||||
t.Error("the synchronous response claimed completion — R-443")
|
||||
}
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if st, ok := r.stackMgr.GetStack("app"); ok && !st.Updating {
|
||||
if st.UpdatePhase != stacks.UpdatePhaseFailed || st.UpdateError != stacks.MsgUpdatePinFailed {
|
||||
t.Errorf("the job's truth must be on the stack: phase=%q err=%q", st.UpdatePhase, st.UpdateError)
|
||||
}
|
||||
return
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("the job never finished")
|
||||
}
|
||||
Reference in New Issue
Block a user