v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s

POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.

The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.

Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.

Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.

Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.

Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-13 11:41:31 +02:00
parent 1552716722
commit 0d402f711d
25 changed files with 2709 additions and 123 deletions
+66 -46
View File
@@ -4,6 +4,7 @@ import (
"crypto/rand"
"encoding/base64"
"encoding/hex"
"errors"
"fmt"
"log"
"math/big"
@@ -236,52 +237,12 @@ func (m *Manager) DeployStack(req DeployRequest) (string, error) {
}
// --- Memory validation ---
var deployWarning string
reservedMB := m.cfg.System.ReservedMemoryMB
totalMB, usedMB, memErr := system.GetMemoryMB()
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
// either never fire (host total) or always fire (host used > guest cap).
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
totalMB = gt
memErr = nil
}
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
usedMB = committedReqMB
}
if memErr != nil {
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
} else {
usableMB := totalMB - reservedMB
newReqMB := ParseMemoryMB(meta.Resources.MemRequest)
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
// Hard block: committed + new request exceeds usable memory
if newReqMB > 0 && usedMB+newReqMB > usableMB {
clearDeploying()
return "", fmt.Errorf(
"Nincs elég memória az alkalmazás telepítéséhez. "+
"Szükséges: %d MB, Elérhető: %d MB "+
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
newReqMB,
usableMB-usedMB,
totalMB,
usedMB,
reservedMB,
)
}
// Soft warning: limits exceed total (overcommit)
_, currentLimitMB := m.CommittedMemory()
newLimitMB := ParseMemoryMB(meta.Resources.MemLimit)
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
deployWarning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
"Normál használat mellett ez nem okoz problémát."
}
// Slice 4: the block moved into memoryVerdict so the guarded update applies the SAME check with
// the SAME wording. Behaviour here is unchanged — same inputs, same log line, same refusal text.
refusal, deployWarning := m.memoryVerdict(ParseMemoryMB(meta.Resources.MemRequest), ParseMemoryMB(meta.Resources.MemLimit), 0, 0)
if refusal != "" {
clearDeploying()
return "", errors.New(refusal)
}
// Debug: log received values (redact passwords/secrets)
@@ -1197,3 +1158,62 @@ func randomAlphanumeric(length int) (string, error) {
}
return string(result), nil
}
// memoryVerdict is the deploy's memory check, extracted so the guarded update uses it unchanged
// (slice 4). releasedReqMB/releasedLimitMB are what the act FREES before it takes the new amount — an
// update replaces the app's own current request, so counting both would refuse an update that fits.
// A deploy releases nothing and passes 0, 0.
//
// Returns the refusal (the deploy's own Hungarian wording, "" = admitted) and the soft overcommit
// warning. An unreadable memory reading admits with a WARN, exactly as the deploy always has.
func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string) {
reservedMB := m.cfg.System.ReservedMemoryMB
totalMB, usedMB, memErr := system.GetMemoryMB()
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
// either never fire (host total) or always fire (host used > guest cap).
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
totalMB = gt
memErr = nil
}
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
usedMB = committedReqMB
}
if memErr != nil {
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
return "", ""
}
usedMB -= releasedReqMB
if usedMB < 0 {
usedMB = 0
}
usableMB := totalMB - reservedMB
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
// Hard block: committed + new request exceeds usable memory
if newReqMB > 0 && usedMB+newReqMB > usableMB {
return fmt.Sprintf(
"Nincs elég memória az alkalmazás telepítéséhez. "+
"Szükséges: %d MB, Elérhető: %d MB "+
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
newReqMB,
usableMB-usedMB,
totalMB,
usedMB,
reservedMB,
), ""
}
// Soft warning: limits exceed total (overcommit)
_, currentLimitMB := m.CommittedMemory()
currentLimitMB -= releasedLimitMB
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
warning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
"Normál használat mellett ez nem okoz problémát."
}
return "", warning
}