v0.237.0: the Update button takes a backup first, and tells the truth (update arc slice 4 — R-448, R-443, R-439)
gates / gates (push) Successful in 13s
gates / gates (push) Successful in 13s
POST /api/stacks/{name}/update is now a guarded job answering 202:
cheap refusals (hold — R-439, busy, migration, deploying, memory via the
deploy's own memoryVerdict, a fixed 2 GB disk floor, and no restorable
Tier-2 copy) → backup-first when the proven copy is older than
update.backup_max_age (24h) → safety dump BEFORE the pin moves → pin →
pull (failure puts the pin back) → up → health (.felhom.yml check or 60 s
settle, update.health_timeout 5m). Not healthy → the app is stopped and
HELD (RestoreHold reason update_failed, same store and gate as R-379) and
the page names the backup to restore from; the pin stays. Success is only
ever update_phase=done after health (R-443). UpdateStack is deleted.
The restorable-unit predicate is EXTRACTED to backup.Tier2UnitRestorePoint
and shared with the backups page (row pinned unchanged). The copy is aged
by the last successful Tier-2 copy, not the manifest created_at — measured
on demo-hp that created_at moves only on definition changes.
Crash safety: update-journal.json before each phase; RecoverUpdates before
the boot sweep, ResumeInterruptedUpdates after the guards are wired.
Three unattended start paths ignored a hold and now honour it: the
drive-return gate (restart + boot recreate) and the nightly volume dump.
The nightly capture and Tier-2 run skip held apps so the restore point
survives. No automatic rollback — measured per-app; route back = restore.
Tests A–H across stacks/backup/api/web/cmd; six red-proofs seen to fail.
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -4,6 +4,7 @@ import (
|
||||
"crypto/rand"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"math/big"
|
||||
@@ -236,52 +237,12 @@ func (m *Manager) DeployStack(req DeployRequest) (string, error) {
|
||||
}
|
||||
|
||||
// --- Memory validation ---
|
||||
var deployWarning string
|
||||
reservedMB := m.cfg.System.ReservedMemoryMB
|
||||
totalMB, usedMB, memErr := system.GetMemoryMB()
|
||||
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
|
||||
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
|
||||
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
|
||||
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
|
||||
// either never fire (host total) or always fire (host used > guest cap).
|
||||
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
|
||||
totalMB = gt
|
||||
memErr = nil
|
||||
}
|
||||
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
|
||||
usedMB = committedReqMB
|
||||
}
|
||||
if memErr != nil {
|
||||
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
|
||||
} else {
|
||||
usableMB := totalMB - reservedMB
|
||||
newReqMB := ParseMemoryMB(meta.Resources.MemRequest)
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
|
||||
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
|
||||
|
||||
// Hard block: committed + new request exceeds usable memory
|
||||
if newReqMB > 0 && usedMB+newReqMB > usableMB {
|
||||
clearDeploying()
|
||||
return "", fmt.Errorf(
|
||||
"Nincs elég memória az alkalmazás telepítéséhez. "+
|
||||
"Szükséges: %d MB, Elérhető: %d MB "+
|
||||
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
|
||||
newReqMB,
|
||||
usableMB-usedMB,
|
||||
totalMB,
|
||||
usedMB,
|
||||
reservedMB,
|
||||
)
|
||||
}
|
||||
|
||||
// Soft warning: limits exceed total (overcommit)
|
||||
_, currentLimitMB := m.CommittedMemory()
|
||||
newLimitMB := ParseMemoryMB(meta.Resources.MemLimit)
|
||||
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
|
||||
deployWarning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
|
||||
"Normál használat mellett ez nem okoz problémát."
|
||||
}
|
||||
// Slice 4: the block moved into memoryVerdict so the guarded update applies the SAME check with
|
||||
// the SAME wording. Behaviour here is unchanged — same inputs, same log line, same refusal text.
|
||||
refusal, deployWarning := m.memoryVerdict(ParseMemoryMB(meta.Resources.MemRequest), ParseMemoryMB(meta.Resources.MemLimit), 0, 0)
|
||||
if refusal != "" {
|
||||
clearDeploying()
|
||||
return "", errors.New(refusal)
|
||||
}
|
||||
|
||||
// Debug: log received values (redact passwords/secrets)
|
||||
@@ -1197,3 +1158,62 @@ func randomAlphanumeric(length int) (string, error) {
|
||||
}
|
||||
return string(result), nil
|
||||
}
|
||||
|
||||
// memoryVerdict is the deploy's memory check, extracted so the guarded update uses it unchanged
|
||||
// (slice 4). releasedReqMB/releasedLimitMB are what the act FREES before it takes the new amount — an
|
||||
// update replaces the app's own current request, so counting both would refuse an update that fits.
|
||||
// A deploy releases nothing and passes 0, 0.
|
||||
//
|
||||
// Returns the refusal (the deploy's own Hungarian wording, "" = admitted) and the soft overcommit
|
||||
// warning. An unreadable memory reading admits with a WARN, exactly as the deploy always has.
|
||||
func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal, warning string) {
|
||||
reservedMB := m.cfg.System.ReservedMemoryMB
|
||||
totalMB, usedMB, memErr := system.GetMemoryMB()
|
||||
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
|
||||
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
|
||||
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
|
||||
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
|
||||
// either never fire (host total) or always fire (host used > guest cap).
|
||||
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
|
||||
totalMB = gt
|
||||
memErr = nil
|
||||
}
|
||||
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
|
||||
usedMB = committedReqMB
|
||||
}
|
||||
if memErr != nil {
|
||||
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
|
||||
return "", ""
|
||||
}
|
||||
usedMB -= releasedReqMB
|
||||
if usedMB < 0 {
|
||||
usedMB = 0
|
||||
}
|
||||
usableMB := totalMB - reservedMB
|
||||
|
||||
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
|
||||
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
|
||||
|
||||
// Hard block: committed + new request exceeds usable memory
|
||||
if newReqMB > 0 && usedMB+newReqMB > usableMB {
|
||||
return fmt.Sprintf(
|
||||
"Nincs elég memória az alkalmazás telepítéséhez. "+
|
||||
"Szükséges: %d MB, Elérhető: %d MB "+
|
||||
"(összesen: %d MB, ebből %d MB használt, %d MB rendszer számára fenntartva)",
|
||||
newReqMB,
|
||||
usableMB-usedMB,
|
||||
totalMB,
|
||||
usedMB,
|
||||
reservedMB,
|
||||
), ""
|
||||
}
|
||||
|
||||
// Soft warning: limits exceed total (overcommit)
|
||||
_, currentLimitMB := m.CommittedMemory()
|
||||
currentLimitMB -= releasedLimitMB
|
||||
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
|
||||
warning = "Az alkalmazások csúcsterhelése meghaladhatja a rendelkezésre álló memóriát. " +
|
||||
"Normál használat mellett ez nem okoz problémát."
|
||||
}
|
||||
return "", warning
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user