820e8efde1
gates / gates (push) Successful in 26s
A drive move persisted through the restore's fresh app.yaml write and dropped the pin: the syncer then copied the catalog verbatim and the next start jumped the app past its ladder (R-700). persistDriveFlip now changes HDD_PATH and nothing else. The restore's write carries the life records (conversion copies, desired_state, update history) from the app.yaml it replaces, and a second conversion no longer overwrites the first kept copy's record (R-697). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
1385 lines
57 KiB
Go
1385 lines
57 KiB
Go
package stacks
|
|
|
|
import (
|
|
"crypto/rand"
|
|
"encoding/base64"
|
|
"encoding/hex"
|
|
"fmt"
|
|
"log"
|
|
"math/big"
|
|
"os"
|
|
"path/filepath"
|
|
"regexp"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/crypto"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// reservedSubdomains lists subdomains reserved for system use.
|
|
var reservedSubdomains = map[string]bool{
|
|
"felhom": true, // controller dashboard
|
|
"files": true, // filebrowser
|
|
"traefik": true, // reverse proxy
|
|
"api": true,
|
|
"www": true,
|
|
"mail": true,
|
|
"smtp": true,
|
|
"ftp": true,
|
|
"admin": true,
|
|
"portal": true,
|
|
"ssh": true,
|
|
"ns1": true,
|
|
"ns2": true,
|
|
"mx": true,
|
|
"pop": true,
|
|
"imap": true,
|
|
}
|
|
|
|
var subdomainRe = regexp.MustCompile(`^[a-z0-9]([a-z0-9-]*[a-z0-9])?$`)
|
|
|
|
// validateSubdomain checks that a subdomain is DNS-safe.
|
|
func validateSubdomain(s string) error {
|
|
if s == "" {
|
|
return util.MsgError("err.stacks.az_aldomain_nem_lehet_ures")
|
|
}
|
|
if len(s) > 63 {
|
|
return fmt.Errorf("az aldomain legfeljebb 63 karakter lehet")
|
|
}
|
|
if !subdomainRe.MatchString(s) {
|
|
return util.MsgError("err.stacks.az_aldomain_csak_kisbetuket_szamokat_es")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// SubdomainInUse checks if a subdomain is already used by any deployed stack
|
|
// other than excludeStack.
|
|
func (m *Manager) SubdomainInUse(subdomain, excludeStack string) bool {
|
|
// Collect stack dirs and metadata under lock, then do I/O outside the lock.
|
|
type candidate struct {
|
|
dir string
|
|
metaSubdomain string
|
|
}
|
|
var candidates []candidate
|
|
|
|
m.mu.RLock()
|
|
for name, stack := range m.stacks {
|
|
if name == excludeStack || !stack.Deployed {
|
|
continue
|
|
}
|
|
candidates = append(candidates, candidate{
|
|
dir: filepath.Dir(stack.ComposePath),
|
|
metaSubdomain: stack.Meta.Subdomain,
|
|
})
|
|
}
|
|
m.mu.RUnlock()
|
|
|
|
for _, c := range candidates {
|
|
appCfg := LoadAppConfig(c.dir)
|
|
if appCfg == nil {
|
|
continue
|
|
}
|
|
if sd, ok := appCfg.Env["SUBDOMAIN"]; ok && sd == subdomain {
|
|
return true
|
|
}
|
|
if _, hasSub := appCfg.Env["SUBDOMAIN"]; !hasSub {
|
|
if c.metaSubdomain == subdomain {
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// AppConfig holds the per-app deployment configuration.
|
|
// Saved as app.yaml in each stack directory after first deployment.
|
|
type AppConfig struct {
|
|
Deployed bool `yaml:"deployed" json:"deployed"`
|
|
DeployedAt string `yaml:"deployed_at" json:"deployed_at"`
|
|
Env map[string]string `yaml:"env" json:"env"`
|
|
LockedFields []string `yaml:"locked_fields" json:"locked_fields"`
|
|
// EmailEnabled is the per-app app-email toggle (default off). When on AND the global toggle is
|
|
// on AND the app has an smtp_mapping, the controller injects the relay SMTP env at compose time.
|
|
EmailEnabled bool `yaml:"email_enabled,omitempty" json:"email_enabled,omitempty"`
|
|
// DesiredState (R-166 / decision D-b) is what the CUSTOMER asked for: DesiredStateRunning or
|
|
// DesiredStateStopped. It is TRI-state, and the third value is the entire safety property:
|
|
//
|
|
// ABSENT ("") MEANS UNKNOWN — IT NEVER MEANS "running".
|
|
//
|
|
// Every app.yaml on every existing box was written before this field existed, so absent is the
|
|
// overwhelmingly common value on upgrade. Reading it as "running" would start, on the next boot
|
|
// after the upgrade, every app its owner deliberately stopped — fleet-wide, silently. Where the
|
|
// state is unknown the boot reconciler falls back to its pre-R-166 behaviour instead of inventing
|
|
// an answer (see internal/bootrecon.isBootOrphan and the §8.1 table it implements).
|
|
//
|
|
// ONE OWNER: the customer's own action writes this and nothing else does. StartStack/StopStack
|
|
// are NOT writers — twelve of their fourteen callers are machines (quiesce, the backup volume
|
|
// dump, app export, the storage gate, migration, the boot reconciler), and recording intent in
|
|
// the primitive would make a nightly backup indistinguishable from the customer pressing Stop,
|
|
// which is the exact confusion this field exists to end. Writers: SetDesiredState's callers.
|
|
DesiredState string `yaml:"desired_state,omitempty" json:"desired_state,omitempty"`
|
|
// InstalledImages records what each compose service is ACTUALLY RUNNING, read from the
|
|
// containers after a successful compose up — never from docker-compose.yml, which the catalog
|
|
// syncer overwrites on a 15-minute cycle with no deployed check at all (measured live:
|
|
// SPIKE-app-update-2026-09-01 §3, where the file said v2.8.5 while the container ran v2.8.6 for
|
|
// 25 minutes). The file is the value that has already moved; the container is the fact.
|
|
//
|
|
// Keyed by COMPOSE SERVICE NAME, not container name: the service name is what the compose file
|
|
// and the catalog template both key on, so it is the only key a comparison can be made against.
|
|
//
|
|
// ABSENT MEANS UNKNOWN AND NEVER MEANS CURRENT (the R-166 rule, applied to an observation
|
|
// instead of an intent). Every app.yaml written before v0.233.0 has no entry here, so absent is
|
|
// the common value on upgrade; a reader that treated it as "up to date" would tell every
|
|
// customer on the fleet that their months-old app is current.
|
|
//
|
|
// WRITTEN BY: Manager.recordInstalledImages ONLY, from StartStack / RestartStack / UpdateStack
|
|
// and the deploy path. READ BY: web.updateBadge (v0.233.0). Nothing takes a DECISION from it.
|
|
InstalledImages map[string]InstalledImage `yaml:"installed_images,omitempty" json:"installed_images,omitempty"`
|
|
// PinnedImages is what this app is SUPPOSED to run, per compose service — the customer's
|
|
// INTENT, and the input the catalog render obeys (v0.235.0, operator ruling 2026-09-06:
|
|
// "freeze the version, keep the fixes flowing").
|
|
//
|
|
// IT IS NOT InstalledImages. That field is an OBSERVATION ("what is running"), written by
|
|
// looking at containers. This one is a DECISION ("what should run"), written only by an act
|
|
// that is entitled to move a version: a deploy, a deliberate update, a restore, or the
|
|
// one-time adoption pass. Letting an observation feed a decision would make a bad reading
|
|
// become a bad deployment — the same category error the desired_state field exists to avoid
|
|
// (R-166). They will normally agree; when they disagree that is a signal, not a bug to
|
|
// paper over.
|
|
//
|
|
// ABSENT MEANS UNPINNED, and unpinned means the app behaves exactly as it did before
|
|
// v0.235.0. It never means "pin to whatever the catalog says now".
|
|
PinnedImages map[string]string `yaml:"pinned_images,omitempty" json:"pinned_images,omitempty"`
|
|
// LastUpdateUndone (v0.263.0, 09 §3 decision 15) is the last update the box UNDID by itself —
|
|
// written only by a successful undo, cleared by the next successful update. The page shows one
|
|
// line from it; the future automatic caller reads it so it never re-presses the same step.
|
|
LastUpdateUndone *UpdateUndone `yaml:"last_update_undone,omitempty" json:"last_update_undone,omitempty"`
|
|
// FailedStep (v0.271.0, R-680) is the ladder step whose update was UNDONE or HELD — written by the
|
|
// undo and by the hold, cleared by the next successful update. The automatic leg never presses it
|
|
// again while the catalog's ladder for this app is the one it failed on (Ladder = LadderPrint); a
|
|
// person still can.
|
|
FailedStep *FailedStep `yaml:"failed_update_step,omitempty" json:"failed_update_step,omitempty"`
|
|
// LastAutoUpdate (v0.271.0, `09` §6.4 part 7) is the automatic leg's last step on this app — the
|
|
// line on the app page („Automatikus frissítés %s-kor — sikeres").
|
|
LastAutoUpdate *AutoUpdateRecord `yaml:"last_auto_update,omitempty" json:"last_auto_update,omitempty"`
|
|
// ConversionCopy (v0.273.0, `09` §6.4 part 10) is the OLD datadir's copy kept after a successful
|
|
// PostgreSQL major conversion, until a backup of the converted app is proven (ReleaseConversionCopies).
|
|
ConversionCopy *ConversionCopy `yaml:"conversion_copy,omitempty" json:"conversion_copy,omitempty"`
|
|
// EarlierConversionCopies (v0.276.0, R-697) are older kept copies a later conversion superseded — a
|
|
// restore to the old major, then the ladder converting again. Released by the same rule as
|
|
// ConversionCopy; without this list the newer record overwrote the older one and its volume was orphaned
|
|
// (TestR697_ASecondConversionDoesNotOrphanTheFirstCopy).
|
|
EarlierConversionCopies []ConversionCopy `yaml:"earlier_conversion_copies,omitempty" json:"earlier_conversion_copies,omitempty"`
|
|
// RestoredLogins (v0.275.0, R-694) are the `type: password` fields whose stored value was GENERATED by a
|
|
// restore (the unit never carries an admin login, D5, and the guest had none — a load of kept data, a
|
|
// removed app, a rebuilt guest) while the app's own login came back with its data. The page then shows
|
|
// no value for them and says the old password is the one that works (restoredLoginFields).
|
|
RestoredLogins []string `yaml:"restored_logins,omitempty" json:"restored_logins,omitempty"`
|
|
}
|
|
|
|
// InstalledImage is one compose service's observed image. See AppConfig.InstalledImages.
|
|
type InstalledImage struct {
|
|
// Ref is the reference the container was created FROM, i.e. docker inspect .Config.Image —
|
|
// e.g. "lscr.io/linuxserver/bookstack:26.05.2". This is what the template pins and what the
|
|
// comparison uses.
|
|
Ref string `yaml:"ref" json:"ref"`
|
|
// Digest is the repo digest of the image behind that reference — the only identifier that
|
|
// cannot move. Empty for an image that was never pulled from a registry (a locally built or
|
|
// imported image has no RepoDigests); an empty digest is recorded, never a skipped entry.
|
|
Digest string `yaml:"digest,omitempty" json:"digest,omitempty"`
|
|
// At is RFC3339 UTC: when this exact Ref+Digest pair was FIRST observed for this service. It is
|
|
// deliberately NOT re-stamped on every restart — an unchanged observation must not rewrite
|
|
// app.yaml (the SetDesiredState rule), and "running since" is more useful than "last looked at".
|
|
At string `yaml:"at" json:"at"`
|
|
}
|
|
|
|
// DeployRequest contains the user-provided values from the deploy form.
|
|
type DeployRequest struct {
|
|
StackName string `json:"stack_name"`
|
|
Values map[string]string `json:"values"` // env_var -> user-provided value
|
|
// KeptData is the household's answer when the app's drive folder already holds old data
|
|
// (`09` §3 decision 36): KeptChoiceFresh here; KeptChoiceUse is carried out by the API as a load
|
|
// from a backup and never reaches DeployStack. "" = no choice was made — refused when old data exists.
|
|
KeptData string `json:"kept_data,omitempty"`
|
|
}
|
|
|
|
// Kept-data choices at install (`09` §3 decision 36).
|
|
const (
|
|
KeptChoiceUse = "use"
|
|
KeptChoiceFresh = "fresh"
|
|
)
|
|
|
|
// DeployStack handles first-time deployment of an app.
|
|
// Returns a warning message (empty if none) and an error if deployment is blocked.
|
|
// 1. Check available memory against app requirements
|
|
// 2. Load metadata (.felhom.yml) to know what fields exist
|
|
// 3. Auto-generate secrets for secret fields (hidden from user)
|
|
// 4. Auto-fill domain from controller config
|
|
// 5. Validate all user-provided values (password, path, required fields)
|
|
// 6. Save app.yaml
|
|
// 7. Run docker compose up -d with env vars
|
|
// 8. Update in-memory stack state
|
|
func (m *Manager) DeployStack(req DeployRequest) (string, error) {
|
|
// Atomically check and set the Deploying flag to prevent concurrent deploys (H1 fix).
|
|
m.mu.Lock()
|
|
sPtr, sOk := m.stacks[req.StackName]
|
|
if !sOk {
|
|
m.mu.Unlock()
|
|
return "", fmt.Errorf("stack %q not found", req.StackName)
|
|
}
|
|
if sPtr.Deploying {
|
|
m.mu.Unlock()
|
|
return "", fmt.Errorf("stack %q is already being deployed — please wait", req.StackName)
|
|
}
|
|
if sPtr.Deployed {
|
|
m.mu.Unlock()
|
|
return "", util.KindErrorf(ErrAlreadyDeployed, "stack %q is already deployed; use update instead", req.StackName)
|
|
}
|
|
sPtr.Deploying = true
|
|
sPtr.DeployError = ""
|
|
m.mu.Unlock()
|
|
|
|
// If any validation below fails, clear the Deploying flag.
|
|
clearDeploying := func() {
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[req.StackName]; ok {
|
|
s.Deploying = false
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
|
|
stack, ok := m.GetStack(req.StackName)
|
|
if !ok {
|
|
clearDeploying()
|
|
return "", fmt.Errorf("stack %q not found", req.StackName)
|
|
}
|
|
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
meta := LoadMetadata(stackDir)
|
|
|
|
// --- Lifecycle gate (defence in depth) ---
|
|
// The API handler refuses this first, with the customer-facing Hungarian message. This second
|
|
// check exists because DeployStack is the manager-level choke point EVERY caller goes through,
|
|
// and metadata is already loaded here — so a future caller that does not route through the API
|
|
// cannot bypass the rule by simply not knowing about it. Deliberately before the first mutation.
|
|
if !meta.CanInstall() {
|
|
clearDeploying()
|
|
return "", fmt.Errorf("stack %q is not installable (lifecycle: %s)", req.StackName, meta.EffectiveLifecycle())
|
|
}
|
|
|
|
// --- Memory validation ---
|
|
// Slice 4: the block moved into memoryVerdict so the guarded update applies the SAME check with
|
|
// the SAME wording. Behaviour here is unchanged — same inputs, same log line, same refusal text.
|
|
refusal, deployWarning := m.memoryVerdict(ParseMemoryMB(meta.Resources.MemRequest), ParseMemoryMB(meta.Resources.MemLimit), 0, 0)
|
|
if refusal != nil {
|
|
clearDeploying()
|
|
return "", refusal
|
|
}
|
|
|
|
// Debug: log received values (redact passwords/secrets)
|
|
m.logger.Printf("[DEBUG] Deploy %s: received %d user values", req.StackName, len(req.Values))
|
|
for k, v := range req.Values {
|
|
if strings.Contains(strings.ToLower(k), "password") || strings.Contains(strings.ToLower(k), "secret") {
|
|
m.logger.Printf("[DEBUG] %s = [REDACTED, len=%d]", k, len(v))
|
|
} else {
|
|
m.logger.Printf("[DEBUG] %s = %q", k, v)
|
|
}
|
|
}
|
|
|
|
// Build the full env map
|
|
env := make(map[string]string)
|
|
var lockedFields []string
|
|
|
|
for _, field := range meta.DeployFields {
|
|
var value string
|
|
|
|
switch field.Type {
|
|
case "domain":
|
|
// Auto-fill from controller config
|
|
value = m.cfg.Customer.Domain
|
|
|
|
case "subdomain":
|
|
// User-editable with default from metadata
|
|
if userVal, ok := req.Values[field.EnvVar]; ok && userVal != "" {
|
|
value = strings.ToLower(strings.TrimSpace(userVal))
|
|
} else if field.Default != "" {
|
|
value = field.Default
|
|
}
|
|
if err := validateSubdomain(value); err != nil {
|
|
clearDeploying()
|
|
return "", err
|
|
}
|
|
if reservedSubdomains[value] {
|
|
clearDeploying()
|
|
return "", util.MsgError("err.stacks.a_z_aldomain_foglalt_rendszer_szamara", value)
|
|
}
|
|
if m.SubdomainInUse(value, req.StackName) {
|
|
clearDeploying()
|
|
return "", util.MsgError("err.stacks.a_z_aldomain_mar_hasznalatban_van", value)
|
|
}
|
|
|
|
case "secret":
|
|
// Use pre-generated value if provided by the deploy page (same value the user saw),
|
|
// otherwise fall back to generating a fresh one.
|
|
if userVal, ok := req.Values[field.EnvVar]; ok && userVal != "" {
|
|
value = userVal
|
|
} else {
|
|
generated, err := generateValue(field.Generate)
|
|
if err != nil {
|
|
clearDeploying()
|
|
return "", fmt.Errorf("generating %s: %w", field.EnvVar, err)
|
|
}
|
|
value = generated
|
|
}
|
|
|
|
case "password":
|
|
// Password fields MUST be filled by the user (via typing or Generálás button).
|
|
// We never silently auto-generate — the user needs to know their password.
|
|
if userVal, ok := req.Values[field.EnvVar]; ok && userVal != "" {
|
|
value = userVal
|
|
} else {
|
|
clearDeploying()
|
|
return "", util.MsgErrorf(ErrRequiredField, "err.stacks.field_required_password", field.Label)
|
|
}
|
|
|
|
default:
|
|
// text, path, select, boolean — use user value or default
|
|
if userVal, ok := req.Values[field.EnvVar]; ok {
|
|
value = userVal
|
|
} else if field.Default != "" {
|
|
value = field.Default
|
|
}
|
|
}
|
|
|
|
// Validate required fields
|
|
if field.Required && value == "" {
|
|
clearDeploying()
|
|
return "", util.MsgErrorf(ErrRequiredField, "err.stacks.field_required", field.Label, field.EnvVar)
|
|
}
|
|
|
|
// Validate path fields exist on the host filesystem
|
|
if field.Type == "path" && value != "" {
|
|
if _, err := os.Stat(value); os.IsNotExist(err) {
|
|
clearDeploying()
|
|
return "", util.KindErrorf(ErrPathMissing, "path %q does not exist for field %q", value, field.Label)
|
|
}
|
|
}
|
|
|
|
if value != "" {
|
|
env[field.EnvVar] = value
|
|
}
|
|
|
|
if field.LockedAfterDeploy {
|
|
lockedFields = append(lockedFields, field.EnvVar)
|
|
}
|
|
}
|
|
|
|
// `09` §3 decision 36 (R-657): an install NEVER runs into an app's old data silently. When the app's
|
|
// private drive folder already holds something, the household chooses: „start fresh" moves it into a
|
|
// dated kept folder (a rename on the same drive — nothing is deleted); „use my kept data" is a load
|
|
// from a backup, which the API performs instead of an install. No choice → refused, nothing moved.
|
|
// Pinned by TestKept_DeployRefusesOverOldDataWithoutAChoice.
|
|
if old := OldAppDataPaths(stack.ComposePath, env["HDD_PATH"]); len(old) > 0 {
|
|
switch req.KeptData {
|
|
case KeptChoiceFresh:
|
|
unit := ""
|
|
if m.keptUnitFn != nil {
|
|
unit = m.keptUnitFn(req.StackName, env["HDD_PATH"])
|
|
}
|
|
kept, err := m.KeepAside(req.StackName, env["HDD_PATH"], old, unit, time.Now())
|
|
if err != nil {
|
|
clearDeploying()
|
|
return "", fmt.Errorf("start fresh: %w", err)
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] Deploy %s: start fresh — the old data (%v) is kept in %s", req.StackName, old, kept)
|
|
case "":
|
|
clearDeploying()
|
|
m.logger.Printf("[WARN] [stacks] Deploy %s REFUSED: the drive already holds its old data %v and no choice was made", req.StackName, old)
|
|
return "", util.KindErrorf(ErrKeptDataChoice, "the drive already holds %s's old data (%s): choose use or fresh", req.StackName, strings.Join(old, ", "))
|
|
default:
|
|
clearDeploying()
|
|
return "", util.KindErrorf(ErrKeptDataChoice, "kept_data %q is not an install choice here (use is a load from a backup)", req.KeptData)
|
|
}
|
|
}
|
|
|
|
// Save app.yaml.
|
|
// CTRL-T2-1: persist the env now, but mark the ON-DISK state Deployed:false
|
|
// until `docker compose up -d` actually succeeds (done in runComposeDeploy).
|
|
// A crash/power-loss during the image-pull window must NOT leave a
|
|
// ghost-deployed stack on disk (Deployed:true with no containers), which
|
|
// DeployStack would then refuse to redeploy. The IN-MEMORY Deployed flag is
|
|
// still set true below to preserve the "no stale Telepítés button during
|
|
// pull" UX; only the durable record waits for success.
|
|
appCfg := &AppConfig{
|
|
Deployed: true, // in-memory truth (see below); the disk write overrides to false
|
|
DeployedAt: time.Now().UTC().Format(time.RFC3339),
|
|
Env: env,
|
|
LockedFields: lockedFields,
|
|
// R-166: deploying an app IS the customer asking for it to run, and this is the
|
|
// intent-before-the-act write (§8.2). Recorded on the transitional Deployed:false write too,
|
|
// which is harmless and correct: nothing reads desired state on a stack that is not deployed
|
|
// (isBootOrphan gates on Deployed first), and if the compose-up then fails, runComposeDeploy
|
|
// reverts Deployed to false — so a failed deploy can never present as an app owed a restart.
|
|
DesiredState: DesiredStateRunning,
|
|
}
|
|
|
|
diskCfg := *appCfg
|
|
diskCfg.Deployed = false // transitional: env saved, not yet marked deployed
|
|
if err := SaveAppConfig(stackDir, &diskCfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
clearDeploying()
|
|
return "", fmt.Errorf("saving app config: %w", err)
|
|
}
|
|
|
|
// Debug: log final env var keys (not values)
|
|
envKeys := make([]string, 0, len(env))
|
|
for k := range env {
|
|
envKeys = append(envKeys, k)
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] Deploying stack %s with %d env vars: [%s]", req.StackName, len(env), strings.Join(envKeys, ", "))
|
|
|
|
// Check which images are available locally before pulling
|
|
if m.isDebug() {
|
|
m.checkLocalImages(req.StackName, stackDir)
|
|
}
|
|
|
|
// Update in-memory stack state. Deploying was already set at the top (H1 fix).
|
|
// The compose-up runs in a goroutine so the API can return immediately
|
|
// and the UI shows progress via polling (image pull can take 30-60s).
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[req.StackName]; ok {
|
|
s.Deployed = true
|
|
s.AppConfig = appCfg
|
|
}
|
|
m.mu.Unlock()
|
|
|
|
// R-681: the install's own journal — removed when it ends either way; found at start = interrupted.
|
|
if err := markInstallPending(stackDir); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] Stack %s: cannot write the install marker (%v) — a restart mid-install would go unreported", req.StackName, err)
|
|
}
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[req.StackName]; ok {
|
|
s.InstallInterrupted = false
|
|
}
|
|
m.mu.Unlock()
|
|
|
|
// Run docker compose up -d asynchronously
|
|
go m.runComposeDeploy(req.StackName, stackDir, env, appCfg)
|
|
|
|
return deployWarning, nil
|
|
}
|
|
|
|
// runComposeDeploy executes docker compose up -d in background.
|
|
// On success it refreshes status; on failure it reverts the deploy state.
|
|
// SetDeployDoneHook registers the callback fired (in the deploy goroutine) when an async deploy
|
|
// ENDS — successfully or not. It is the seam R-536 needed: the deploy's outcome is known here and
|
|
// nowhere else, and the hub event that asserts an app is installed must hang off the outcome rather
|
|
// than off the acceptance.
|
|
//
|
|
// Same shape as SetMigrationDoneHook, deliberately: the policy (which event, what wording) lives in
|
|
// the caller, and this package only reports what happened.
|
|
func (m *Manager) SetDeployDoneHook(fn func(name string, ok bool, detail string)) {
|
|
m.deployDoneHook = fn
|
|
}
|
|
|
|
func (m *Manager) runComposeDeploy(name, stackDir string, env map[string]string, appCfg *AppConfig) {
|
|
start := time.Now()
|
|
_, composeErr := m.composeExecWithEnv(stackDir, env, "up", "-d")
|
|
|
|
if composeErr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] Stack %s deploy failed after %.1fs: %v", name, time.Since(start).Seconds(), composeErr)
|
|
// R-649 (v0.266.0, operator ruling 2026-09-23): a failed install REMOVES what it started, so
|
|
// „not installed" never stands over running containers (R-634's promise, for the deploy's OWN
|
|
// failures — e.g. a dependency whose healthcheck never passes after its container started).
|
|
// `down` WITHOUT -v: containers and the network go, named volumes stay — a reinstall of an app
|
|
// removed with „keep my data" must find its data again. A failed `down` is logged and the record
|
|
// still reads not-deployed; the household's Remove clears what is left (halfStateEvidence).
|
|
if _, downErr := m.composeExecWithEnv(stackDir, env, "down"); downErr != nil {
|
|
m.logger.Printf("[ERROR] [stacks] Stack %s: removing what the failed deploy started ALSO failed: %v — containers may remain; Remove clears them", name, downErr)
|
|
} else {
|
|
m.logger.Printf("[INFO] [stacks] Stack %s: the failed deploy's containers were removed (volumes kept) — R-649", name)
|
|
}
|
|
// Revert in-memory and disk state
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Deployed = false
|
|
s.Deploying = false
|
|
s.DeployError = composeErr.Error()
|
|
s.AppConfig = nil
|
|
}
|
|
// Also revert the shared appCfg under lock (C03 fix)
|
|
appCfg.Deployed = false
|
|
m.mu.Unlock()
|
|
// Save reverted state to disk with encryption (H05 fix)
|
|
meta := LoadMetadata(stackDir)
|
|
_ = SaveAppConfig(stackDir, appCfg, m.encKey, SensitiveEnvVars(&meta))
|
|
// R-536: the deploy ended, and it ended badly. Say so — the alternative is the silence that
|
|
// let an accept-time „Alkalmazás telepítve" stand as the last word on an app that never ran.
|
|
//
|
|
// The app.yaml is deliberately NOT deleted here: it is the crash-safe record written with
|
|
// Deployed:false, it carries the settings the customer typed, and a redeploy reuses them. The
|
|
// state the surfaces read is `not_deployed`, which is the fact that matters.
|
|
clearInstallPending(stackDir) // R-681: the install ended (badly) and said so
|
|
if m.deployDoneHook != nil {
|
|
m.deployDoneHook(name, false, composeErr.Error())
|
|
}
|
|
return
|
|
}
|
|
|
|
m.logger.Printf("[INFO] [stacks] Stack %s deployed successfully (took %.1fs)", name, time.Since(start).Seconds())
|
|
|
|
// CTRL-T2-1: compose up -d succeeded — only NOW mark deployed on disk.
|
|
// (DeployStack wrote the env with Deployed:false; flip it true here so the
|
|
// durable record matches reality and survives a restart.)
|
|
meta := LoadMetadata(stackDir)
|
|
if err := SaveAppConfig(stackDir, appCfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
// Running but not durably recorded as deployed. Revert so the customer
|
|
// can cleanly redeploy rather than be stuck with a half-recorded stack.
|
|
m.logger.Printf("[ERROR] [stacks] Stack %s: compose succeeded but persisting deployed state failed: %v — reverting", name, err)
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Deployed = false
|
|
s.Deploying = false
|
|
s.DeployError = "deploy succeeded but state could not be saved: " + err.Error()
|
|
s.AppConfig = nil
|
|
}
|
|
m.mu.Unlock()
|
|
clearInstallPending(stackDir)
|
|
return
|
|
}
|
|
|
|
clearInstallPending(stackDir) // R-681: the durable record says deployed — the install is over
|
|
|
|
// Clear deploying flag
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Deploying = false
|
|
}
|
|
m.mu.Unlock()
|
|
|
|
// Record what this deploy actually installed, per compose service (v0.233.0). Runs AFTER the
|
|
// SaveAppConfig above so it loads an app.yaml that already reads deployed=true. A failure here
|
|
// never fails the deploy — see recordInstalledImages.
|
|
deployEnv := m.stackEnv(stackDir)
|
|
m.recordInstalledImages(name, stackDir, deployEnv)
|
|
|
|
// Pin what we just deployed FROM (v0.235.0). The stack dir's compose file IS what the deploy
|
|
// used, so it is both the pin's source and the definition stored beside it. A failure here is
|
|
// logged and never fails the deploy — the app is up, and an unpinned app simply keeps
|
|
// pre-v0.235.0 behaviour.
|
|
if pin, data, err := PinFromCompose(ComposePathIn(stackDir)); err != nil {
|
|
m.logger.Printf("[WARN] [stacks] pin %s: cannot pin from the deployed compose file: %v", name, err)
|
|
} else if err := m.SetPin(name, stackDir, pin, data); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] pin %s: %v", name, err)
|
|
} else {
|
|
// v0.263.2: the deployed version's .felhom.yml — the probe an undo will judge it by later.
|
|
m.storeAppliedMetaFrom(name, stackDir, filepath.Join(stackDir, ".felhom.yml"))
|
|
}
|
|
|
|
// Post-deploy container state check (async, non-blocking)
|
|
m.logPostStartStatus(name, stackDir, deployEnv)
|
|
|
|
_ = m.RefreshStatus()
|
|
|
|
// R-536: ONLY NOW is „Alkalmazás telepítve" a true sentence — the compose up succeeded, the
|
|
// durable record says deployed, the images are recorded and the status has been refreshed. The
|
|
// observed state rides along rather than being asserted: a stack that is still `starting` is
|
|
// installed, and the detail says which it is instead of the event implying health it has not
|
|
// measured.
|
|
if m.deployDoneHook != nil {
|
|
state := ""
|
|
if s, ok := m.GetStack(name); ok {
|
|
state = string(s.State)
|
|
}
|
|
m.deployDoneHook(name, true, state)
|
|
}
|
|
}
|
|
|
|
// UpdateStackConfig updates non-locked fields for a deployed stack.
|
|
func (m *Manager) UpdateStackConfig(name string, values map[string]string) error {
|
|
m.logger.Printf("[INFO] [stacks] Updating config for stack %s", name)
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] UpdateStackConfig called: name=%q, %d values to update", name, len(values))
|
|
}
|
|
|
|
stack, ok := m.GetStack(name)
|
|
if !ok {
|
|
return fmt.Errorf("stack %q not found", name)
|
|
}
|
|
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
appCfg := LoadAppConfig(stackDir)
|
|
if appCfg == nil || !appCfg.Deployed {
|
|
return fmt.Errorf("stack %q is not deployed yet", name)
|
|
}
|
|
|
|
if appCfg.Env == nil {
|
|
appCfg.Env = make(map[string]string)
|
|
}
|
|
|
|
lockedSet := make(map[string]bool)
|
|
for _, f := range appCfg.LockedFields {
|
|
lockedSet[f] = true
|
|
}
|
|
|
|
meta := LoadMetadata(stackDir)
|
|
var changedKeys []string
|
|
for key, val := range values {
|
|
if lockedSet[key] {
|
|
return fmt.Errorf("field %q is locked and cannot be changed after deployment", key)
|
|
}
|
|
if appCfg.Env[key] != val {
|
|
changedKeys = append(changedKeys, key)
|
|
}
|
|
appCfg.Env[key] = val
|
|
}
|
|
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] UpdateStackConfig %s: changed keys: [%s], locked keys: %d", name, strings.Join(changedKeys, ", "), len(lockedSet))
|
|
}
|
|
|
|
if err := SaveAppConfig(stackDir, appCfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
return fmt.Errorf("saving updated config: %w", err)
|
|
}
|
|
|
|
// Use stackEnv which loads decrypted values for docker compose (C01 fix).
|
|
// appCfg.Env may contain encrypted values from LoadAppConfig.
|
|
env := m.stackEnv(stackDir)
|
|
if _, err := m.composeExecCustomEnv(stackDir, env, "up", "-d"); err != nil {
|
|
return fmt.Errorf("restarting with new config: %w", err)
|
|
}
|
|
|
|
m.logger.Printf("[INFO] [stacks] Stack %s config updated and restarted", name)
|
|
return m.RefreshStatus()
|
|
}
|
|
|
|
// RedeployFromEnv writes app.yaml from the given FULL env (encrypting secret fields) and (re-)deploys
|
|
// the stack with `docker compose up -d`, which re-pulls the pinned image. Used by the restore-from-unit
|
|
// flow (Phase 2b): unlike UpdateStackConfig it sets the full env INCLUDING locked secrets — which were
|
|
// recovered from the guest's own app.yaml, never regenerated. Caller is responsible for the gate.
|
|
func (m *Manager) RedeployFromEnv(name string, env map[string]string) error {
|
|
if err := m.PersistUnitRedeployConfig(name, env); err != nil {
|
|
return err
|
|
}
|
|
return m.upFromAppConfig(name)
|
|
}
|
|
|
|
// upFromAppConfig is RedeployFromEnv's up-and-report tail: `compose up -d` from the stored app.yaml.
|
|
func (m *Manager) upFromAppConfig(name string) error {
|
|
stack, ok := m.GetStack(name)
|
|
if !ok {
|
|
return fmt.Errorf("stack %q not found", name)
|
|
}
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
deployEnv := m.stackEnv(stackDir) // decrypts secrets back for compose
|
|
if _, err := m.composeExecCustomEnv(stackDir, deployEnv, "up", "-d"); err != nil {
|
|
return fmt.Errorf("compose up: %w", err)
|
|
}
|
|
m.logPostStartStatus(name, stackDir, deployEnv)
|
|
return m.RefreshStatus()
|
|
}
|
|
|
|
// PersistUnitRedeployConfig is the PERSIST half of RedeployFromEnv: it writes app.yaml from the full
|
|
// env (encrypting secret fields, recording locked fields) and marks the stack deployed in memory —
|
|
// and starts NOTHING.
|
|
//
|
|
// Split out for R-47. The restore paths must place the app's definition and then bring up only the
|
|
// database service for the dump replay; calling RedeployFromEnv there would end in a full
|
|
// `compose up -d` BEFORE the replay, which is exactly the race (H4) this work removes.
|
|
// RedeployFromEnv itself is this function plus the unchanged up-and-report tail, so its public
|
|
// behaviour is identical to before the split.
|
|
func (m *Manager) PersistUnitRedeployConfig(name string, env map[string]string) error {
|
|
stack, ok := m.GetStack(name)
|
|
if !ok {
|
|
return fmt.Errorf("stack %q not found", name)
|
|
}
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
meta := LoadMetadata(stackDir)
|
|
|
|
// R-694: which admin logins did the restore have to GENERATE? Exactly the `type: password` fields the
|
|
// guest held no value for before this write (the unit never carries one) — read BEFORE it is replaced.
|
|
prior := LoadAppConfigDecrypted(stackDir, m.encKey)
|
|
cfg := &AppConfig{
|
|
Deployed: true,
|
|
DeployedAt: time.Now().UTC().Format(time.RFC3339),
|
|
Env: env,
|
|
}
|
|
for _, f := range meta.DeployFields {
|
|
if f.LockedAfterDeploy {
|
|
cfg.LockedFields = append(cfg.LockedFields, f.EnvVar)
|
|
}
|
|
}
|
|
cfg.RestoredLogins = restoredLoginFields(name, meta, prior, env)
|
|
carryLifeRecords(m.logger, name, LoadAppConfig(stackDir), cfg)
|
|
if len(cfg.RestoredLogins) > 0 {
|
|
m.logger.Printf("[INFO] [stacks] %s: the restore generated %v — the app's own login came back with its data; the page will not show the new value as the password", name, cfg.RestoredLogins)
|
|
}
|
|
if err := SaveAppConfig(stackDir, cfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
return fmt.Errorf("saving app config: %w", err)
|
|
}
|
|
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[name]; ok {
|
|
s.Deployed = true
|
|
s.AppConfig = cfg
|
|
}
|
|
m.mu.Unlock()
|
|
|
|
m.logger.Printf("[INFO] [stacks] Redeploying %s from recovery unit with %d env vars", name, len(env))
|
|
return nil
|
|
}
|
|
|
|
// composeExecWithEnv runs a compose command with custom env vars injected. Used by the initial deploy
|
|
// path (DeployStack), which builds env from the deploy values rather than from app.yaml via stackEnv —
|
|
// so USERDATA_PATH must be injected here too (mirrors stackEnv), else the FIRST deploy resolves
|
|
// ${USERDATA_PATH} to "" and binds a bogus root-owned dir at the container root.
|
|
func (m *Manager) composeExecWithEnv(dir string, env map[string]string, args ...string) (string, error) {
|
|
cmdEnv := os.Environ()
|
|
for k, v := range env {
|
|
cmdEnv = append(cmdEnv, fmt.Sprintf("%s=%s", k, v))
|
|
}
|
|
cmdEnv = append(cmdEnv, fmt.Sprintf("DOMAIN=%s", m.cfg.Customer.Domain))
|
|
cmdEnv = withPathVars(cmdEnv, env["HDD_PATH"], m.sysDataPath, m.GetImportRoot())
|
|
return m.composeExecCustomEnv(dir, cmdEnv, args...)
|
|
}
|
|
|
|
// withPathVars appends the two derived path variables to a "K=V" env slice:
|
|
//
|
|
// USERDATA_PATH=<hdd>/userdata — per-app, on the app's OWN drive (when hdd is non-empty)
|
|
// IMPORT_PATH=<importRoot> — CANONICAL, on the system drive (when importRoot is non-empty)
|
|
//
|
|
// Shared by BOTH compose-env builders (stackEnv for start/redeploy, composeExecWithEnv for the initial
|
|
// deploy) so the variables always resolve — the initial-deploy path missing USERDATA_PATH bound a
|
|
// bogus root-owned dir at the container root, and IMPORT_PATH has the identical failure mode.
|
|
//
|
|
// An unresolvable importRoot is left UNSET on purpose (the caller logs it): compose then fails loudly
|
|
// on an unresolved ${IMPORT_PATH} rather than silently falling back to a per-drive path, which would
|
|
// recreate the dead-drop-zone shape R-75 exists to remove.
|
|
func withPathVars(cmdEnv []string, hdd, sysDataPath, importRoot string) []string {
|
|
if hdd != "" {
|
|
// R-203: UserdataDir takes a NAMESPACE ROOT, not a bare drive path. Passing `hdd` straight in
|
|
// bound <hdd>/userdata, which equals the namespace root only on an ENROLLED drive. On the
|
|
// system-data fallback it is one segment short, so the app wrote to a directory the off-site
|
|
// capture set never looked at — and the run still reported ok. Measured live on demo-hp.
|
|
cmdEnv = append(cmdEnv, "USERDATA_PATH="+appbackup.UserdataDir(appbackup.NamespaceRootFor(hdd, sysDataPath)))
|
|
}
|
|
if importRoot != "" {
|
|
cmdEnv = append(cmdEnv, "IMPORT_PATH="+importRoot)
|
|
}
|
|
return cmdEnv
|
|
}
|
|
|
|
// GetDeployFields returns the deployment fields for a stack (for the deploy form).
|
|
func (m *Manager) GetDeployFields(name string) (*Metadata, *AppConfig, error) {
|
|
stack, ok := m.GetStack(name)
|
|
if !ok {
|
|
return nil, nil, fmt.Errorf("stack %q not found", name)
|
|
}
|
|
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
meta := LoadMetadata(stackDir)
|
|
appCfg := LoadAppConfig(stackDir)
|
|
|
|
return &meta, appCfg, nil
|
|
}
|
|
|
|
// UpdateOptionalConfig updates optional env vars in app.yaml and restarts the stack if deployed.
|
|
// Only updates env vars that are listed in the metadata's optional_config sections.
|
|
func (m *Manager) UpdateOptionalConfig(stackName string, values map[string]string) error {
|
|
m.logger.Printf("[INFO] [stacks] Updating optional config for stack %s", stackName)
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] UpdateOptionalConfig called: stack=%q, %d values provided", stackName, len(values))
|
|
}
|
|
|
|
stack, ok := m.GetStack(stackName)
|
|
if !ok {
|
|
return fmt.Errorf("stack %q not found", stackName)
|
|
}
|
|
|
|
// Build a set of allowed env vars from optional_config
|
|
allowed := make(map[string]bool)
|
|
for _, group := range stack.Meta.OptionalConfig {
|
|
for _, field := range group.Fields {
|
|
allowed[field.EnvVar] = true
|
|
}
|
|
}
|
|
if len(allowed) == 0 {
|
|
return fmt.Errorf("no optional config fields defined for %s", stackName)
|
|
}
|
|
|
|
if m.isDebug() {
|
|
allowedKeys := make([]string, 0, len(allowed))
|
|
for k := range allowed {
|
|
allowedKeys = append(allowedKeys, k)
|
|
}
|
|
m.logger.Printf("[DEBUG] [stacks] UpdateOptionalConfig %s: allowed fields: [%s]", stackName, strings.Join(allowedKeys, ", "))
|
|
}
|
|
|
|
// Load existing app.yaml (or create empty one)
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
appCfg := LoadAppConfig(stackDir)
|
|
if appCfg == nil {
|
|
appCfg = &AppConfig{
|
|
Env: make(map[string]string),
|
|
}
|
|
}
|
|
if appCfg.Env == nil {
|
|
appCfg.Env = make(map[string]string)
|
|
}
|
|
|
|
// Update only allowed env vars
|
|
changed := false
|
|
for key, val := range values {
|
|
if !allowed[key] {
|
|
m.logger.Printf("[WARN] [stacks] Ignoring non-optional env var: %s", key)
|
|
continue
|
|
}
|
|
if appCfg.Env[key] != val {
|
|
appCfg.Env[key] = val
|
|
changed = true
|
|
m.logger.Printf("[INFO] [stacks] Updated optional config %s for %s", key, stackName)
|
|
}
|
|
}
|
|
|
|
if !changed {
|
|
return nil
|
|
}
|
|
|
|
// Save app.yaml
|
|
meta := LoadMetadata(stackDir)
|
|
if err := SaveAppConfig(stackDir, appCfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
return fmt.Errorf("saving app config: %w", err)
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] Saved updated app.yaml for %s", stackName)
|
|
|
|
// If deployed, recreate containers to pick up new env vars
|
|
// (docker compose restart does NOT pick up new env vars — must use up -d)
|
|
if stack.Deployed {
|
|
// R-166 — the THIRD customer-intent point, alongside the API action switch and deploy/import.
|
|
// This branch runs `up -d`, so the customer editing an app's settings ends with the app
|
|
// RUNNING; recording that keeps intent and reality in step. Written before the act (§8.2).
|
|
//
|
|
// Deliberately inside the `stack.Deployed` branch only: the other branch starts nothing, so
|
|
// it expresses no opinion about whether the app should run. Set on the already-loaded appCfg
|
|
// rather than through SetDesiredState so it rides the save just above instead of rewriting
|
|
// app.yaml twice — the load-then-save is what makes that safe (SaveAppConfig copies-and-
|
|
// overlays, so no other field is disturbed).
|
|
if appCfg.DesiredState != DesiredStateRunning {
|
|
appCfg.DesiredState = DesiredStateRunning
|
|
if err := SaveAppConfig(stackDir, appCfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
return fmt.Errorf("recording desired state before applying the new config: %w", err)
|
|
}
|
|
m.mu.Lock()
|
|
if s, ok := m.stacks[stackName]; ok && s.AppConfig != nil {
|
|
s.AppConfig.DesiredState = DesiredStateRunning
|
|
}
|
|
m.mu.Unlock()
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] Restarting %s to apply new optional config", stackName)
|
|
env := m.stackEnv(stackDir)
|
|
if _, err := m.composeExecCustomEnv(stackDir, env, "up", "-d"); err != nil {
|
|
return fmt.Errorf("restart after config update: %w", err)
|
|
}
|
|
m.logPostStartStatus(stackName, stackDir, env)
|
|
}
|
|
|
|
return m.RefreshStatus()
|
|
}
|
|
|
|
// DriveLive reports whether an app's data drive is a live mountpoint right now.
|
|
//
|
|
// It is the SAME signal the userdata belt uses (manager.go, the `isMountPoint` seam) rather than a
|
|
// second implementation, so the two can never disagree about whether a drive is there — a drift that
|
|
// would be invisible until one of them acted on it. The system/local path is legitimately not a
|
|
// mountpoint and is never gated, exactly as the belt treats it.
|
|
//
|
|
// R-171: exported because the boot reconciler must ask this question and lives in another package.
|
|
// Before v0.190.0 nothing asked it on that path, so the sweep started apps whose drive was absent —
|
|
// observed live on 2026-08-02 (audits/DIAG-bootrecon-drive-absent-2026-08-02.md).
|
|
func (m *Manager) DriveLive(hddPath string) bool {
|
|
if hddPath == "" || hddPath == m.sysDataPath {
|
|
return true // SSD-resident: no external drive to be absent
|
|
}
|
|
return m.isMountPoint(hddPath)
|
|
}
|
|
|
|
// LoadAppConfigByName reads app.yaml for a named stack. Returns nil if not found.
|
|
func (m *Manager) LoadAppConfigByName(stackName string) *AppConfig {
|
|
stack, ok := m.GetStack(stackName)
|
|
if !ok {
|
|
return nil
|
|
}
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
return LoadAppConfig(stackDir)
|
|
}
|
|
|
|
// PreviewDeployValues generates the auto-field values that will be used at deploy time:
|
|
// domain from controller config and freshly-generated secrets. These values are shown
|
|
// on the deploy page so the user can see (and note down) their passwords before deploying.
|
|
// Pass them back in DeployRequest.Values so the same values are saved to app.yaml.
|
|
func (m *Manager) PreviewDeployValues(name string) (map[string]string, error) {
|
|
stack, ok := m.GetStack(name)
|
|
if !ok {
|
|
return nil, fmt.Errorf("stack %q not found", name)
|
|
}
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
meta := LoadMetadata(stackDir)
|
|
|
|
result := make(map[string]string)
|
|
for _, field := range meta.DeployFields {
|
|
switch field.Type {
|
|
case "domain":
|
|
// Show the base domain. The subdomain is now a separate user-editable field.
|
|
result[field.EnvVar] = m.cfg.Customer.Domain
|
|
case "secret":
|
|
if field.Generate == "" {
|
|
continue
|
|
}
|
|
val, err := generateValue(field.Generate)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("generating preview for %s: %w", field.EnvVar, err)
|
|
}
|
|
result[field.EnvVar] = val
|
|
}
|
|
}
|
|
return result, nil
|
|
}
|
|
|
|
// --- App config persistence ---
|
|
|
|
func LoadAppConfig(stackDir string) *AppConfig {
|
|
path := filepath.Join(stackDir, "app.yaml")
|
|
data, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nil
|
|
}
|
|
cfg := &AppConfig{}
|
|
if err := yaml.Unmarshal(data, cfg); err != nil {
|
|
log.Printf("[WARN] [stacks] LoadAppConfig: %v", err)
|
|
return nil
|
|
}
|
|
return cfg
|
|
}
|
|
|
|
func SaveAppConfig(stackDir string, cfg *AppConfig, encKey []byte, sensitiveVars []string) error {
|
|
encryptedCount := 0
|
|
// COPY-AND-OVERLAY, never a field-by-field rebuild (the R-100 lesson, v0.181.0).
|
|
//
|
|
// This used to be a struct literal naming five fields. That shape is safe exactly until someone
|
|
// adds a sixth: the new field is silently dropped on every save, and because the save path is
|
|
// shared by nine call sites the loss shows up far from the code that caused it. R-100 shipped
|
|
// with two live instances of precisely this bug (offboxConfigHandler and ApplyOffsiteTarget both
|
|
// rebuilt a target field-by-field and erased LastSuccess).
|
|
//
|
|
// A value copy carries EVERY field the struct has, including ones added after this line was
|
|
// written, so it is safe by construction. Only Env is rebuilt below — it is the one field that
|
|
// needs transforming (encryption), and it must not alias the caller's map.
|
|
//
|
|
// LIMITATION, measured not assumed (TestSaveAppConfig_UnknownYAMLKeysAreDropped): keys present in
|
|
// the on-disk YAML that this struct does not model are NOT preserved — the round-trip goes
|
|
// through the struct, so yaml.Unmarshal discards them before this function ever sees them. That
|
|
// is unchanged by R-166 and is why every writer must load-then-save rather than construct.
|
|
saveCfg := *cfg
|
|
saveCfg.Env = make(map[string]string, len(cfg.Env))
|
|
sensitiveSet := make(map[string]bool, len(sensitiveVars))
|
|
for _, v := range sensitiveVars {
|
|
sensitiveSet[v] = true
|
|
}
|
|
for k, v := range cfg.Env {
|
|
if encKey != nil && sensitiveSet[k] && !crypto.IsEncrypted(v) && v != "" {
|
|
enc, err := crypto.Encrypt(encKey, v)
|
|
if err != nil {
|
|
// H10 (fail-closed): NEVER persist a sensitive value in plaintext.
|
|
// Earlier code logged a WARN and fell through to a plaintext write;
|
|
// that leaked the secret to disk. Abort the save instead — callers
|
|
// already propagate this error and the deploy fails cleanly.
|
|
return fmt.Errorf("encrypting sensitive env var %q (refusing to persist plaintext): %w", k, err)
|
|
}
|
|
saveCfg.Env[k] = enc
|
|
encryptedCount++
|
|
continue
|
|
}
|
|
saveCfg.Env[k] = v
|
|
}
|
|
|
|
log.Printf("[DEBUG] [stacks] SaveAppConfig: saving %s — %d env vars, %d encrypted, %d sensitive fields",
|
|
stackDir, len(saveCfg.Env), encryptedCount, len(sensitiveVars))
|
|
|
|
data, err := yaml.Marshal(saveCfg)
|
|
if err != nil {
|
|
log.Printf("[ERROR] [stacks] SaveAppConfig: failed to marshal config for %s: %v", stackDir, err)
|
|
return fmt.Errorf("marshaling app config: %w", err)
|
|
}
|
|
path := filepath.Join(stackDir, "app.yaml")
|
|
header := "# Auto-generated by felhom-controller — do not edit locked fields manually\n"
|
|
content := header + string(data)
|
|
|
|
// Atomic write: write to .tmp then rename (H04 fix)
|
|
tmpPath := path + ".tmp"
|
|
if err := os.WriteFile(tmpPath, []byte(content), 0600); err != nil {
|
|
log.Printf("[ERROR] [stacks] SaveAppConfig: failed to save %s: %v", path, err)
|
|
return fmt.Errorf("writing %s: %w", tmpPath, err)
|
|
}
|
|
if err := os.Rename(tmpPath, path); err != nil {
|
|
_ = os.Remove(tmpPath)
|
|
log.Printf("[ERROR] [stacks] SaveAppConfig: failed to save %s: %v", path, err)
|
|
return fmt.Errorf("renaming %s to %s: %w", tmpPath, path, err)
|
|
}
|
|
log.Printf("[INFO] [stacks] SaveAppConfig: saved config for %s", filepath.Base(stackDir))
|
|
return nil
|
|
}
|
|
|
|
// LoadAppConfigDecrypted loads app.yaml and decrypts any encrypted values.
|
|
func LoadAppConfigDecrypted(stackDir string, encKey []byte) *AppConfig {
|
|
cfg := LoadAppConfig(stackDir)
|
|
if cfg == nil {
|
|
return cfg
|
|
}
|
|
if encKey == nil {
|
|
log.Printf("[DEBUG] [stacks] LoadAppConfigDecrypted: no encryption key, returning raw config for %s", stackDir)
|
|
return cfg
|
|
}
|
|
cfg.Env = crypto.DecryptMap(encKey, cfg.Env)
|
|
return cfg
|
|
}
|
|
|
|
// SensitiveEnvVars returns the env var names for secret/password fields from metadata.
|
|
func SensitiveEnvVars(meta *Metadata) []string {
|
|
var vars []string
|
|
for _, f := range meta.DeployFields {
|
|
if f.Type == "secret" || f.Type == "password" {
|
|
vars = append(vars, f.EnvVar)
|
|
}
|
|
}
|
|
return vars
|
|
}
|
|
|
|
// nonPortableSecrets is the register of secrets that must NEVER travel in an on-drive recovery unit
|
|
// even though the catalog types them `secret` — i.e. credentials whose reach is NOT bounded by
|
|
// physical possession of the drive, because they authenticate against a service published to the
|
|
// internet. Keyed by catalog SLUG (never empty — LoadMetadata falls back to the directory name).
|
|
//
|
|
// It is a CODE register, not a catalog flag, deliberately: the D5 ruling is a security boundary, and
|
|
// a boundary a catalog push can silently move is not a boundary (the R-97a lesson — an invariant that
|
|
// only configuration enforced). Adding an app whose `type: secret` field gates an internet-reachable
|
|
// login means adding a row here.
|
|
//
|
|
// - vaultwarden/ADMIN_TOKEN gates the /admin panel, served on the app's own public web port.
|
|
var nonPortableSecrets = map[string]map[string]bool{
|
|
"vaultwarden": {"ADMIN_TOKEN": true},
|
|
}
|
|
|
|
// PortableSecretEnvVars returns the env-var names of secrets that TRAVEL inside the on-drive recovery
|
|
// unit (D5), in deterministic metadata order.
|
|
//
|
|
// The ruling (operator, 2026-07-30): `type: secret` travels, `type: password` does not, minus
|
|
// nonPortableSecrets. The line is drawn on REACH, not on whether a secret is nominally resettable:
|
|
//
|
|
// - Every `type: secret` field either decrypts data sitting on the SAME drive (the 5 declared
|
|
// data_keys, plus encryption keys the catalog labels as such but never flagged — see R-127) or
|
|
// authenticates to a container on an internal compose network with no external listener (the 18
|
|
// DB/root passwords, and the signing secrets). Possessing it adds nothing to possessing the
|
|
// drive, which is exactly D2's argument for keeping the DATA plaintext.
|
|
// - Every `type: password` field is an admin/UI login for a published service, so its blast radius
|
|
// is NOT bounded by the drive. Those stay in the guest and are regenerated on restore (O4).
|
|
//
|
|
// Excluding the `type: password` class is what licenses the plaintext ruling; the two are coupled and
|
|
// must not be relaxed independently.
|
|
func PortableSecretEnvVars(meta *Metadata) []string {
|
|
blocked := nonPortableSecrets[meta.Slug]
|
|
var vars []string
|
|
for _, f := range meta.DeployFields {
|
|
if f.Type == "secret" && !blocked[f.EnvVar] {
|
|
vars = append(vars, f.EnvVar)
|
|
}
|
|
}
|
|
return vars
|
|
}
|
|
|
|
// --- Secret generation ---
|
|
|
|
const alphanumChars = "abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789"
|
|
|
|
func generateValue(spec string) (string, error) {
|
|
if spec == "" {
|
|
return "", fmt.Errorf("empty generator spec")
|
|
}
|
|
parts := strings.SplitN(spec, ":", 2)
|
|
if len(parts) != 2 {
|
|
return "", fmt.Errorf("invalid generator spec: %q (expected type:param)", spec)
|
|
}
|
|
|
|
switch parts[0] {
|
|
case "password":
|
|
length := 0
|
|
if _, err := fmt.Sscanf(parts[1], "%d", &length); err != nil || length <= 0 {
|
|
return "", fmt.Errorf("invalid password length: %q", parts[1])
|
|
}
|
|
return randomAlphanumeric(length)
|
|
case "hex":
|
|
byteLen := 0
|
|
if _, err := fmt.Sscanf(parts[1], "%d", &byteLen); err != nil || byteLen <= 0 {
|
|
return "", fmt.Errorf("invalid hex length: %q", parts[1])
|
|
}
|
|
b := make([]byte, byteLen)
|
|
if _, err := rand.Read(b); err != nil {
|
|
return "", fmt.Errorf("reading random bytes: %w", err)
|
|
}
|
|
return hex.EncodeToString(b), nil
|
|
case "base64key":
|
|
byteLen := 0
|
|
if _, err := fmt.Sscanf(parts[1], "%d", &byteLen); err != nil || byteLen <= 0 {
|
|
return "", fmt.Errorf("invalid base64key length: %q", parts[1])
|
|
}
|
|
b := make([]byte, byteLen)
|
|
if _, err := rand.Read(b); err != nil {
|
|
return "", fmt.Errorf("reading random bytes: %w", err)
|
|
}
|
|
return "base64:" + base64.StdEncoding.EncodeToString(b), nil
|
|
case "static":
|
|
return parts[1], nil
|
|
default:
|
|
return "", fmt.Errorf("unknown generator type: %q", parts[0])
|
|
}
|
|
}
|
|
|
|
// GenerateSecretForField generates a replacement value for a stack's RESETTABLE secret deploy-field
|
|
// (O4: the restore-from-unit path uses this — via backup.SetSecretGenerator — when a resettable
|
|
// secret cannot be recovered from the guest's app.yaml, so the app redeploys with a fresh credential
|
|
// instead of a blank one that fails compose-up).
|
|
//
|
|
// Returns ok=false when the field is unknown, has no generator spec, or — deliberately — is a
|
|
// DATA-ENCRYPTING key: data-keys are NEVER generated (regenerating one would render stored data
|
|
// unreadable; the restore's fail-closed gate refuses before this point, this is defense-in-depth).
|
|
// The generated VALUE is never logged — names only.
|
|
func (m *Manager) GenerateSecretForField(stackName, envVar string) (string, bool) {
|
|
s, ok := m.GetStack(stackName)
|
|
if !ok {
|
|
return "", false
|
|
}
|
|
meta := LoadMetadata(filepath.Dir(s.ComposePath))
|
|
for _, f := range meta.DeployFields {
|
|
if f.EnvVar != envVar {
|
|
continue
|
|
}
|
|
if f.DataKey {
|
|
m.logger.Printf("[WARN] [stacks] GenerateSecretForField(%s/%s): refusing — field is a data-encrypting key", stackName, envVar)
|
|
return "", false
|
|
}
|
|
if (f.Type != "secret" && f.Type != "password") || f.Generate == "" {
|
|
return "", false
|
|
}
|
|
value, err := generateValue(f.Generate)
|
|
if err != nil || value == "" {
|
|
m.logger.Printf("[ERROR] [stacks] GenerateSecretForField(%s/%s): generator %q failed: %v", stackName, envVar, f.Generate, err)
|
|
return "", false
|
|
}
|
|
return value, true
|
|
}
|
|
return "", false
|
|
}
|
|
|
|
// InjectMissingFields checks deployed stacks for new deploy_fields that are not
|
|
// yet in app.yaml and auto-generates values for secret/domain fields.
|
|
// Called after sync (for updated stacks) and on startup (for all deployed stacks).
|
|
func (m *Manager) InjectMissingFields(stackNames []string) {
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] InjectMissingFields: checking %d stacks", len(stackNames))
|
|
}
|
|
|
|
count := 0
|
|
for _, name := range stackNames {
|
|
stack, ok := m.GetStack(name)
|
|
if !ok {
|
|
continue
|
|
}
|
|
count++
|
|
|
|
stackDir := filepath.Dir(stack.ComposePath)
|
|
meta := LoadMetadata(stackDir)
|
|
appCfg := LoadAppConfig(stackDir)
|
|
if appCfg == nil || !appCfg.Deployed {
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] InjectMissingFields: skipping %s (not deployed or no app config)", name)
|
|
}
|
|
continue
|
|
}
|
|
|
|
if m.isDebug() {
|
|
m.logger.Printf("[DEBUG] [stacks] InjectMissingFields: checking stack %s — %d deploy fields, %d existing env vars",
|
|
name, len(meta.DeployFields), len(appCfg.Env))
|
|
}
|
|
|
|
var injected []string
|
|
for _, field := range meta.DeployFields {
|
|
if _, exists := appCfg.Env[field.EnvVar]; exists {
|
|
continue // already present
|
|
}
|
|
|
|
switch field.Type {
|
|
case "secret":
|
|
if field.Generate == "" {
|
|
m.logger.Printf("[WARN] [stacks] Stack %s: new secret field %s has no generator — skipping", name, field.EnvVar)
|
|
continue
|
|
}
|
|
value, err := generateValue(field.Generate)
|
|
if err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] Stack %s: failed to generate %s: %v", name, field.EnvVar, err)
|
|
continue
|
|
}
|
|
appCfg.Env[field.EnvVar] = value
|
|
if field.LockedAfterDeploy {
|
|
appCfg.LockedFields = append(appCfg.LockedFields, field.EnvVar)
|
|
}
|
|
injected = append(injected, field.EnvVar)
|
|
|
|
case "domain":
|
|
appCfg.Env[field.EnvVar] = m.cfg.Customer.Domain
|
|
if field.LockedAfterDeploy && !containsStr(appCfg.LockedFields, field.EnvVar) {
|
|
appCfg.LockedFields = append(appCfg.LockedFields, field.EnvVar)
|
|
}
|
|
injected = append(injected, field.EnvVar)
|
|
|
|
case "subdomain":
|
|
// Auto-fill from field default or metadata subdomain
|
|
val := field.Default
|
|
if val == "" {
|
|
val = meta.Subdomain
|
|
}
|
|
if val == "" {
|
|
m.logger.Printf("[WARN] [stacks] Stack %s: new subdomain field %s has no default — skipping", name, field.EnvVar)
|
|
continue
|
|
}
|
|
appCfg.Env[field.EnvVar] = val
|
|
if field.LockedAfterDeploy && !containsStr(appCfg.LockedFields, field.EnvVar) {
|
|
appCfg.LockedFields = append(appCfg.LockedFields, field.EnvVar)
|
|
}
|
|
injected = append(injected, field.EnvVar)
|
|
|
|
default:
|
|
m.logger.Printf("[WARN] [stacks] Stack %s: new field %s (type=%s) requires manual configuration", name, field.EnvVar, field.Type)
|
|
}
|
|
}
|
|
|
|
if len(injected) > 0 {
|
|
if err := SaveAppConfig(stackDir, appCfg, m.encKey, SensitiveEnvVars(&meta)); err != nil {
|
|
m.logger.Printf("[ERROR] [stacks] Stack %s: failed to save app.yaml after injection: %v", name, err)
|
|
continue
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] Stack %s: injected missing fields: %s", name, strings.Join(injected, ", "))
|
|
}
|
|
}
|
|
m.logger.Printf("[INFO] [stacks] InjectMissingFields: processed %d stacks", count)
|
|
}
|
|
|
|
func containsStr(slice []string, s string) bool {
|
|
for _, v := range slice {
|
|
if v == s {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func randomAlphanumeric(length int) (string, error) {
|
|
result := make([]byte, length)
|
|
for i := range result {
|
|
n, err := rand.Int(rand.Reader, big.NewInt(int64(len(alphanumChars))))
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
result[i] = alphanumChars[n.Int64()]
|
|
}
|
|
return string(result), nil
|
|
}
|
|
|
|
// memoryVerdict is the deploy's memory check, extracted so the guarded update uses it unchanged
|
|
// (slice 4). releasedReqMB/releasedLimitMB are what the act FREES before it takes the new amount — an
|
|
// update replaces the app's own current request, so counting both would refuse an update that fits.
|
|
// A deploy releases nothing and passes 0, 0.
|
|
//
|
|
// Returns the refusal (nil = admitted) and the soft overcommit warning.
|
|
//
|
|
// v0.253.0 (R-557): the refusal is an ERROR rather than a sentence, and it carries BOTH its kind
|
|
// (ErrNotEnoughMemory, so api.deployStatusFor still answers 409) and its message key (so the
|
|
// household reads it in its own language). One value where there used to be a sentence plus a
|
|
// wrapper at each call site. An unreadable memory reading admits with a WARN, exactly as the deploy
|
|
// always has.
|
|
func (m *Manager) memoryVerdict(newReqMB, newLimitMB, releasedReqMB, releasedLimitMB int) (refusal error, warning string) {
|
|
reservedMB := m.cfg.System.ReservedMemoryMB
|
|
totalMB, usedMB, memErr := system.GetMemoryMB()
|
|
// F1: the controller container cannot read the guest's RAM cap from /proc (no lxcfs) or its own
|
|
// cgroup (the cap is on the LXC ancestor). Prefer the guest cap from the Docker daemon (runs in the
|
|
// LXC). And use the controller's OWN committed-memory accounting for "used" — accurate and cheap —
|
|
// rather than host /proc RSS, which is unobservable-per-guest and would otherwise make this guard
|
|
// either never fire (host total) or always fire (host used > guest cap).
|
|
if gt, ok := system.GuestMemTotalMB(); ok && gt > 0 {
|
|
totalMB = gt
|
|
memErr = nil
|
|
}
|
|
if committedReqMB, _ := m.CommittedMemory(); committedReqMB > 0 || memErr == nil {
|
|
usedMB = committedReqMB
|
|
}
|
|
if memErr != nil {
|
|
m.logger.Printf("[WARN] [stacks] Cannot read system memory: %v — skipping memory check", memErr)
|
|
return nil, ""
|
|
}
|
|
usedMB -= releasedReqMB
|
|
if usedMB < 0 {
|
|
usedMB = 0
|
|
}
|
|
usableMB := totalMB - reservedMB
|
|
|
|
m.logger.Printf("[INFO] [stacks] Memory check: total=%dMB, reserved=%dMB, usable=%dMB, committed_used=%dMB, new_req=%dMB, remaining=%dMB",
|
|
totalMB, reservedMB, usableMB, usedMB, newReqMB, usableMB-usedMB-newReqMB)
|
|
|
|
// Hard block: committed + new request exceeds usable memory
|
|
if newReqMB > 0 && usedMB+newReqMB > usableMB {
|
|
return util.MsgErrorf(ErrNotEnoughMemory, "err.stacks.not_enough_memory",
|
|
newReqMB,
|
|
usableMB-usedMB,
|
|
totalMB,
|
|
usedMB,
|
|
reservedMB,
|
|
), ""
|
|
}
|
|
|
|
// Soft warning: limits exceed total (overcommit)
|
|
_, currentLimitMB := m.CommittedMemory()
|
|
currentLimitMB -= releasedLimitMB
|
|
if newLimitMB > 0 && currentLimitMB+newLimitMB > totalMB {
|
|
warning = msgHU("err.stacks.memory_overcommit_warning")
|
|
}
|
|
return nil, warning
|
|
}
|
|
|
|
// loginAppliedEveryStart is the register of `type: password` fields whose app APPLIES the env value at
|
|
// EVERY start, so a value generated at a restore IS the login afterwards (R-694, measured 2026-09-26 from
|
|
// each image's entrypoint at the catalog's tag, `audits/version-travel-2026-09-26/D4/`): code-server's
|
|
// s6 run script passes $PASSWORD to `code-server --auth password` on every start and stores none. The six
|
|
// other apps with such a field (crafty-controller, gokapi, grafana, kimai, nextcloud, paperless-ngx) use it
|
|
// only at first initialisation — the restored data's login wins. Code, not a catalog flag, like
|
|
// nonPortableSecrets: it decides what a household is told about how to get into its own app.
|
|
var loginAppliedEveryStart = map[string]map[string]bool{
|
|
"code-server": {"PASSWORD": true},
|
|
}
|
|
|
|
// restoredLoginFields names the `type: password` fields a restore GENERATED — no value in the guest's
|
|
// app.yaml before, a value now — for an app whose login lives in its data. Pinned by TestR694_*.
|
|
func restoredLoginFields(app string, meta Metadata, prior *AppConfig, env map[string]string) []string {
|
|
var out []string
|
|
for _, f := range meta.DeployFields {
|
|
if f.Type != "password" || env[f.EnvVar] == "" || loginAppliedEveryStart[app][f.EnvVar] {
|
|
continue
|
|
}
|
|
if prior != nil && prior.Env[f.EnvVar] != "" {
|
|
continue // the guest kept the household's own value — not generated
|
|
}
|
|
out = append(out, f.EnvVar)
|
|
}
|
|
return out
|
|
}
|