c1a73b24b3
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
3907 lines
187 KiB
Go
3907 lines
187 KiB
Go
package main
|
||
|
||
import (
|
||
"context"
|
||
"encoding/json"
|
||
"errors"
|
||
"flag"
|
||
"fmt"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/dockerexec"
|
||
"io"
|
||
"log"
|
||
"net/http"
|
||
"os"
|
||
"os/signal"
|
||
"path/filepath"
|
||
"sort"
|
||
"syscall"
|
||
"time"
|
||
|
||
"crypto/subtle"
|
||
"strings"
|
||
"sync"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/agentapi"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/api"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/appbackup"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/appexport"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/assets"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/backup"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/backupwindow"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/bootrecon"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/bootstrap"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/channelhealth"
|
||
cf "gitea.dooplex.hu/admin/felhom-controller/internal/cloudflare"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/crypto"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/fillwatch"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/infra"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/integrations"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/mailrelay"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/metrics"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/monitor"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/offsiteapply"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/quiesce"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/recovery"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/report"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/scheduler"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/selftest"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/selfupdate"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/setup"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/stacks"
|
||
catalogsync "gitea.dooplex.hu/admin/felhom-controller/internal/sync"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/web"
|
||
)
|
||
|
||
var (
|
||
// Set at build time via ldflags
|
||
Version = "dev"
|
||
BuildTime = "unknown"
|
||
GitCommit = "unknown"
|
||
)
|
||
|
||
func main() {
|
||
// Hidden re-exec mode (NAS verify-before-commit): the web server re-execs THIS binary as
|
||
// `felhom-controller --netprobe <dir>` with uid/gid-1000 credentials to prove a media app can
|
||
// write through a freshly-verified network share (SPIKE-nas-verify Q2). Handled before flag
|
||
// parsing; the exit code is the probe verdict (see web/netprobe.go).
|
||
if len(os.Args) == 3 && os.Args[1] == "--netprobe" {
|
||
os.Exit(web.NetProbeChild(os.Args[2]))
|
||
}
|
||
|
||
configPath := flag.String("config", "/opt/docker/felhom-controller/controller.yaml", "Path to configuration file")
|
||
showVersion := flag.Bool("version", false, "Show version and exit")
|
||
printResetCode := flag.Bool("print-reset-code", false, "Customer-claim escape hatch (v0.122.0, F-4): print a fresh one-time local claim/reset code to stdout, then exit. Root-gated by reachability (docker exec). Same gate consumes it.")
|
||
printInfraImages := flag.Bool("print-infra-images", false, "Print every controller-managed infra image (one per line) and exit. Read by the golden bake (felhom-agent configs/build-golden.sh) so the appliance image pre-pulls exactly what THIS controller version will request.")
|
||
recoverOffsiteCheck := flag.Bool("recover-offsite-check", false, "R-200 diagnostic: read the customer recovery code from STDIN, recover the offsite repository password from the hub-held sealed escrow via the agent, and report whether it matches the one on disk — BY HASH. Compares, never installs; writes nothing. Exit 0 = match, 2 = clean mismatch, 1 = a step failed.")
|
||
recoverOffsiteInstall := flag.Bool("recover-offsite-install", false, "R-200: read the customer recovery code from STDIN, recover the offsite repository password, print both hashes, and — with --confirm-install — PLACE it so the existing repository reopens. Without --confirm-install it is a dry run that writes nothing.")
|
||
confirmInstall := flag.Bool("confirm-install", false, "Required alongside --recover-offsite-install to actually write the recovered repository password. Deliberately a second invocation so the hashes are seen before any write is possible.")
|
||
// §7.5 (R-241): the operator levers for a running abandonment countdown. The automatic 30-day
|
||
// ending is deliberately NOT built (R-245) — what is built is the path that actually happens,
|
||
// which is the customer getting in touch and support needing something to press.
|
||
abandonStatus := flag.Bool("abandon-status", false, "R-241: print the state of this box's off-site abandonment countdown (set-aside path, due date, days left) and exit. Read-only.")
|
||
abandonExtend := flag.Int("abandon-extend", 0, "R-241 (operator): extend a running abandonment countdown by N days from now, then exit. Refuses when no countdown is running.")
|
||
abandonStop := flag.Bool("abandon-stop", false, "R-241 (operator): stop a running abandonment countdown, then exit. The set-aside history is kept and nothing is deleted. Refuses when no countdown is running.")
|
||
listRestoreHolds := flag.Bool("restore-holds", false, "R-379 (operator): list apps this controller is deliberately holding stopped after a failed database restore whose rollback also failed, then exit. Read-only.")
|
||
clearRestoreHold := flag.String("clear-restore-hold", "", "R-379 (operator): clear the restore hold on APP so it can start again, then exit. Clearing does NOT repair the database — check the app's data first; the undo copies are in its unit's db-dumps dir.")
|
||
flag.Parse()
|
||
|
||
if *showVersion {
|
||
fmt.Printf("felhom-controller %s (built %s, commit %s)\n", Version, BuildTime, GitCommit)
|
||
os.Exit(0)
|
||
}
|
||
|
||
// Config-free by design: the golden bake runs this against a bare `docker run` of the controller
|
||
// image, where no controller.yaml, data dir or settings exist yet. It must never touch either.
|
||
if *printInfraImages {
|
||
for _, img := range infra.Images() {
|
||
fmt.Println(img)
|
||
}
|
||
os.Exit(0)
|
||
}
|
||
|
||
// R-200 (v0.196.0) — the INSTALL sibling. Same fetch/unseal path, same stdin discipline for R;
|
||
// the difference is that it places the recovered password so a rebuilt box reopens the off-site
|
||
// history it inherited. Two invocations by design: without --confirm-install it prints the hashes
|
||
// and writes nothing.
|
||
if *recoverOffsiteInstall {
|
||
cfg, err := config.LoadPermissive(*configPath)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "recover-offsite-install: loading config: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
sett, err := settings.Load(cfg.Paths.DataDir+"/settings.json", log.New(os.Stderr, "", 0))
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "recover-offsite-install: loading settings: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
ac, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "recover-offsite-install: agent channel: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
os.Exit(backup.RecoverAndInstall(backup.RecoveryCheckDeps{
|
||
Manager: backup.NewManager(cfg, sett, log.New(os.Stderr, "", 0)),
|
||
Recoverer: ac,
|
||
In: os.Stdin,
|
||
Out: os.Stdout,
|
||
Err: os.Stderr,
|
||
}, *confirmInstall))
|
||
}
|
||
|
||
// R-200 (v0.195.0) — the offsite-key recovery check. A `docker exec` escape hatch in the shape of
|
||
// --print-reset-code above, and deliberately NOT a page or a browser-reachable API: the
|
||
// customer-facing flow is designed on top of a chain that has been walked, and this is the walk.
|
||
// R arrives on STDIN so it never appears in argv, `ps`, shell history or a session transcript.
|
||
if *recoverOffsiteCheck {
|
||
cfg, err := config.LoadPermissive(*configPath)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "recover-offsite-check: loading config: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
sett, err := settings.Load(cfg.Paths.DataDir+"/settings.json", log.New(os.Stderr, "", 0))
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "recover-offsite-check: loading settings: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
ac, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "recover-offsite-check: agent channel: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
os.Exit(backup.RunRecoveryCheck(backup.RecoveryCheckDeps{
|
||
Manager: backup.NewManager(cfg, sett, log.New(os.Stderr, "", 0)),
|
||
Recoverer: ac,
|
||
In: os.Stdin,
|
||
Out: os.Stdout,
|
||
Err: os.Stderr,
|
||
}))
|
||
}
|
||
|
||
// §7.5 (R-241) — the operator's abandonment levers. Grouped in one block, before the server
|
||
// starts, exactly like the other CLI subcommands: each loads config + settings, acts, and exits.
|
||
// R-379 (operator): the way OUT of a hold. A CLI subcommand rather than an API route, matching
|
||
// --abandon-stop above: the controller's HTTP surface is authenticated as the CUSTOMER, and
|
||
// clearing a hold is a decision that should follow an operator looking at the database. Reaching
|
||
// this needs `docker exec` into the guest, which is the operator gate this project already uses.
|
||
if *listRestoreHolds || *clearRestoreHold != "" {
|
||
cfg, err := config.LoadPermissive(*configPath)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "restore-hold: loading config: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
lg := log.New(os.Stderr, "", 0)
|
||
sett, err := settings.Load(cfg.Paths.DataDir+"/settings.json", lg)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "restore-hold: loading settings: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
if *listRestoreHolds {
|
||
holds := sett.ListRestoreHolds()
|
||
if len(holds) == 0 {
|
||
fmt.Println("no restore holds in force")
|
||
os.Exit(0)
|
||
}
|
||
for _, h := range holds {
|
||
fmt.Printf("%s\theld since %s\n replay error : %s\n rollback err : %s\n", h.Stack, h.At, h.ReplayError, h.RollbackErr)
|
||
}
|
||
os.Exit(0)
|
||
}
|
||
cleared, err := sett.ClearRestoreHold(*clearRestoreHold)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "restore-hold: clearing %s: %v\n", *clearRestoreHold, err)
|
||
os.Exit(1)
|
||
}
|
||
if !cleared {
|
||
// "there was nothing to clear" is not success — a silent 0 here would let an operator
|
||
// believe they had unblocked an app whose name they mistyped.
|
||
fmt.Fprintf(os.Stderr, "no restore hold in force for %q — nothing was changed\n", *clearRestoreHold)
|
||
os.Exit(2)
|
||
}
|
||
// THE RESTART IS NOT OPTIONAL, and saying so is the difference between a route that works
|
||
// and one an operator believes worked. This runs as a SECOND process: it clears the hold in
|
||
// settings.json, but the RUNNING controller holds its own in-memory Settings and keeps
|
||
// refusing until it reloads. Measured on demo-hp 2026-08-22 — the clear succeeded, the file
|
||
// was correct, and the customer's start button still refused until the controller restarted.
|
||
//
|
||
// There is also a lost-update window while both processes hold the file: if the running
|
||
// controller saves settings.json from its own copy before the restart, the clear is undone.
|
||
// Restarting promptly closes it. A route that clears through the running controller would
|
||
// remove both problems and is the right shape later; it needs an operator tier the
|
||
// controller's HTTP surface does not have today (it authenticates as the CUSTOMER, and a
|
||
// customer clearing their own hold is precisely what the hold exists to prevent).
|
||
fmt.Printf("restore hold cleared for %s in settings.json.\n", *clearRestoreHold)
|
||
fmt.Printf(" NOW RESTART THE CONTROLLER, or it will keep refusing to start the app:\n")
|
||
fmt.Printf(" systemctl restart felhom-controller-bootstrap.service\n")
|
||
fmt.Printf(" Check the app's data first — the undo copies are in its unit's db-dumps dir.\n")
|
||
os.Exit(0)
|
||
}
|
||
|
||
if *abandonStatus || *abandonExtend > 0 || *abandonStop {
|
||
cfg, err := config.LoadPermissive(*configPath)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "abandon: loading config: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
lg := log.New(os.Stderr, "", 0)
|
||
sett, err := settings.Load(cfg.Paths.DataDir+"/settings.json", lg)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "abandon: loading settings: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
mgr := backup.NewManager(cfg, sett, lg)
|
||
st := mgr.AbandonStatus()
|
||
switch {
|
||
case *abandonStop:
|
||
if serr := mgr.StopAbandon(); serr != nil {
|
||
fmt.Fprintf(os.Stderr, "abandon-stop: %v\n", serr)
|
||
os.Exit(2)
|
||
}
|
||
fmt.Printf("abandonment STOPPED — the set-aside history at %s is kept; nothing was deleted\n", st.RepoPath)
|
||
case *abandonExtend > 0:
|
||
due, eerr := mgr.ExtendAbandon(*abandonExtend)
|
||
if eerr != nil {
|
||
fmt.Fprintf(os.Stderr, "abandon-extend: %v\n", eerr)
|
||
os.Exit(2)
|
||
}
|
||
fmt.Printf("abandonment EXTENDED by %d day(s) — the set-aside history at %s is now deleted on %s\n",
|
||
*abandonExtend, st.RepoPath, due.Format("2006-01-02"))
|
||
default:
|
||
if !st.Active && !st.PurgeRequested {
|
||
fmt.Println("no abandonment countdown is running on this box")
|
||
os.Exit(0)
|
||
}
|
||
if st.PurgeRequested {
|
||
fmt.Printf("set-aside history DELETED at %s; awaiting the hub to drop the sealed package\n", st.RepoPath)
|
||
os.Exit(0)
|
||
}
|
||
fmt.Printf("abandonment countdown RUNNING\n set-aside history : %s\n chosen on : %s\n deleted on : %s\n days left : %d\n",
|
||
st.RepoPath, st.StartedAt.Format("2006-01-02"), st.DueAt.Format("2006-01-02"), st.DaysLeft)
|
||
}
|
||
os.Exit(0)
|
||
}
|
||
|
||
if *printResetCode {
|
||
cfg, err := config.LoadPermissive(*configPath)
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "print-reset-code: loading config: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
sett, err := settings.Load(cfg.Paths.DataDir+"/settings.json", log.New(os.Stderr, "", 0))
|
||
if err != nil {
|
||
fmt.Fprintf(os.Stderr, "print-reset-code: loading settings: %v\n", err)
|
||
os.Exit(1)
|
||
}
|
||
os.Exit(web.PrintLocalResetCode(sett, cfg))
|
||
}
|
||
|
||
startTime := time.Now()
|
||
|
||
// --- Load configuration ---
|
||
// Use LoadPermissive to tolerate minimal configs (e.g. only domain set by docker-setup.sh).
|
||
// If even that fails (file missing/unreadable), fall back to defaults.
|
||
cfg, err := config.LoadPermissive(*configPath)
|
||
if err != nil {
|
||
cfg = config.Default()
|
||
log.Printf("[WARN] Config load failed (%s), using defaults: %v", *configPath, err)
|
||
}
|
||
|
||
logger, logBuffer := setupLogger(cfg)
|
||
// fix-6 (CAMPAIGN-3): load the pre-restart debug-ring window back immediately, so a controller
|
||
// restart / container recreation no longer wipes the exact evidence an operator needs. The spill
|
||
// lives on the SSD state dir (DataDir — holds settings.json, survives recreation), NEVER a NAS path.
|
||
ringSpillPath := filepath.Join(cfg.Paths.DataDir, "debug-ring.log")
|
||
logBuffer.LoadFrom(ringSpillPath)
|
||
// v0.116.0: the debug ring is the controller_log_tail source (report self-tail channel).
|
||
report.SetControllerLogSource(logBuffer.Lines)
|
||
|
||
// --- Bootstrap ingestion (slice 8A → v0.40.0 onboarding, doc 03 §6) ---
|
||
// On first run, if this controller is not yet configured AND the host agent's provisioning
|
||
// back-half attached a bootstrap.json config mount, PULL the full controller.yaml from the hub
|
||
// (using the bootstrap's retrieval passphrase), merge in the per-guest local_api block, and come
|
||
// up CONFIGURED — skipping setup mode. Idempotent (never clobbers an existing controller.yaml)
|
||
// and fail-safe (a malformed/absent bootstrap, or a hub outage at first boot, leaves us in setup
|
||
// mode). The adapter marks a transient hub-unreachable error as retryable (the rest are permanent).
|
||
pull := func(hubURL, customerID, retrievalPassword string) (string, error) {
|
||
y, perr := report.PullConfig(hubURL, customerID, retrievalPassword)
|
||
if perr != nil && errors.Is(perr, report.ErrHubUnreachable) {
|
||
return "", fmt.Errorf("%w: %w", bootstrap.ErrPullTransient, perr)
|
||
}
|
||
return y, perr
|
||
}
|
||
cfg = bootstrap.MaybeIngest(*configPath, cfg, logger, pull)
|
||
|
||
// --- Wire system package debug logging ---
|
||
if cfg.Logging.Level == "debug" {
|
||
system.DebugLogger = logger
|
||
}
|
||
|
||
// --- Setup mode: if no customer ID configured, run setup wizard ---
|
||
if setup.NeedsSetup(cfg) {
|
||
logger.Printf("[INFO] felhom-controller %s — setup mode", Version)
|
||
runSetupMode(cfg, logger)
|
||
return
|
||
}
|
||
|
||
// --- Local API connectivity probe (slice 8A) ---
|
||
// When seeded with a local-API endpoint, prove the controller↔agent channel at startup and
|
||
// learn this guest's mounts (placement view). Non-fatal — the controller runs regardless; a
|
||
// failure is logged for diagnosis. The full /backup/due quiesce loop lands in 8B.
|
||
probeLocalAPI(cfg, logger)
|
||
|
||
logger.Printf("[INFO] felhom-controller %s starting (customer: %s, domain: %s)",
|
||
Version, cfg.Customer.ID, cfg.Customer.Domain)
|
||
|
||
// --- Load settings ---
|
||
settingsPath := cfg.Paths.DataDir + "/settings.json"
|
||
sett, err := settings.Load(settingsPath, logger)
|
||
if err != nil {
|
||
logger.Fatalf("[FATAL] Failed to load settings from %s: %v", settingsPath, err)
|
||
}
|
||
sett.SetDebug(cfg.Logging.Level == "debug")
|
||
// v0.256.0 (R-558): the language the hub says this box should START in. Consulted by
|
||
// GetLanguage ONLY when the household has never chosen — an explicit choice in settings.json
|
||
// always wins, so a config pull cannot unseat it. A config change restarts the controller, so
|
||
// binding it once here is enough. Pinned by TestConfigLanguageIsWiredInMain.
|
||
sett.SetConfigLanguage(cfg.Customer.Language)
|
||
|
||
// --- Auto-discover storage paths from deployed apps ---
|
||
discoveredPaths := discoverHDDPaths(cfg.Paths.StacksDir, logger)
|
||
sett.AutoDiscoverStoragePaths(discoveredPaths, cfg.Paths.HDDPath, logger)
|
||
|
||
// --- Load or create encryption key ---
|
||
encKeyPath := filepath.Join(cfg.Paths.DataDir, "encryption.key")
|
||
encKey, err := crypto.LoadOrCreateKey(encKeyPath)
|
||
if err != nil {
|
||
logger.Fatalf("[FATAL] Failed to load encryption key: %v", err)
|
||
}
|
||
logger.Printf("[INFO] Encryption key loaded from %s", encKeyPath)
|
||
|
||
// --- Initialize stack manager ---
|
||
stackMgr, err := stacks.NewManager(cfg, logger)
|
||
if err != nil {
|
||
logger.Fatalf("[FATAL] Failed to initialize stack manager: %v", err)
|
||
}
|
||
stackMgr.SetEncryptionKey(encKey)
|
||
stacks.SetControllerVersion(Version) // v0.287.0: a template's min_controller is enforced against this
|
||
|
||
// Initial stack scan
|
||
if err := stackMgr.ScanStacks(); err != nil {
|
||
logger.Printf("[WARN] Initial stack scan failed: %v", err)
|
||
}
|
||
|
||
// Inject missing deploy fields for all deployed stacks on startup
|
||
if names := stackMgr.DeployedStackNames(); len(names) > 0 {
|
||
stackMgr.InjectMissingFields(names)
|
||
}
|
||
|
||
// Migrate existing plaintext passwords to encrypted
|
||
stackMgr.MigrateEncryption()
|
||
|
||
// --- First-boot base-infrastructure bring-up ---
|
||
// We are guaranteed configured here (setup.NeedsSetup returned false above), so deploy the base
|
||
// stack (traefik-public network → traefik → cloudflared → filebrowser) the controller needs for
|
||
// routing + external access. Runs in a goroutine so a slow first-boot image pull never delays the
|
||
// web server; non-fatal (idempotent + single-flight, the health loop re-attempts each tick).
|
||
go func() {
|
||
if err := stackMgr.EnsureBaseStack(); err != nil {
|
||
logger.Printf("[WARN] [infra] first-boot base-stack bring-up: %v", err)
|
||
}
|
||
}()
|
||
|
||
// --- Initialize catalog syncer ---
|
||
syncer := catalogsync.New(cfg, logger, stackMgr.ScanStacks, func(updated []string) {
|
||
stackMgr.InjectMissingFields(updated)
|
||
})
|
||
// v0.235.0 — the render seam. Without this line the syncer copies the catalog verbatim, i.e.
|
||
// the pre-v0.235.0 product; the freeze is INERT unless it is wired here, which is why a test
|
||
// walks this file's AST for the call rather than trusting the package to be correct alone.
|
||
syncer.SetRenderPlanFn(stackMgr.RenderPlanFor)
|
||
// Start() is deliberately NOT called here — see the comment beside the call further down, after
|
||
// the pin adoption pass.
|
||
defer syncer.Stop()
|
||
|
||
// --- Graceful shutdown context ---
|
||
ctx, cancel := context.WithCancel(context.Background())
|
||
defer cancel()
|
||
|
||
// --- Quiesce loop (slice 8B): app-consistent backup around the agent vzdump ---
|
||
// Runs only when the local API is configured (a provisioned guest) and quiesce is enabled.
|
||
// Recover FIRST (restart any stacks left stopped by a crash mid-quiesce), then start the loop.
|
||
quiesceLoop := startQuiesceLoop(ctx, cfg, sett, stackMgr, logger)
|
||
|
||
// --- R-166: recover apps left stopped by an interrupted app-data operation ---
|
||
// A volume dump, an offsite reconstitution or a `.fab` export stops the app, works on its data,
|
||
// and starts it again. A controller killed inside that window used to leave the app down with
|
||
// NOTHING on disk explaining it — and a stopped app has zero containers, which the boot
|
||
// reconciler below read as a deliberate customer stop and left alone, indefinitely.
|
||
//
|
||
// ORDERING IS LOAD-BEARING (§8.4) and this call must COMPLETE, not merely be reached, before the
|
||
// boot-reconcile goroutine is launched: an app the marker already explains must not also be
|
||
// reported as an unexplained boot orphan. Same position and same reason as the quiesce Recover
|
||
// immediately above.
|
||
// The guard is built HERE rather than taken from the backup manager because that manager is not
|
||
// constructed until ~40 lines below — and moving its construction up to suit this would be a far
|
||
// wider change than moving one object down. It is handed to the manager (SetAppStopGuard) and to
|
||
// the exporter later, so all three share ONE guard over ONE file.
|
||
//
|
||
// R-174: the starter is GATED, never the raw manager. Recover runs here, at startup — exactly
|
||
// when an external drive may not have come back — and `Manager.StartStack` has no drive gate of
|
||
// its own, so the un-gated version started apps onto missing drives. The gate is the SAME
|
||
// `driveStartGate` the boot sweep uses (holder #3 of bootDriveGate), so the two cannot disagree.
|
||
appStopGuard := backup.NewAppStopGuard(filepath.Join(cfg.Paths.DataDir, "appstop-state.json"), logger)
|
||
appStopGuard.SetStarter(gatedAppStopStarter{
|
||
inner: stackMgr,
|
||
gate: driveStartGate{mgr: stackMgr, sett: sett},
|
||
logger: logger,
|
||
})
|
||
appStopRecovery := appStopGuard.Recover()
|
||
|
||
// --- R-166: desired-state backfill (running-only) ---
|
||
// Converge the apps whose intent is unambiguous — deployed and observed UP — so the fleet stops
|
||
// depending on legacy inference without waiting for a button press. NEVER backfills "stopped":
|
||
// zero containers cannot distinguish a deliberate stop from a power cut, and that inference is
|
||
// the defect. Runs after the two recoveries so a just-restarted app is counted as running.
|
||
stackMgr.BackfillDesiredState()
|
||
|
||
// --- v0.234.0: installed-images backfill (complete observations only) ---
|
||
// v0.233.0 recorded what an app installed only from the four bring-up paths, so an app nobody
|
||
// restarts carried no record — and therefore no „Naprakész"/„Frissítés elérhető" badge — for as
|
||
// long as it went untouched. On a quiet box that is EVERY app, which is the box we most want to
|
||
// be able to see. This seeds the absences by READING the containers: it starts nothing, restarts
|
||
// nothing and writes no compose file. It never overwrites an existing record, and it refuses to
|
||
// seed an app it cannot observe COMPLETELY — a partial record renders as "behind" on an app that
|
||
// is current, and a confident wrong answer is worse than the silence it replaces.
|
||
// Placed here, beside the desired-state backfill and after the two recoveries, for the same
|
||
// reason: a just-restarted app is observed in its settled state.
|
||
stackMgr.BackfillInstalledImages()
|
||
|
||
// --- v0.235.0: pin adoption (complete AND matching observations only) ---
|
||
// Give every deployed app a pin for what it is already running, so the render has something to
|
||
// obey. Runs immediately after the backfill so an app that pass has just observed can be pinned
|
||
// in the same boot. It reads and writes FILES only — no container is started, stopped or touched.
|
||
// An app it cannot pin confidently is left UNPINNED and keeps pre-v0.235.0 behaviour, loudly.
|
||
stackMgr.AdoptPins()
|
||
// v0.264.0 (R-646): record the pinned version's .felhom.yml for apps pinned before v0.263.2 that are
|
||
// current with the catalog — the probe their first undo will judge them by. Files only.
|
||
stackMgr.BackfillAppliedMeta()
|
||
|
||
// --- Start the catalog syncer, AFTER adoption ---
|
||
// ORDERING IS LOAD-BEARING. Start() fires an immediate sync in a goroutine. Started at its
|
||
// original place — before adoption — that first sync would run while every app was still
|
||
// unpinned, copy the catalog verbatim over a deployed app, and hand the next restart a version
|
||
// change: precisely the behaviour this release removes, once per boot. Adoption first means the
|
||
// very first sync of a boot already obeys the pins.
|
||
syncer.Start()
|
||
|
||
// --- R-52: boot desired-state reconciliation ---
|
||
// A deployed app that missed its boot start used to stay down until a human noticed (F5: immich
|
||
// and calibre-web sat Exited for ~18 h while ten siblings came back). One bounded start-once
|
||
// sweep, deliberately AFTER the quiesce recovery above so the two never race for the same stack,
|
||
// and entirely inside deadAppBootGrace so a successful recovery is silent and a failed one still
|
||
// alerts honestly. Never touches an app the customer stopped — see internal/bootrecon.
|
||
// R-171: hand the boot sweep the settings it needs to answer "is this app's drive live?" BEFORE
|
||
// the goroutine starts — an unwired gate is silently the pre-v0.190.0 behaviour that started apps
|
||
// onto absent drives. TestMainWiresBootDriveGate walks this file's AST for the assignment.
|
||
// --- Slice 4: guarded-update crash recovery (Scenario G) ---
|
||
// BEFORE the boot reconciler, and the order is load-bearing: an update interrupted after `up` is
|
||
// marked Updating here, and bootDriveGate refuses an Updating app — started after the sweep, the
|
||
// sweep could bring up a half-updated app with no record of why. Pin-backs for updates interrupted
|
||
// before anything ran happen here too. The resumed health wait itself is launched further down,
|
||
// once the backup side is wired, because a resumed update that fails must be able to HOLD.
|
||
if resumed := stackMgr.RecoverUpdates(); len(resumed) > 0 {
|
||
logger.Printf("[WARN] [update] %d interrupted update(s) will resume after the backup side is wired: %v", len(resumed), resumed)
|
||
}
|
||
|
||
bootDriveSettings = sett
|
||
bootQuiesceLoop = quiesceLoop
|
||
bootAppStopGuard = appStopGuard
|
||
bootStackMgr = stackMgr
|
||
go runBootReconcile(ctx, stackMgr, logger)
|
||
|
||
// --- Start CPU collector ---
|
||
cpuCollector := system.NewCPUCollector(5 * time.Second)
|
||
cpuCollector.Start(ctx)
|
||
defer cpuCollector.Stop()
|
||
|
||
// --- Initialize metrics store + collector ---
|
||
metricsDBPath := "/opt/docker/felhom-controller/data/metrics.db"
|
||
metricsStore, err := metrics.NewMetricsStore(metricsDBPath, logger)
|
||
if err != nil {
|
||
logger.Printf("[WARN] Failed to initialize metrics store: %v — monitoring disabled", err)
|
||
} else {
|
||
logger.Printf("[INFO] Metrics store opened at %s", metricsDBPath)
|
||
}
|
||
|
||
if metricsStore != nil {
|
||
defer metricsStore.Close()
|
||
metricsHDDPath := cfg.Paths.HDDPath
|
||
if p := sett.GetDefaultStoragePath(); p != "" {
|
||
metricsHDDPath = p
|
||
}
|
||
metricsCollector := metrics.NewMetricsCollector(metricsStore, cpuCollector, metricsHDDPath, logger)
|
||
metricsCollector.Start(ctx)
|
||
defer metricsCollector.Stop()
|
||
logger.Println("[INFO] Metrics collector started (60s interval)")
|
||
}
|
||
|
||
// Deprecation notice for ping UUIDs (Healthchecks pinging retired — the Hub
|
||
// now owns monitoring; disk-tier backup moved to the host agent in slice 8C).
|
||
uuids := cfg.Monitoring.PingUUIDs
|
||
if uuids.Heartbeat != "" || uuids.SystemHealth != "" || uuids.DBDump != "" || uuids.Backup != "" || uuids.BackupIntegrity != "" {
|
||
logger.Println("[INFO] Healthchecks ping UUIDs configured but no longer used — monitoring is now handled by the Hub")
|
||
}
|
||
|
||
// --- Initialize backup manager (app-data only: DB dumps + Docker-volume tars) ---
|
||
var backupMgr *backup.Manager
|
||
// R-550: the restore the last stop interrupted, found when the record loads; raised once the notifier exists.
|
||
var interruptedRestore *backup.RestoreOpResult
|
||
stackProv := &stackAdapter{
|
||
mgr: stackMgr,
|
||
getStoragePaths: func() []settings.StoragePath { return sett.GetStoragePaths() },
|
||
encKey: encKey,
|
||
}
|
||
if cfg.Backup.Enabled {
|
||
backupMgr = backup.NewManager(cfg, sett, logger)
|
||
// R-550: load the persisted restore record BEFORE any page can ask for it.
|
||
interruptedRestore = loadRestoreRecordAtStartup(backupMgr, cfg.Paths.DataDir, logger)
|
||
// R-166: use the guard that already ran Recover at startup, not a second one over the same
|
||
// file (see SetAppStopGuard — one file, one owner).
|
||
backupMgr.SetAppStopGuard(appStopGuard)
|
||
backupMgr.SetStackProvider(stackProv)
|
||
backupMgr.SetVersion(Version)
|
||
// O4: restore-from-unit generates a replacement for an unrecoverable RESETTABLE secret
|
||
// (data-keys stay fail-closed) so the app redeploys with a fresh credential, not a blank one.
|
||
backupMgr.SetSecretGenerator(stackMgr.GenerateSecretForField)
|
||
// R-7b: after a shares restore re-adds definitions to the registry, smb.conf must be
|
||
// re-rendered or the restored shares exist on paper but are not exported. A seam rather than a
|
||
// direct call — the backup package must not depend on the stacks package.
|
||
backupMgr.SetSharesReconciler(stackMgr.ReconcileSamba)
|
||
}
|
||
|
||
// --- Slice 4: the guarded update's backup side ---
|
||
// Wired unconditionally: with backup disabled the adapter answers "no restore point" and every
|
||
// update is refused with the no-backup sentence — the precondition cannot be met, which is true.
|
||
// An UNWIRED manager also refuses (fail closed); TestSlice4_UpdateGuardsAreWiredAtStartup walks
|
||
// this file for the call, because a seam built and never wired has shipped here seven times.
|
||
stackMgr.SetUpdateGuards(&updateGuardsAdapter{b: backupMgr, q: quiesceLoop})
|
||
// `09` §3 decision 36: a start-fresh moves the removed app's own unit in with its kept files.
|
||
stackMgr.SetKeptUnitFinder(backupMgr.RemovedUnitOnDrive)
|
||
// v0.271.0 (`09` §6.4 part 7): the automatic update leg. The switch is read at the leg's start and
|
||
// before every press; W is the SAME effective window every nightly leg reads. The files-may-change
|
||
// mark asks the backup side's truth table (decisions 25/26), never a second definition of "whole".
|
||
// Pinned by TestUpdateLegIsWiredAtStartup (an AST walk of this file).
|
||
stackMgr.SetUpdateLeg(stacks.UpdateLegOptions{
|
||
Enabled: sett.GetAppUpdateUnattended,
|
||
WindowStart: func() string {
|
||
return backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule)
|
||
},
|
||
FreshWholeCopy: func(ctx context.Context, name string) (bool, string) {
|
||
if backupMgr == nil {
|
||
return false, "backup is disabled on this box"
|
||
}
|
||
return backupMgr.FreshWholeCopy(ctx, name, cfg.Update.BackupMaxAgeDuration(), time.Now())
|
||
},
|
||
Location: budapestLoc(),
|
||
})
|
||
// v0.238.1: the nightly legs (capture, Tier 2, volume dump) leave an app alone WHILE it is being
|
||
// updated, not only once it is held — found live in Scenario F, see backup.Manager.isHeld.
|
||
if backupMgr != nil {
|
||
backupMgr.SetUpdatingCheck(stackMgr.IsUpdating)
|
||
// R-671 (v0.272.0): a restore that lifts an update hold removes the undo copies that hold kept.
|
||
backupMgr.SetUndoCopyRemover(stackMgr.RemoveUndoCopies)
|
||
}
|
||
if n := stackMgr.ResumeInterruptedUpdates(ctx); n > 0 {
|
||
logger.Printf("[WARN] [update] resumed %d interrupted update(s)", n)
|
||
}
|
||
|
||
// SLICE 2: the offsite apply-bridge is launched further down, AFTER the self-updater is constructed
|
||
// (R-71a: the bridge's settle-gate reads the updater's floor/update-running state to defer the
|
||
// consume past a managed day-0 floor-update). See "offsite apply-bridge" below.
|
||
|
||
// --- Wire the data-migration engine (B1) + backup↔migration mutual exclusion (Change 3) ---
|
||
stackMgr.SetMigrationDeps(sett, func() bool { return backupMgr != nil && backupMgr.IsRunning() })
|
||
if backupMgr != nil {
|
||
backupMgr.SetMigrationRunningCheck(stackMgr.IsMigrating)
|
||
}
|
||
// RecoverMigration is started AFTER the web server's done-hook is wired (below), so a resumed
|
||
// decommission-migration still finalizes the source decommission on completion.
|
||
|
||
// --- Initialize alert manager ---
|
||
alertMgr := web.NewAlertManager(logger)
|
||
|
||
// --- Initialize notifier ---
|
||
notifier := notify.New(cfg.Hub.URL, cfg.Hub.APIKey, cfg.Customer.ID, sett, logger, cfg.Logging.Level == "debug")
|
||
// R-550: an interrupted restore is raised once, now that the notifier exists.
|
||
reportInterruptedRestore(notifier, interruptedRestore)
|
||
|
||
// R-97a: wire the quiesce loop's hub-event seam. It MUST happen here and not in
|
||
// startQuiesceLoop, because the notifier is constructed after it — and it must happen at all,
|
||
// or the seam is inert (the "built but never wired" trap, four recorded instances). Reachability
|
||
// is covered by TestQuiesceTierNotifierIsWired in this package.
|
||
if quiesceLoop != nil {
|
||
quiesceLoop.SetTierNotifier(quiesceTierNotifier{n: notifier})
|
||
}
|
||
|
||
// v0.264.0 (`09` §3 decision 15, §6.4 part 2): an undone or held app update is TOLD — one household
|
||
// mail + an operator event each. The held sentence is the hold's OWN sentence, rendered by the backup
|
||
// side in whichever language is asked for, so the mail says exactly what the page says. Pinned by
|
||
// TestUpdateEventSinkIsWiredAtStartup — a seam built and never wired is this project's commonest
|
||
// defect, and this one would fail silently: no event, no mail, no error.
|
||
stackMgr.SetUpdateEventSink(func(ev stacks.UpdateEvent) {
|
||
switch ev.Kind {
|
||
case stacks.UpdateEventUndone:
|
||
notifier.NotifyAppUpdateUndone(ev.App, ev.From, ev.To, ev.At)
|
||
case stacks.UpdateEventHeld:
|
||
d := notify.AppUpdateDetails{App: ev.App, StackName: ev.App, From: ev.From, To: ev.To,
|
||
At: ev.At.UTC().Format(time.RFC3339), CopyTier: ev.CopyTier}
|
||
if !ev.CopyDate.IsZero() {
|
||
d.CopyDate = ev.CopyDate.UTC().Format(time.RFC3339)
|
||
}
|
||
if backupMgr != nil && ev.CopyTier > 0 {
|
||
d.CopyHolds = backupMgr.UpdateCopyHoldsKey(ev.App, ev.CopyTier) // R-647: the key, never the Hungarian phrase
|
||
}
|
||
var hold settings.RestoreHold
|
||
haveHold := false
|
||
if ev.HoldRecorded && backupMgr != nil {
|
||
hold, haveHold = backupMgr.UpdateHold(ev.App)
|
||
}
|
||
if haveHold { // R-659: the details name the copy the SENTENCE names, not the precondition's
|
||
d.CopyTier, d.CopyDate, d.CopyHolds = hold.CopyTier, hold.CopyDate, ""
|
||
if hold.CopyTier > 0 {
|
||
d.CopyHolds = backupMgr.UpdateCopyHoldsKey(ev.App, hold.CopyTier)
|
||
}
|
||
}
|
||
notifier.NotifyAppUpdateHeld(d, func(lang string) string {
|
||
if ev.HoldRecorded && backupMgr != nil {
|
||
if held, why := backupMgr.RestoreHoldForLang(ev.App, lang); held {
|
||
return why
|
||
}
|
||
}
|
||
return util.Text(lang, "update.error.hold_unsaved")
|
||
})
|
||
// R-659 (v0.268.0; operator ruling 2026-09-24, option A): no copy on this box brings the app
|
||
// back whole — the household was just told support is informed; this is that information.
|
||
if haveHold && hold.NoWholeCopy {
|
||
notifier.NotifyAppHoldNoWholeCopy(notify.AppHoldNoWholeCopyDetails{App: ev.App, StackName: ev.App,
|
||
From: ev.From, To: ev.To, At: ev.At.UTC().Format(time.RFC3339), CopiesSeen: hold.CopiesSeen, UndoState: hold.UndoState})
|
||
}
|
||
}
|
||
})
|
||
|
||
// R-166 §2.4: report an interrupted app-data operation to the operator, HERE, because the
|
||
// recovery itself had to run before the boot reconciler (line ~236) and the notifier does not
|
||
// exist until this line. An interrupted operation means the controller died mid-backup and that
|
||
// backup did not complete — operator-grade news even when every app came back.
|
||
//
|
||
// It rides the EXISTING `backup_failed` event type rather than a new one: a new type needs the
|
||
// hub's allowedEventTypes + customerMessages pair changed, which is a wire change, and this
|
||
// release ships no hub change. A controller emitting an unlisted type gets a flat 400 from
|
||
// POST /event. Reachability is covered by TestAppStopRecoveryIsWired.
|
||
// R-174: `Alarming()` and not merely `!= nil`. A recovery that ONLY refused starts is the drive
|
||
// gate working as designed, and `backup_failed` is customer-enabled by default — reporting a
|
||
// deliberate hold through it would email the customer "A biztonsági mentés sikertelen!" about an
|
||
// app nothing is wrong with. The refusal is already logged at WARN with its reason.
|
||
if appStopRecovery != nil && appStopRecovery.Alarming() {
|
||
notifier.NotifyBackupFailed(appStopRecovery.Message(), appStopRecovery.Detail())
|
||
} else if appStopRecovery != nil {
|
||
logger.Printf("[WARN] [appstop] %s — not alarming: %s", appStopRecovery.Message(), appStopRecovery.Detail())
|
||
}
|
||
|
||
// --- Initialize the app-email SMTP shim (mailrelay) ---
|
||
// In-process shim: apps → shim → hub → Resend (the Resend key stays hub-side). It runs only
|
||
// when the controller has a hub (URL+key) AND the operational kill-switch is on; the runtime
|
||
// global app-email toggle then decides whether it actually listens (Lifecycle.Apply). Listeners
|
||
// bind to the app Docker network only — never published to the host/internet.
|
||
var mailShim *mailrelay.Lifecycle
|
||
if cfg.Hub.URL != "" && cfg.Hub.APIKey != "" && cfg.MailRelay.HardEnabled() {
|
||
mailShim = mailrelay.NewLifecycle(mailrelay.Options{
|
||
PlainAddr: cfg.MailRelay.PlainListen,
|
||
TLSAddr: cfg.MailRelay.TLSListen,
|
||
PlainNoTLSAddr: cfg.MailRelay.PlainNoTLSListen,
|
||
ServiceName: cfg.MailRelay.ShimHost,
|
||
Policy: mailrelay.NewPolicy(cfg.MailRelay.FromDomains),
|
||
Forwarder: mailrelay.NewHubForwarder(cfg.Hub.URL, cfg.Hub.APIKey),
|
||
Logger: logger,
|
||
})
|
||
if err := mailShim.Apply(sett.AppEmailEnabled()); err != nil {
|
||
logger.Printf("[ERROR] [mailrelay] could not start app-email shim: %v", err)
|
||
}
|
||
} else {
|
||
logger.Printf("[INFO] [mailrelay] app-email shim unavailable (no hub configured or disabled in config)")
|
||
}
|
||
|
||
// --- Initialize self-updater ---
|
||
var updater *selfupdate.Updater
|
||
if cfg.SelfUpdate.Enabled {
|
||
// Phase 1: the host agent performs the container swap. Build a (nil-able) agent client from the
|
||
// provisioned local-API config; when absent (un-provisioned guest) self-update is unavailable.
|
||
var swapAgent selfupdate.AgentSwapper
|
||
if cfg.LocalAPI.Endpoint != "" && cfg.LocalAPI.Token != "" {
|
||
if ac, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint); err != nil {
|
||
logger.Printf("[WARN] Self-update: agent client init failed (%v) — updates unavailable", err)
|
||
} else {
|
||
swapAgent = ac
|
||
}
|
||
}
|
||
updater = selfupdate.NewUpdater(&cfg.SelfUpdate, &cfg.Git, Version, cfg.Paths.DataDir, swapAgent, logger, cfg.Logging.Level == "debug")
|
||
updater.SetBackupRunningCheck(func() bool {
|
||
return backupMgr != nil && backupMgr.IsRunning()
|
||
})
|
||
// v0.261.0 — the two-way lock between the controller's own update and a guarded app update.
|
||
// BOTH directions are wired here because this is the only place that holds both objects, and
|
||
// the dependency must not exist in either package (stacks never imports selfupdate).
|
||
//
|
||
// The app-update side is NOT redundant with the backup check above: the update's `backing-up`
|
||
// phase does take the backup single-flight, but `checking`, `safety-dump`, `pinning`,
|
||
// `pulling`, `starting` and `verifying` do not — and the last two are where the new version
|
||
// may already have touched the customer's data.
|
||
updater.SetAppUpdatingCheck(func() bool {
|
||
if stackMgr == nil {
|
||
return false
|
||
}
|
||
// v0.271.0: the whole automatic update leg counts, not only a step in flight — a controller
|
||
// swap between two steps would cancel the rest of the night's leg.
|
||
legActive, _ := stackMgr.UpdateLegState()
|
||
return stackMgr.AnyUpdating() || legActive
|
||
})
|
||
if stackMgr != nil {
|
||
stackMgr.SetSelfUpdatingCheck(updater.IsUpdateRunning)
|
||
}
|
||
// Check for post-update state (did a previous update succeed or fail?)
|
||
if state := updater.VerifyStartup(); state != nil {
|
||
notifier.NotifyControllerUpdated(state.PreviousVersion, state.TargetVersion, state.Status == "success")
|
||
}
|
||
logger.Printf("[INFO] Self-update enabled (check every %s, auto-update: %v, auto-update time: %s)",
|
||
cfg.SelfUpdate.CheckInterval, cfg.SelfUpdate.AutoUpdate, cfg.SelfUpdate.AutoUpdateTime)
|
||
}
|
||
|
||
// SLICE 2: the offsite apply-bridge — on startup (async, non-blocking) reconcile the hub-served
|
||
// offsite descriptor into a configured key-only offbox target (fail-safe, idempotent, no blind
|
||
// TOFU). The config_refresh self-restart re-runs this after a descriptor change (new process →
|
||
// startup). Launched HERE (after the self-updater is built) so the R-71a settle-gate can read the
|
||
// updater's floor/update-running state and defer the one-time-password consume past a managed
|
||
// day-0 floor-update (the F10 race). The gate is wired ONLY when an updater exists — with no update
|
||
// mechanism there is no floor-update to race, so the bridge reconciles immediately (Settle nil).
|
||
// offsiteBridge is hoisted out of the block below ONLY so the recovery screen can drive one
|
||
// reconcile between placing a recovered key and reading the repository (R-219). nil when off-site
|
||
// is not configured for this customer, which the web seam treats as "skip".
|
||
var offsiteBridge *offsiteapply.Bridge
|
||
offsiteRegistrar := offsiteapply.HubRegistrar{HubURL: cfg.Hub.URL, CustomerID: cfg.Customer.ID, APIKey: cfg.Hub.APIKey}
|
||
if backupMgr != nil && cfg.Hub.URL != "" && cfg.Hub.APIKey != "" {
|
||
// The orphan move-aside is the HUB's now (decision 69): the box's pinned key reaches only the
|
||
// append-only rclone server and cannot rename a directory.
|
||
backupMgr.SetOffsiteMoveAside(offsiteRegistrar.MoveAside)
|
||
// Decision 68: retention on the append-only tier happens only inside a hub-opened window.
|
||
backupMgr.SetOffsiteWindowClient(offsiteapply.HubWindowClient{Registrar: offsiteRegistrar})
|
||
// Decision 74: a due set-aside deletion is the hub's (after its own 7-day delay).
|
||
backupMgr.SetOffsiteAbandonClient(offsiteapply.HubAbandonClient{Registrar: offsiteRegistrar})
|
||
}
|
||
if backupMgr != nil && cfg.Offsite.Enabled && cfg.Hub.URL != "" && cfg.Hub.APIKey != "" {
|
||
bridge := &offsiteapply.Bridge{
|
||
Cfg: cfg,
|
||
// Decision 69 (v0.289.0): the box sends its PUBLIC key to the hub's registrar, which pins it
|
||
// append-only; the box never receives the sub-account password.
|
||
Registrar: offsiteRegistrar,
|
||
Scanner: offsiteapply.KeyscanScanner{},
|
||
KeyGen: offsiteapply.ED25519KeyGen{},
|
||
Prober: offsiteapply.PinnedProber{},
|
||
Existing: func() string {
|
||
b, err := os.ReadFile(filepath.Join(cfg.Paths.DataDir, "offbox", "ssh_key"))
|
||
if err != nil {
|
||
return ""
|
||
}
|
||
return string(b)
|
||
},
|
||
Enabler: offsiteapply.EnablerFunc(func(ctx context.Context, host, user string, port int, repoPath, priv, kh string, quotaGB int) error {
|
||
tgt := &settings.OffboxTarget{Enabled: true, Host: host, User: user, Port: port, RepoPath: repoPath, Schedule: "daily", QuotaGB: quotaGB}
|
||
stage := func(ctx context.Context, pw string) error {
|
||
ac, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
return ac.StageEscrowSecret(ctx, pw)
|
||
}
|
||
return backupMgr.ApplyOffsiteTarget(ctx, tgt, priv, kh, stage)
|
||
}),
|
||
MarkerPath: filepath.Join(cfg.Paths.DataDir, "offbox", "applied_marker"),
|
||
Logger: logger,
|
||
}
|
||
if updater != nil {
|
||
// Thin adapter over the updater's OWN knowledge (StackDataProvider pattern) — the bridge
|
||
// never fetches the floor a second way. floorKnown = the floor has been learned from a
|
||
// report ACK yet (GetFloor() != "").
|
||
u := updater
|
||
bridge.Settle = offsiteapply.SettleFunc(func() (string, string, bool, bool) {
|
||
floor := u.GetFloor()
|
||
return Version, floor, u.IsUpdateRunning(), floor != ""
|
||
})
|
||
}
|
||
offsiteBridge = bridge
|
||
go func() {
|
||
// ReconcileWhenSettled runs the settle-gate FIRST (its own bounds), then Reconcile under a
|
||
// fresh 3-minute context — the gate's wait never eats the reconcile budget.
|
||
if err := bridge.ReconcileWhenSettled(context.Background()); err != nil {
|
||
logger.Printf("[WARN] [offsite-apply] reconcile: %v (retries on next config refresh/restart)", err)
|
||
}
|
||
}()
|
||
}
|
||
|
||
// --- Initialize scheduler ---
|
||
sched := scheduler.New(logger)
|
||
sched.SetDebug(cfg.Logging.Level == "debug")
|
||
|
||
// Existing periodic tasks (migrated from ad-hoc goroutines)
|
||
// F7: 30s left the dashboard list lagging Docker health by up to ~30s after a deploy/state change.
|
||
// 10s (matching the health-probes cadence) tightens it; RefreshStatus is a cheap `docker ps`-based
|
||
// refresh of the in-memory map, so 10s does not meaningfully load Docker. (The deploy page itself
|
||
// already polls per-stack every 3s; this is for the dashboard/stacks list.)
|
||
sched.Every("status-refresh", 10*time.Second, func(ctx context.Context) error {
|
||
return stackMgr.RefreshStatus()
|
||
})
|
||
sched.Every("stack-scan", 2*time.Minute, func(ctx context.Context) error {
|
||
return stackMgr.ScanStacks()
|
||
})
|
||
sched.Every("health-probes", 10*time.Second, func(ctx context.Context) error {
|
||
return stackMgr.RunHealthProbes()
|
||
})
|
||
|
||
// System health check — refreshes dashboard alerts and notifies on changes.
|
||
// Healthchecks.io pinging has been retired (the Hub now owns monitoring).
|
||
healthInterval, err := time.ParseDuration(cfg.Monitoring.SystemHealthInterval)
|
||
if err != nil {
|
||
healthInterval = 5 * time.Minute
|
||
}
|
||
sched.Every("system-health", healthInterval, func(ctx context.Context) error {
|
||
healthReport := monitor.RunHealthCheck(cfg, cpuCollector, sett.GetStoragePaths(), sett.GetSMBSettings(), logger)
|
||
// Self-heal the base stack: call unconditionally every tick. EnsureBaseStack is single-flight
|
||
// + idempotent (skips running stacks ⇒ a cheap 3× docker-inspect no-op when healthy), so there
|
||
// is no need to couple to the health-report issue strings. Runs in a goroutine — never blocks
|
||
// or fails the health job.
|
||
go func() {
|
||
if err := stackMgr.EnsureBaseStack(); err != nil {
|
||
logger.Printf("[WARN] [infra] self-heal base-stack bring-up: %v", err)
|
||
}
|
||
}()
|
||
// Refresh dashboard alerts from health report
|
||
updateAvailable := false
|
||
latestVersion := ""
|
||
if updater != nil {
|
||
status := updater.GetStatus()
|
||
if status.LastCheck != nil {
|
||
updateAvailable = status.LastCheck.UpdateAvailable
|
||
latestVersion = status.LastCheck.LatestVersion
|
||
}
|
||
}
|
||
alertMgr.Refresh(healthReport, cfg, backupMgr, updateAvailable, latestVersion, sett.GetStoragePaths())
|
||
// Notify on health status changes
|
||
notifier.NotifyHealthChange(healthReport.Status, healthReport.Issues, healthReport.Warnings)
|
||
return nil
|
||
})
|
||
|
||
// fix-3 (CAMPAIGN-3): a deployed app that is not running must be LOUD, not silent (the campaign's
|
||
// CWA sat dead 4 h; F11 then produced 4 silently-dead NAS apps per reboot). Runs on its own short
|
||
// cadence (faster than the 5-min health cycle) so a dead app surfaces quickly. Logs nothing on the
|
||
// quiet path (no ring spam — fix-6 friendly); the dashboard banner is state-based (self-clears) and
|
||
// the hub event fires once per running→down transition (notifier tracks it). A boot grace skips the
|
||
// controller's own startup settle so apps that legitimately take 30–60 s to come up don't alert.
|
||
//
|
||
// F-OBS (Campaign 8): this check had NO positive observable at default `info` level. Its per-cycle
|
||
// scheduler line is emitted through Scheduler.dbg(), which is gated on logging.level==debug and so
|
||
// is never PRODUCED on a default box (not merely filtered — the always-DEBUG ring cannot capture
|
||
// it either), and a 30 s interval also puts it on the scheduler's quiet path. So on every customer
|
||
// box, "no alarms" was indistinguishable from "the detector never ran" — the precise fallacy this
|
||
// project now has a standing rule against, and it directly undermines confidence in the F-CRIT-1
|
||
// fix in the field.
|
||
//
|
||
// The observable is a PERIODIC SUMMARY, not a line per run: at 30 s a per-run line is 2880
|
||
// lines/day of pure noise, which is what made the original author choose silence. Every
|
||
// deadAppHeartbeatEvery-th scan emits one INFO carrying the scan count and what it found, so an
|
||
// operator can always answer "is it running, and what does it see?" from a default box — and a
|
||
// STALLED detector is visible as the heartbeat stopping.
|
||
sched.Every("deadapp-check", 30*time.Second, func(ctx context.Context) error {
|
||
if time.Since(startTime) < deadAppBootGrace {
|
||
return nil // still inside the startup settle window
|
||
}
|
||
dead, states := scanDeployedAppRunStates(stackMgr, quiesceLoop, appStopGuard, backupMgr)
|
||
alertMgr.SetDeadAppAlerts(dead)
|
||
notifier.NotifyAppStartFailures(states)
|
||
// R-514: a worker OOM-killed inside a running container leaves the app „Fut". Surface it.
|
||
ooms, oerr := stackMgr.ScanOOMKilled()
|
||
if oerr != nil {
|
||
logger.Printf("[WARN] [deadapp] OOM scan failed: %v", oerr)
|
||
} else {
|
||
for _, o := range ooms {
|
||
logger.Printf("[WARN] [deadapp] %s: container %s was OOM-killed (started %s; kills %d, limit %s, peak %s)", o.Stack, o.Container, o.StartedAt, o.Kills, o.MemLimit, o.Peak)
|
||
notifier.NotifyAppOOM(o.Stack, o.Container, o.StartedAt, o.Kills, o.MemLimit, o.Peak)
|
||
}
|
||
}
|
||
// v0.269.0 (`09` §3 decision 28, R-667): a crash loop or an OOM storm is STOPPED by the box.
|
||
stopUnhealthyApps(logger, unhealthyDeps{
|
||
verdicts: stackMgr.ObserveUnhealthy,
|
||
suppressed: func() map[string]bool {
|
||
return unionSuppressed(unionSuppressed(quiesceLoop.SuppressedStacks(), appStopGuard.SuppressedStacks()), stackMgr.UpdatingStacks())
|
||
},
|
||
busy: func(name string) (bool, string) {
|
||
if backupMgr == nil {
|
||
return false, ""
|
||
}
|
||
return backupMgr.UpdateBusy(name)
|
||
},
|
||
holdKind: func(name string) string {
|
||
if backupMgr == nil {
|
||
return ""
|
||
}
|
||
return backupMgr.HoldKind(name)
|
||
},
|
||
stop: stackMgr.StopStack,
|
||
hold: func(name, kind string, at time.Time) (int, error) {
|
||
if backupMgr == nil {
|
||
return 0, fmt.Errorf("backup is not enabled on this box — the stop cannot be held")
|
||
}
|
||
return backupMgr.HoldUnhealthy(name, kind, at)
|
||
},
|
||
notify: func(v stacks.UnhealthyVerdict, trip int) {
|
||
notifier.NotifyAppStoppedUnhealthy(v.Stack, v.Kind, v.Count, v.Window, trip)
|
||
},
|
||
}, ooms, time.Now())
|
||
deadAppScans++
|
||
noteDeadAppScan(logger, deadAppScans, len(states), len(dead))
|
||
noteUnknownIntentSuppressions(logger, states)
|
||
return nil
|
||
})
|
||
|
||
// fix-6 (CAMPAIGN-3): periodically spill the debug ring to the SSD state dir so a HARD crash loses
|
||
// at most ~60 s of the window (a clean shutdown spills too). ≤30s interval → auto-quiet (no
|
||
// per-cycle scheduler line). SpillTo is atomic (tmp+rename) so it can never corrupt the ring file.
|
||
sched.Every("ring-spill", 30*time.Second, func(ctx context.Context) error {
|
||
return logBuffer.SpillTo(ringSpillPath)
|
||
})
|
||
|
||
// --- Central hub pusher (declared early so backup closure can reference it) ---
|
||
var hubPusher *report.Pusher
|
||
// escrowConfirmer is hoisted so the web server (built later) can read its Scenario-F
|
||
// stale-blob flag (SetEscrowStale below). nil when no hub is configured.
|
||
var escrowConfirmer *report.EscrowAutoConfirmer
|
||
if cfg.Hub.URL != "" && cfg.Hub.APIKey != "" {
|
||
hubPusher = report.NewPusher(&cfg.Hub, logger, cfg.Logging.Level == "debug")
|
||
// SLICE 3 — hub-verified escrow auto-confirm (long-lived: the mismatch warn dedupes per hash,
|
||
// not per 15-min cycle). Flips offbox pending→escrowed ONLY when the hub-recorded hash of the
|
||
// escrowed password matches the local repo password's hash; never un-confirms. v0.127.0 adds
|
||
// the escrowed-state STALE re-check (Scenario F — warn + card flag, never a state change).
|
||
escrowConfirmer = &report.EscrowAutoConfirmer{
|
||
Pending: func() bool {
|
||
return backupMgr != nil && backupMgr.OffboxConfigured() &&
|
||
sett.GetOffboxTarget() != nil && sett.GetOffboxTarget().EscrowState == "pending"
|
||
},
|
||
Escrowed: func() bool {
|
||
return backupMgr != nil && backupMgr.OffboxConfigured() &&
|
||
sett.GetOffboxTarget() != nil && sett.GetOffboxTarget().EscrowState == "escrowed"
|
||
},
|
||
LocalHash: func() (string, bool) {
|
||
if backupMgr == nil {
|
||
return "", false
|
||
}
|
||
return backupMgr.OffboxRepoPasswordHash()
|
||
},
|
||
Flip: func() error {
|
||
return sett.UpdateOffboxStatus(func(o *settings.OffboxTarget) {
|
||
o.EscrowState = "escrowed"
|
||
o.CeremonyCompletedAt = "" // v0.138.0: clear the awaiting-card stamp on confirm
|
||
})
|
||
},
|
||
Wipe: func(ctx context.Context) error {
|
||
ac, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
return ac.WipeStagedEscrowSecret(ctx)
|
||
},
|
||
// v0.199.0 (R-204 item 4 / R-193): remember whether the HUB holds a sealed recovery
|
||
// package for us — recorded on EVERY ACK, including when no off-site target exists, which
|
||
// is precisely the case that needs it. Half of the stranded-rebuild predicate in
|
||
// backup.needsOffsiteCredential; the other half (no repository password) is local.
|
||
RecordPresence: sett.SetHubEscrowIdentityPresent,
|
||
// v0.201.0 (R-222): whether the hub is ALSO keeping an EARLIER sealed package, so the
|
||
// recovery screen can name that situation instead of blaming the customer's typing.
|
||
// v0.206.0 (R-241): the superseded fact ALSO closes out an abandonment. When the hub stops
|
||
// reporting a superseded package, the sealed package that protected the deleted set-aside
|
||
// history is gone too — both halves are away and the question is finished (Scenario F).
|
||
// Chained onto the existing recorder rather than added as a second ACK consumer, for the
|
||
// reason RecordPresence's own comment gives.
|
||
RecordSuperseded: func(present bool, at string) error {
|
||
if backupMgr != nil {
|
||
backupMgr.ClearAbandonPurgeIfConfirmed(present)
|
||
}
|
||
return sett.SetHubEscrowSuperseded(present, at)
|
||
},
|
||
// v0.206.0 (R-241): the hash the hub's package COVERS. This is the fact the recovery
|
||
// screen's shape (c) reads, and the comparison against it was already being computed here
|
||
// on every ACK and discarded. Wired at the same point as the two above, deliberately —
|
||
// a second wiring site in this file is how six features got built and never reached.
|
||
RecordEscrowKeyHash: sett.SetHubEscrowKeySHA256,
|
||
Logger: logger,
|
||
}
|
||
// Wire hub verification: update settings when hub reports customer status
|
||
hubPusher.OnPushResponse = func(resp *report.PushResponse) {
|
||
if resp.CustomerBlocked {
|
||
sett.SetHubVerified(false, time.Now())
|
||
logger.Printf("[WARN] Customer blocked on Hub — new deployments may be restricted")
|
||
} else {
|
||
sett.SetHubVerified(true, time.Now())
|
||
}
|
||
// Phase 2 managed updates: the ACK carries the operator-enforced minimum version (FLOOR).
|
||
// Hand it to the updater and reconcile — if the box is below the floor it auto-updates to
|
||
// the floor (reusing the Phase 1 swap). This rides the existing report cycle; no new timer.
|
||
// latest_version stays informational (the customer's opt-in button), NOT the auto-target.
|
||
if updater != nil {
|
||
updater.SetFloor(resp.MinControllerVersion)
|
||
updater.MaybeAutoUpdate()
|
||
}
|
||
// Config-refresh (v0.26.0): the ACK also carries the per-customer config_version. On a
|
||
// change vs. the last-applied version, re-pull controller.yaml (re-merging local_api from
|
||
// bootstrap.json) and self-restart so the new config loads. Pull-based — the hub never
|
||
// connects into the box. Rides this same report cycle; no new timer, no agent. First-ever
|
||
// ACK records the baseline without restarting; a failed pull keeps the current config and
|
||
// retries next cycle; an unchanged version is a no-op (so no restart storm).
|
||
cr := &report.ConfigRefresher{
|
||
Applied: sett.GetAppliedConfigVersion,
|
||
Record: sett.SetAppliedConfigVersion,
|
||
Refresh: func() error { return bootstrap.RefreshConfig(*configPath, logger, pull) },
|
||
Restart: func() { api.GracefulSelfRestart(logger) },
|
||
Logger: logger,
|
||
}
|
||
cr.Reconcile(resp.ConfigVersion)
|
||
// SLICE 3: run the escrow auto-confirm on the same ACK (after the config refresh decision —
|
||
// a refresh-restart re-enters here anyway on the next cycle).
|
||
escrowConfirmer.Reconcile(resp.Escrow)
|
||
// v0.111.0: operator log-tail requests ride the same ACK (pull pattern — the hub
|
||
// never reaches in). The NEXT report cycle collects + ships the tails; the hub
|
||
// clears its pending request on receipt. An empty list clears any stale local set.
|
||
report.SetPendingLogTails(resp.LogTailRequests)
|
||
// v0.116.0: the controller's OWN ring, same pattern (selftail.go).
|
||
report.SetPendingControllerLog(resp.ControllerLogRequested)
|
||
// v0.122.0 (F-4): cache the hub-delivered claim-code state (idempotent by
|
||
// generation). The web gate reads it on the next request — no restart needed.
|
||
claimSync := &report.ClaimSync{Settings: sett, Logger: logger}
|
||
claimSync.Reconcile(resp.Claim)
|
||
}
|
||
// Wire hub push status into alert manager for dashboard alerts
|
||
alertMgr.SetHubPushStatus(func() web.HubPushStatusData {
|
||
s := hubPusher.GetStatus()
|
||
return web.HubPushStatusData{
|
||
LastAttempt: s.LastAttempt,
|
||
LastSuccess: s.LastSuccess,
|
||
LastError: s.LastError,
|
||
Consecutive: s.Consecutive,
|
||
}
|
||
})
|
||
}
|
||
|
||
// Backup daily jobs
|
||
if cfg.Backup.Enabled && backupMgr != nil {
|
||
// v0.168.0: ONE customer setting (the window start W) drives all three nightly legs at fixed
|
||
// offsets — db-dump at W, tier-2 at W+60m, off-box at W+105m — so they can never be misordered.
|
||
// The window resolves settings > controller.yaml > "02:30"; a UI save fans out via UpdateDaily.
|
||
win := backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule)
|
||
dbLeg, tier2Leg, offboxLeg := backupwindow.LegTimes(win)
|
||
|
||
// App-data backup: daily database dumps. Disk-tier (restic snapshots,
|
||
// cross-drive, integrity check, infra backup) has moved to the host agent.
|
||
sched.Daily("db-dump", dbLeg, func(ctx context.Context) error {
|
||
err := backupMgr.RunDBDumps(ctx)
|
||
if err != nil {
|
||
notifier.NotifyDBDumpFailed("Adatbázis mentés sikertelen", err.Error())
|
||
} else {
|
||
notifier.NotifyDBDumpCompleted(notify.DBDumpDetails{})
|
||
}
|
||
return err
|
||
})
|
||
|
||
// ── R-218, THE CONSUME HALF (v0.203.0) ────────────────────────────────────────────────
|
||
//
|
||
// The declaration half shipped in v0.201.0 and works: a stranded box says
|
||
// `offsite.state=needs_credential`, and the hub's `offsiteheal` re-stages the one-time secret
|
||
// and logs *"the box re-consumes on its next cycle"*.
|
||
//
|
||
// **THERE WAS NO NEXT CYCLE.** `Reconcile` ran exactly twice in a process's life: once at
|
||
// start-up (the goroutine above, whose own comment says "retries on next config refresh/
|
||
// restart") and once when the recovery screen drives it (R-219). Both fire BEFORE the hub has
|
||
// anything staged, because the hub only stages in response to the declaration those runs
|
||
// precede.
|
||
//
|
||
// Measured on the R-201 re-walk, 2026-08-06: unlock reconcile 11:43:07 · hub staged 11:44:57
|
||
// saying "next cycle" · a full report cycle ran 11:55:46 · still unconsumed at 12:06. A guest
|
||
// command line moved it in **18 seconds**, which proves the credential, the target and the key
|
||
// were all fine and only the trigger was missing.
|
||
//
|
||
// So: re-run the SAME reconcile, on a tick, for exactly as long as the box itself says it needs
|
||
// a credential. The signal is the box's own `OffboxReportStatus().State` — the very statement
|
||
// the hub acts on, so the two can never disagree about whether a retry is wanted.
|
||
//
|
||
// WHY A POLL AND NOT AN ACK FLAG: the customer-facing text promises "amint megvannak" (no
|
||
// deadline) and the backups card promises "within a day"; a 5-minute tick is inside both by a
|
||
// wide margin, and it needs no hub change. If either promise ever tightens to minutes, revisit.
|
||
//
|
||
// IT STOPS BY CONSTRUCTION: the instant a target exists `OffboxReportStatus` stops returning
|
||
// the declared state, so a healthy box does no work and makes no noise. The settle gate is
|
||
// deliberately kept — `ReconcileWhenSettled` waits for floor knowledge exactly as at start-up.
|
||
sched.Every("offsite-credential-retry", 5*time.Minute, func(ctx context.Context) error {
|
||
if offsiteBridge == nil {
|
||
return nil // off-site not configured for this customer
|
||
}
|
||
attempted, err := offsiteBridge.RetryIfDeclared(ctx, func() bool {
|
||
st := backupMgr.OffboxReportStatus()
|
||
return st != nil && st.State == backup.OffsiteStateNeedsCredential
|
||
})
|
||
if !attempted {
|
||
return nil // a target exists (or we never needed one) — no work, and no log line
|
||
}
|
||
if err != nil {
|
||
// NOT a job failure: the box still declares, so the next tick tries again. Logged at
|
||
// WARN because a credential that never arrives is exactly what this exists to surface.
|
||
logger.Printf("[WARN] [offsite-apply] credential retry: %v (the box still declares a need; retrying)", err)
|
||
return nil
|
||
}
|
||
logger.Printf("[INFO] [offsite-apply] credential retry: the staged credential was collected and the tier applied")
|
||
return nil
|
||
})
|
||
|
||
// Cache refresh: every 5 minutes. Recompute the effective window each pass so the cached
|
||
// "next DB dump" follows a runtime window change (the UI save also refreshes immediately).
|
||
sched.Every("backup-cache", 5*time.Minute, func(ctx context.Context) error {
|
||
curDB, _, _ := backupwindow.LegTimes(backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule))
|
||
backupMgr.RefreshCache(scheduler.NextDailyRun(curDB))
|
||
return nil
|
||
})
|
||
|
||
// v0.273.0 (`09` §6.4 part 10, B5): a PostgreSQL conversion keeps the OLD datadir's copy until a
|
||
// backup of the converted app is proven; this releases it. Hourly — a copy outliving its backup by
|
||
// an hour costs disk, never data. Pinned by TestConvert_ReleaseIsWiredAtStartup.
|
||
sched.Every("conversion-copy-release", 1*time.Hour, func(ctx context.Context) error {
|
||
stackMgr.ReleaseConversionCopies(ctx)
|
||
return nil
|
||
})
|
||
|
||
// Tier 2: off-drive copy of each HDD app's recovery unit + userdata (auto-enabled, auto-target).
|
||
// Runs after the DB dump so it copies a fresh unit.
|
||
backupMgr.SetTier2Notifier(func(stackName, destLabel string, dur time.Duration, err error) {
|
||
// DISPLAY BOUNDARY (R-7b): this event's StackName lands in Hungarian operator/customer copy
|
||
// ("Másodlagos mentés elkészült: <név>"), so the reserved shares key is mapped here. Caught
|
||
// by live validation — the first demo run pushed a literal "_shares" before this line.
|
||
display := backup.DisplayStackName(stackName)
|
||
if err != nil {
|
||
notifier.NotifyCrossDriveFailed(notify.CrossDriveDetails{
|
||
StackName: display, Method: "rsync", DestPath: destLabel,
|
||
Duration: dur.Round(time.Second).String(), Error: err.Error(),
|
||
})
|
||
} else {
|
||
notifier.NotifyCrossDriveCompleted(notify.CrossDriveDetails{
|
||
StackName: display, Method: "rsync", DestPath: destLabel,
|
||
Duration: dur.Round(time.Second).String(),
|
||
})
|
||
}
|
||
})
|
||
sched.Daily("tier2-backup", tier2Leg, func(ctx context.Context) error {
|
||
backupMgr.RunAllTier2()
|
||
return nil
|
||
})
|
||
|
||
// Off-box (NAS) restic-SFTP backup (Part B): the off-site leg. A failure (incl. a fail-fast dead-NAS
|
||
// error) alerts the operator via the allowlisted backup_failed event. Daily after Tier 2.
|
||
backupMgr.SetOffboxNotify(func(dur time.Duration, snapshots int, err error) {
|
||
if err != nil {
|
||
// F-DIAG: a distinct message per cause, and the detail SANITISED — a raw err.Error()
|
||
// carries the sftp:user@host:/path repo reference off the box.
|
||
notifier.NotifyBackupFailed("Off-box (NAS) mentés sikertelen",
|
||
backupMgr.OffsiteFailureMessage(err, dur))
|
||
}
|
||
})
|
||
// R-203 (operator half): a completed off-site run could NOT capture a directory an app declares
|
||
// MANDATORY. It is not a failed run — what was captured is real and the snapshot count records
|
||
// it — but the app is NOT fully protected, and until v0.197.0 that reached nobody: the run said
|
||
// ok, the card said ok, the hub said ok, and one WARN line inside the container said otherwise.
|
||
//
|
||
// It rides the EXISTING per-run digest (backup_run_failures) rather than a new event type: a
|
||
// new type is a two-repo change (the hub's allowedEventTypes drops anything unlisted), and the
|
||
// digest is already operator-only and already means "this run did not fully do its job".
|
||
// Part D (v0.279.0): a RUNNING app whose newest copy holds no database dump and no volume tar. Rides
|
||
// the same operator-only digest as R-203 (no new event type — the hub drops unlisted ones), once per
|
||
// app per tier per day (the backup package dedupes).
|
||
backupMgr.SetHollowCopyNotify(func(app, tier string) {
|
||
d := notify.BackupRunFailuresDetails{RunKind: "data-check", Attempted: 1, Failed: 1,
|
||
Apps: []notify.RunFailureDetail{{App: app, Leg: "hollow-" + tier,
|
||
Reason: "the app is running but its newest " + tier + " copy holds no database dump and no volume tar"}}}
|
||
notifier.NotifyBackupRunFailures(fmt.Sprintf(
|
||
"Backup holds NO DATA: %s is running, but its newest %s copy carries no database dump and no volume tar — "+
|
||
"the app is not protected on that tier. Check for a hold on a running app (R-704's shape) or a skipped dump.",
|
||
app, tier), d)
|
||
})
|
||
backupMgr.SetOffboxGapNotify(func(gaps map[string][]string) {
|
||
apps := make([]string, 0, len(gaps))
|
||
for app := range gaps {
|
||
apps = append(apps, app)
|
||
}
|
||
sort.Strings(apps) // deterministic message + details
|
||
d := notify.BackupRunFailuresDetails{RunKind: "offsite", Attempted: len(apps), Failed: len(apps)}
|
||
var parts []string
|
||
for _, app := range apps {
|
||
sort.Strings(gaps[app])
|
||
reason := "declared MANDATORY data directories not captured: " + strings.Join(gaps[app], ", ")
|
||
d.Apps = append(d.Apps, notify.RunFailureDetail{App: app, Leg: "offsite-userdata", Reason: reason})
|
||
parts = append(parts, fmt.Sprintf("%s (%s)", app, strings.Join(gaps[app], ", ")))
|
||
}
|
||
notifier.NotifyBackupRunFailures(fmt.Sprintf(
|
||
"Offsite backup INCOMPLETE: %d app(s) had a directory they declare MANDATORY missing from the snapshot — %s. "+
|
||
"The run itself succeeded and what it captured is real, but these apps are NOT fully protected off-site. "+
|
||
"Check that the declared path exists on disk and that the app's data root resolves where the backup looks (R-203).",
|
||
len(apps), strings.Join(parts, "; ")), d)
|
||
})
|
||
// R-158 / R-167 (D-c, operator half): a per-app Tier-1 recovery-unit capture failed. Until
|
||
// v0.191.0 this was a `[WARN]` line and nothing else — /backups/apps is the page you open to
|
||
// ask whether ONE app is backed up, and it was the one page that never said. Fires per app;
|
||
// the capture loop continues, so three failing apps produce three events and one failing app
|
||
// does not silence its siblings.
|
||
//
|
||
// OPERATOR-TIER, not backup_failed — a customer can take no action on a capture failure, and
|
||
// backup_failed is customer-enabled by default. The space figures ride along because the
|
||
// overwhelmingly likely cause is a full filesystem and they answer "why" without a login.
|
||
// No controller-side cooldown: the hub owns it.
|
||
backupMgr.SetUnitNotify(func(stackName string, err error, usage *backup.UnitSpace) {
|
||
d := notify.RecoveryUnitFailureDetails{App: stackName, Error: err.Error()}
|
||
if usage != nil {
|
||
d.TargetPath, d.UsedGB, d.AvailGB = usage.Path, usage.UsedGB, usage.AvailGB
|
||
d.TotalGB, d.UsedPercent, d.SpaceKnown = usage.TotalGB, usage.UsedPercent, true
|
||
}
|
||
notifier.NotifyRecoveryUnitCaptureFailed(fmt.Sprintf(
|
||
"Recovery unit capture FAILED for %q — the app has no fresh local (Tier-1) backup, and Tier-2/Tier-3 have nothing to copy. %s. Error: %v",
|
||
stackName, usage.String(), err), d)
|
||
})
|
||
// R-182: the ONE operator digest per backup run. The per-app events above are the RECORD
|
||
// (the hub routes them record-only); this is the NOTIFICATION, sent once at the end of a run
|
||
// and only when something actually failed. A clean run sends nothing — and that silence is
|
||
// safe because the hub's own daily deadline check raises expected_backup_missed from report
|
||
// freshness, independently of any mail this box chooses to send.
|
||
backupMgr.SetRunSummaryNotify(func(rs backup.RunSummary) {
|
||
d := notify.BackupRunFailuresDetails{
|
||
RunID: rs.RunID, RunKind: rs.RunKind,
|
||
Failed: rs.Failed, Attempted: rs.Attempted,
|
||
}
|
||
for _, a := range rs.Apps {
|
||
d.Apps = append(d.Apps, notify.RunFailureDetail{App: a.App, Leg: a.Leg, Reason: a.Reason})
|
||
}
|
||
if rs.Usage != nil {
|
||
d.TargetPath, d.UsedGB, d.AvailGB = rs.Usage.Path, rs.Usage.UsedGB, rs.Usage.AvailGB
|
||
d.TotalGB, d.UsedPercent, d.SpaceKnown = rs.Usage.TotalGB, rs.Usage.UsedPercent, true
|
||
}
|
||
notifier.NotifyBackupRunFailures(rs.Message, d)
|
||
})
|
||
// R-379/R-380: an app HELD after a failed database restore whose rollback also failed.
|
||
//
|
||
// ROUTED THROUGH `backup_run_failures`, and the reuse is deliberate but imperfect — say so
|
||
// rather than let a future reader assume it was the natural fit. That type is operator-only
|
||
// (`notify.operatorOnlyEvents`), carries severity `error` (a VALID token — `warn` would
|
||
// coerce to `info` and email nobody, which is R-329 and still live elsewhere), and its
|
||
// payload is exactly {app, leg, reason}. What it is NOT is a restore event: it says "backup
|
||
// run". **No operator-facing RESTORE event type exists**, and minting one is a two-repo
|
||
// change with a hub deploy, which this task's baselines hold fixed. The leg is named
|
||
// `restore-hold` so the record is unambiguous to whoever reads it.
|
||
//
|
||
// DELIBERATELY NOT `backup_failed`: customer-enabled by default, Hungarian "A biztonsági
|
||
// mentés sikertelen!", and this is a deliberate hold rather than a backup failure. R-171 one
|
||
// path over.
|
||
backupMgr.SetRestoreHoldNotify(func(stack string, replayErr, rollbackErr error) {
|
||
d := notify.BackupRunFailuresDetails{RunKind: "offsite-reconstitute", Failed: 1, Attempted: 1}
|
||
reason := "rollback failed"
|
||
if rollbackErr != nil {
|
||
reason = rollbackErr.Error()
|
||
}
|
||
d.Apps = append(d.Apps, notify.RunFailureDetail{App: stack, Leg: "restore-hold", Reason: reason})
|
||
msg := fmt.Sprintf("App %q is HELD STOPPED: its off-site database restore failed AND the rollback to the customer's own pre-restore copy also failed. "+
|
||
"The app will not start from any path until the hold is cleared. Replay error: %v. Rollback error: %v", stack, replayErr, rollbackErr)
|
||
notifier.NotifyBackupRunFailures(msg, d)
|
||
})
|
||
// 3a: the pre-push enlargement gate blocked an app's userdata push (config+DB still saved). Edge-
|
||
// triggered by the engine (only NEW blocks notify), so the hub's per-event-type cooldown suffices —
|
||
// no controller-side timer (the hub owns cooldown).
|
||
backupMgr.SetOffboxEnlargeBlockedNotifier(func(stack string, estBytes int64, usedGB, quotaGB int) {
|
||
notifier.NotifyOffboxEnlargeBlocked(fmt.Sprintf(
|
||
"A(z) %s teljes távoli mentése (~%s) túllépné a tárhelykeretet (%d/%d GB). A konfiguráció és az adatbázis továbbra is mentésre kerül; nagyobb kerethez vedd fel velünk a kapcsolatot.",
|
||
stack, appbackup.HumanizeBytes(estBytes), usedGB, quotaGB))
|
||
})
|
||
// v0.142.0 offsite-repo continuity: push a hub event on the orphaned/reset transitions (once per
|
||
// transition — the Manager guards nightly re-fire). Operator-visible on the customer page.
|
||
backupMgr.SetOffboxOrphanEvent(func(eventType, renamedTo string) {
|
||
switch eventType {
|
||
case "offbox_repo_orphaned":
|
||
notifier.PushEvent("offbox_repo_orphaned", "warning",
|
||
"A távoli mentési tároló elárvult: a benne lévő mentések egy korábbi, már nem elérhető kulccsal készültek (újratelepítés). Új mentés a tároló visszaállításáig nem készül.", nil)
|
||
case "offbox_repo_reset":
|
||
notifier.PushEvent("offbox_repo_reset", "info",
|
||
"A távoli mentési tároló visszaállítva: a régi előzmény félretéve (nem törölve), és egy üres, új tároló jött létre a mostani kulccsal.", map[string]string{"renamed_to": renamedTo})
|
||
case "offbox_abandon_completed":
|
||
// R-241: the ONLY event in the product that reports a customer's off-site history
|
||
// being deleted. It is fired after the deletion, not before — the operator wants to
|
||
// know it happened, and a pre-announcement that then fails would be worse than silence.
|
||
notifier.PushEvent("offbox_abandon_completed", "info",
|
||
"A korábbi távoli mentések a türelmi idő lejártával törlésre kerültek, az ügyfél döntése alapján. A hozzájuk tartozó lezárt helyreállítási csomag eltávolítását is kértük.", map[string]string{"deleted_path": renamedTo})
|
||
}
|
||
})
|
||
// v0.271.0 (`09` §3 decision 11, §6.4.2 point 2): the automatic update leg is CHAINED to the
|
||
// off-site job — it runs when this function's off-site half has ended, on every path: configured
|
||
// or not, ok, failed, any of RunOffboxBackup's early returns, even a panic. A box with no off-site
|
||
// target runs it at W+105m. One call site, no signal to go stale.
|
||
sched.Daily("offbox-backup", offboxLeg, func(ctx context.Context) error {
|
||
return chainUpdateLeg(ctx, func(ctx context.Context) error {
|
||
t := sett.GetOffboxTarget()
|
||
if t == nil || !t.Enabled || t.Schedule != "daily" || !backupMgr.OffboxConfigured() {
|
||
logger.Printf("[INFO] [offbox] no scheduled off-site target on this box — the off-site leg does nothing; the update leg runs now")
|
||
return nil // not configured / not scheduled
|
||
}
|
||
return backupMgr.RunOffboxBackup(ctx)
|
||
}, func(ctx context.Context) { stackMgr.RunUpdateLeg(ctx, "after-offsite") })
|
||
})
|
||
// R-241 — the abandonment terminal step. DAILY and not on the backup leg, deliberately: it must
|
||
// run on a box whose off-site tier is NOT configured for runs (an abandoning box may be sitting
|
||
// with escrow pending), and tying it to the backup leg would make the deletion depend on a
|
||
// condition that has nothing to do with it.
|
||
//
|
||
// It is quiet by construction: on every box with no countdown it returns immediately and logs
|
||
// nothing, which is asserted (TestR241_Sweep_QuietWhenNothingDue).
|
||
sched.Daily("offsite-abandon-sweep", "05:10", func(ctx context.Context) error {
|
||
deleted, err := backupMgr.AbandonSweep(ctx)
|
||
if err != nil {
|
||
// NOT a job failure: the countdown stays due and tomorrow's sweep retries. A transport
|
||
// blip must never silently abandon the abandonment.
|
||
logger.Printf("[WARN] [offbox] abandonment sweep: %v (the countdown stays due and retries)", err)
|
||
return nil
|
||
}
|
||
if deleted {
|
||
logger.Printf("[INFO] [offbox] abandonment sweep: the set-aside history was deleted this cycle")
|
||
}
|
||
return nil
|
||
})
|
||
// R-359 — the off-site store gets checked. Nothing verified it before this: the whole-guest
|
||
// tier has verify jobs, the tier holding the customer's documents and photos had none, and we
|
||
// would have found out at restore time with a customer waiting.
|
||
//
|
||
// DAILY JOB, WEEKLY BEHAVIOUR, and that is the point. It asks "is the last successful check
|
||
// older than the max age?" rather than "is it Sunday?", so a box that was switched off on its
|
||
// check day is checked the next day it is on. R-341 is exactly the other shape: a dated check
|
||
// quietly missed and never caught up. No `Weekly` primitive was added to the scheduler —
|
||
// due-ness is smaller, catches up, and is what R-86 already chose for restore-tests.
|
||
//
|
||
// 06:00 CHOSEN FROM THE LIVE SCHEDULE, not from a document. Observed on demo-hp 2026-08-30:
|
||
// db-dump 02:30, tier2-backup 03:30, fill-watch 03:30, metrics-prune 04:00, offbox-backup
|
||
// 04:15, offsite-abandon-sweep 05:10, whole-guest gate [04:30, 08:30). 06:00 is 1h45 clear of
|
||
// the off-site leg (which runs ~2m20s, measured) and 50 min clear of the sweep. The backup
|
||
// WINDOW is customer-configurable, so no fixed time is collision-proof for every box — but a
|
||
// collision costs one skipped day, not a missed check, because due-ness makes tomorrow try
|
||
// again. What would be a real defect is a slot that collides EVERY night; this is not one.
|
||
sched.Daily("offsite-integrity", "06:00", func(ctx context.Context) error {
|
||
runOffsiteIntegrityCheck(ctx, backupMgr, notifier, logger, false)
|
||
return nil
|
||
})
|
||
|
||
// R-87 — the nightly off-site PROOF: one app, its newest snapshot, restored read-only and
|
||
// judged, then deleted.
|
||
//
|
||
// 05:30 CHOSEN FROM THE LIVE SCHEDULE, read off demo-hp on 2026-08-31 and not from a
|
||
// document: db-dump 02:30, tier2-backup 03:30, offbox-backup 04:15 (2m52s measured),
|
||
// offsite-abandon-sweep 05:10, offsite-integrity 06:00 (40.3s measured at 100% depth). 05:30
|
||
// sits in the measured empty gap — 20 min after the sweep starts and 30 min before the
|
||
// integrity check, whose flag it shares.
|
||
//
|
||
// THE WHOLE RUN IS SECONDS, so the slot has room it does not need: all eight apps back to
|
||
// back measured 25 s on this box, and this job does ONE. The backup WINDOW is
|
||
// customer-configurable, so no fixed time is collision-proof on every box — but a collision
|
||
// costs one skipped day rather than a missed proof, because due-ness is per SNAPSHOT and
|
||
// tomorrow tries the same app again. That is the identical argument the integrity job's own
|
||
// slot comment makes, and it is the reason skip-if-busy is the right policy here too.
|
||
sched.Daily("offsite-proof", "05:30", func(ctx context.Context) error {
|
||
runOffsiteProof(ctx, backupMgr, notifier, logger)
|
||
return nil
|
||
})
|
||
}
|
||
|
||
// Metrics prune — daily at 04:00
|
||
if metricsStore != nil {
|
||
sched.Daily("metrics-prune", "04:00", func(ctx context.Context) error {
|
||
deleted, err := metricsStore.Prune(30 * 24 * time.Hour)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
logger.Printf("[INFO] Pruned %d old metric rows", deleted)
|
||
return nil
|
||
})
|
||
}
|
||
|
||
// --- R-167 (decision D-c, customer half): warn BEFORE a filesystem fills ---
|
||
//
|
||
// Nothing warned before v0.191.0. The first sign of a full filesystem was a backup that did not
|
||
// happen, and the only related signal — the healthcheck's generic `health_degraded` at 90% — looked
|
||
// at REGISTERED STORAGE PATHS ONLY, so the docker area and the system-data area (the one holding
|
||
// every driveless app's recovery unit) were invisible, and it never reported a free-byte figure or
|
||
// named a drive.
|
||
//
|
||
// CADENCE: DAILY, at 03:30. A fill is a slow-moving quantity — the thing that fills a disk is a
|
||
// customer's photo library or a nightly backup, not a spike — so a shorter interval buys no
|
||
// earlier warning and only costs statfs calls. 03:30 is deliberately BEFORE the nightly app-data
|
||
// legs (db-dump / tier2 / offbox), so a customer who is about to lose a backup to lack of space
|
||
// hears about it while there is still a night's margin, rather than after the failure.
|
||
// The interval is NOT a cooldown: repeats are impossible because the check is edge-triggered per
|
||
// filesystem, and the hub owns cooldown regardless.
|
||
fillWatcher := fillwatch.New(
|
||
filepath.Join(cfg.Paths.DataDir, "fillwatch-state.json"), logger,
|
||
func() []fillwatch.Target { return fillTargets(cfg, sett) },
|
||
func(p string) *fillwatch.Usage {
|
||
di := system.GetDiskUsage(p)
|
||
if di == nil {
|
||
return nil // §8.4 — unreadable is NOT full; never fabricate a zero here
|
||
}
|
||
return &fillwatch.Usage{
|
||
UsedPercent: di.UsedPercent, AvailGB: di.AvailGB,
|
||
UsedGB: di.UsedGB, TotalGB: di.TotalGB,
|
||
}
|
||
})
|
||
if notifier != nil {
|
||
fillWatcher.SetNotify(func(e fillwatch.Event) {
|
||
// The CUSTOMER's event — deliberately not operator-only. A customer can free space,
|
||
// delete files or add a drive, so this alert is theirs (D-c). The dynamic Hungarian
|
||
// message carries the label and the free space; the hub has NO customerMessages entry for
|
||
// these two types precisely so the template cannot discard them.
|
||
notifier.PushEvent(e.Band.EventType(), e.Band.Severity(), e.Message, map[string]any{
|
||
"path": e.Target.Path, "label": e.Target.Label,
|
||
"used_percent": e.Usage.UsedPercent, "avail_gb": e.Usage.AvailGB,
|
||
"total_gb": e.Usage.TotalGB, "band": e.Band.String(),
|
||
})
|
||
})
|
||
}
|
||
// `09` §3 decision 36: the warning names the kept folders on the filling drive, with their sizes —
|
||
// space the household can free on the kept-data list. Nothing is deleted by the box (D3).
|
||
fillWatcher.SetExtra(func(t fillwatch.Target) string { return keptSpaceSentence(stackMgr, t.Path) })
|
||
sched.Daily("fill-watch", "03:30", func(ctx context.Context) error { return fillWatcher.Check() })
|
||
|
||
// AND ONCE SHORTLY AFTER STARTUP. A box that BOOTS with a filesystem already over the line must
|
||
// warn now, not up to 24 hours later — that is the R-100 shape, a real fault visible only after a
|
||
// deadline elapses. The hub's own checkers handle exactly this case deliberately ("already-breached
|
||
// keys are left UNSEEDED at init so their first Check emits" — the F2 lesson, storage_fill.go), and
|
||
// a daily-only schedule here would have been the same gap one component over.
|
||
//
|
||
// It is SAFE because the check is edge-triggered against PERSISTED state: a filesystem the customer
|
||
// has already been warned about stays silent, so this adds a warning only where one is genuinely
|
||
// owed. The delay lets mounts settle and the drive gate tick first, so a drive that is merely slow
|
||
// to come back reads as unreadable (§8.4, skipped) rather than as anything at all.
|
||
go func() {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-time.After(fillWatchStartupDelay):
|
||
}
|
||
if err := fillWatcher.Check(); err != nil {
|
||
logger.Printf("[WARN] [fillwatch] startup check failed: %v", err)
|
||
}
|
||
}()
|
||
|
||
// --- Central hub reporting schedule ---
|
||
if hubPusher != nil {
|
||
if cfg.Hub.Enabled {
|
||
pushInterval, err := time.ParseDuration(cfg.Hub.PushInterval)
|
||
if err != nil {
|
||
pushInterval = 15 * time.Minute
|
||
}
|
||
sched.Every("hub-report", pushInterval, func(ctx context.Context) error {
|
||
r := report.BuildReport(cfg, *configPath, stackMgr, backupMgr, cpuCollector, metricsStore, Version, sett.GetStoragePaths(), sett.GetGeoRestriction(), sett.GetSMBSettings(), logger)
|
||
r.Claimed = sett.GetClaimed() // v0.122.0 (F-4): set-only claim flag for the hub
|
||
r.Language = sett.GetLanguage() // v0.247.0 (i18n): the hub e-mails follow it (slice 3)
|
||
if err := hubPusher.Push(r); err != nil {
|
||
return err
|
||
}
|
||
// Drain pending events (e.g., DR recovery completed) after successful push
|
||
if events := sett.DrainPendingEvents(); len(events) > 0 {
|
||
for _, ev := range events {
|
||
notifier.Notify(ev.EventType, ev.Severity, ev.Message, ev.Details)
|
||
}
|
||
logger.Printf("[INFO] Drained %d pending events to Hub", len(events))
|
||
}
|
||
return nil
|
||
})
|
||
logger.Printf("[INFO] Hub reporting enabled (every %s to %s)", pushInterval, cfg.Hub.URL)
|
||
} else {
|
||
logger.Printf("[INFO] Hub reporting disabled — will send disabled notification to %s", cfg.Hub.URL)
|
||
}
|
||
}
|
||
|
||
// Self-update scheduler jobs
|
||
if cfg.SelfUpdate.Enabled && updater != nil {
|
||
// Periodic version check (populates UI, never triggers update)
|
||
checkInterval, ciErr := time.ParseDuration(cfg.SelfUpdate.CheckInterval)
|
||
if ciErr != nil {
|
||
checkInterval = 6 * time.Hour
|
||
}
|
||
sched.Every("selfupdate-check", checkInterval, func(ctx context.Context) error {
|
||
result := updater.CheckForUpdate()
|
||
if result.UpdateAvailable {
|
||
logger.Printf("[INFO] Update available: %s -> %s", result.CurrentVersion, result.LatestVersion)
|
||
}
|
||
return nil
|
||
})
|
||
|
||
// Auto-update (daily, fires after typical backup completion)
|
||
if cfg.SelfUpdate.AutoUpdate {
|
||
sched.Daily("selfupdate-auto", cfg.SelfUpdate.AutoUpdateTime, func(ctx context.Context) error {
|
||
result := updater.CheckForUpdate()
|
||
if !result.UpdateAvailable {
|
||
return nil
|
||
}
|
||
if err := updater.TriggerUpdate("auto"); err != nil {
|
||
logger.Printf("[WARN] Auto-update skipped: %v", err)
|
||
}
|
||
return nil
|
||
})
|
||
}
|
||
}
|
||
|
||
// Storage watchdog (disk disconnect/reconnect detection) has moved to the host
|
||
// agent (slice 8C) — the controller no longer owns disk-level monitoring.
|
||
|
||
// --- Asset syncer (download from Hub) ---
|
||
var assetsSyncer *assets.Syncer
|
||
if cfg.Hub.Enabled && cfg.Assets.SyncEnabled && cfg.Hub.URL != "" && cfg.Hub.APIKey != "" {
|
||
assetsDir := filepath.Join(cfg.Paths.DataDir, "assets")
|
||
assetsSyncer = assets.New(cfg.Hub.URL, cfg.Hub.APIKey, assetsDir, "/usr/share/felhom/assets", logger, cfg.Logging.Level == "debug")
|
||
go func() {
|
||
time.Sleep(10 * time.Second)
|
||
if err := assetsSyncer.Sync(ctx); err != nil {
|
||
logger.Printf("[WARN] Initial asset sync failed: %v", err)
|
||
}
|
||
}()
|
||
sched.Daily("asset-sync", cfg.Assets.SyncSchedule, func(ctx context.Context) error {
|
||
return assetsSyncer.Sync(ctx)
|
||
})
|
||
logger.Printf("[INFO] Asset sync enabled (daily at %s from Hub)", cfg.Assets.SyncSchedule)
|
||
}
|
||
|
||
// --- Startup self-test ---
|
||
selfTestResult := selftest.Run(cfg, sett, logger)
|
||
|
||
sched.Start(ctx)
|
||
defer sched.Stop()
|
||
|
||
// Generate recovery info file if retrieval password is set
|
||
if rp := sett.GetRetrievalPassword(); rp != "" {
|
||
go func() {
|
||
info := recovery.Info{
|
||
CustomerID: cfg.Customer.ID,
|
||
RetrievalPassword: rp,
|
||
HubURL: cfg.Hub.URL,
|
||
SupportEmail: "support@felhom.eu",
|
||
SupportURL: "https://felhom.eu/kapcsolat",
|
||
}
|
||
if err := recovery.GenerateRecoveryFile(info, Version, cfg.Paths.DataDir); err != nil {
|
||
logger.Printf("[WARN] Failed to generate recovery-info.txt: %v", err)
|
||
}
|
||
}()
|
||
}
|
||
|
||
// Fire startup pings + hub report immediately (don't wait for first scheduler tick)
|
||
go func() {
|
||
time.Sleep(5 * time.Second) // Let all subsystems fully initialize
|
||
|
||
// Push controller startup event to Hub
|
||
notifier.NotifyControllerStarted(Version, map[string]interface{}{
|
||
"selftest_pass": selfTestResult.Pass,
|
||
"selftest_warn": selfTestResult.Warn,
|
||
"selftest_fail": selfTestResult.Fail,
|
||
})
|
||
|
||
// Hub report
|
||
if hubPusher != nil {
|
||
if cfg.Hub.Enabled {
|
||
r := report.BuildReport(cfg, *configPath, stackMgr, backupMgr, cpuCollector, metricsStore, Version, sett.GetStoragePaths(), sett.GetGeoRestriction(), sett.GetSMBSettings(), logger)
|
||
r.Claimed = sett.GetClaimed() // v0.122.0 (F-4): set-only claim flag for the hub
|
||
r.Language = sett.GetLanguage() // v0.247.0 (i18n): the hub e-mails follow it (slice 3)
|
||
var pushErr error
|
||
for attempt := 1; attempt <= 3; attempt++ {
|
||
pushErr = hubPusher.Push(r)
|
||
if pushErr == nil {
|
||
logger.Println("[INFO] Startup hub report sent")
|
||
break
|
||
}
|
||
logger.Printf("[WARN] Startup hub report attempt %d/3 failed: %v", attempt, pushErr)
|
||
if attempt < 3 {
|
||
time.Sleep(15 * time.Second)
|
||
}
|
||
}
|
||
if pushErr != nil {
|
||
logger.Printf("[WARN] Startup hub report failed after 3 attempts — next scheduled push in %s", cfg.Hub.PushInterval)
|
||
}
|
||
} else {
|
||
// Send a minimal "disabled" notification so hub knows reporting is intentionally off
|
||
r := &report.Report{
|
||
Version: 1,
|
||
CustomerID: cfg.Customer.ID,
|
||
CustomerName: cfg.Customer.Name,
|
||
ControllerVersion: Version,
|
||
Timestamp: time.Now().UTC(),
|
||
ReportingDisabled: true,
|
||
Health: report.HealthReport{Status: "disabled", Issues: []string{}, Warnings: []string{}},
|
||
Stacks: report.StacksReport{Deployed: []string{}, Available: []string{}},
|
||
Containers: report.ContainerReport{List: []report.ContainerDetailReport{}},
|
||
}
|
||
hubPusher.PushOnce(r)
|
||
}
|
||
}
|
||
|
||
// Initial self-update check (so settings page shows version info quickly)
|
||
if updater != nil {
|
||
time.Sleep(25 * time.Second) // Additional delay after hub report
|
||
result := updater.CheckForUpdate()
|
||
if result.UpdateAvailable {
|
||
logger.Printf("[INFO] Startup: update available %s -> %s", result.CurrentVersion, result.LatestVersion)
|
||
} else if result.Error != "" {
|
||
logger.Printf("[DEBUG] Startup version check: %s", result.Error)
|
||
}
|
||
}
|
||
}()
|
||
|
||
// Initial backup cache population (don't block startup)
|
||
if cfg.Backup.Enabled && backupMgr != nil {
|
||
go func() {
|
||
curDB, _, _ := backupwindow.LegTimes(backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule))
|
||
backupMgr.RefreshCache(scheduler.NextDailyRun(curDB))
|
||
}()
|
||
}
|
||
|
||
// Sync notification preferences to hub on startup (handles hub DB rebuild recovery)
|
||
if notifier.IsEnabled() {
|
||
go func() {
|
||
prefs := sett.GetNotificationPrefs()
|
||
if prefs.Email != "" {
|
||
if err := notifier.SyncPreferences(prefs.Email, prefs.EnabledEvents, prefs.CooldownHours); err != nil {
|
||
logger.Printf("[WARN] Failed to sync notification preferences on startup: %v", err)
|
||
}
|
||
}
|
||
}()
|
||
}
|
||
|
||
// Initial alert refresh (so alerts appear immediately, not after first 5min health check)
|
||
go func() {
|
||
report := monitor.RunHealthCheck(cfg, cpuCollector, sett.GetStoragePaths(), sett.GetSMBSettings(), logger)
|
||
alertMgr.Refresh(report, cfg, backupMgr, false, "")
|
||
}()
|
||
|
||
// --- Out-of-cycle report trigger (v0.139.0, Direction 1) ---
|
||
// ONE canonical fire closure (full BuildReport + Claimed + Push) behind a debounced,
|
||
// coalescing trigger — user actions with hub-side effects (geo, escrow claim, settings
|
||
// save, offsite toggle, app deploy/remove, customer claim) round-trip in seconds instead
|
||
// of the next ~15-min cycle. The scheduled hub-report job above stays the reconciliation
|
||
// backbone; the trigger is best-effort on top (its failures degrade to the cycle).
|
||
// nil when hub reporting is off → the api/web seams stay unset (strict no-op).
|
||
var reportTrigger *report.Trigger
|
||
if hubPusher != nil && cfg.Hub.Enabled {
|
||
fireReport := func() error {
|
||
rep := report.BuildReport(cfg, *configPath, stackMgr, backupMgr, cpuCollector, metricsStore, Version, sett.GetStoragePaths(), sett.GetGeoRestriction(), sett.GetSMBSettings(), logger)
|
||
rep.Claimed = sett.GetClaimed() // v0.122.0 (F-4): set-only claim flag for the hub
|
||
rep.Language = sett.GetLanguage() // v0.247.0 (i18n): the hub e-mails follow it (slice 3)
|
||
return hubPusher.Push(rep)
|
||
}
|
||
reportTrigger = report.NewTrigger(fireReport, logger)
|
||
go reportTrigger.Run(ctx)
|
||
|
||
// Direction-2 immediate-sync (v0.140.0): hold a hanging GET against the hub's wait channel
|
||
// and fire the trigger the instant operator intent moves — so an operator save round-trips in
|
||
// seconds instead of on the next ~15-min cycle. Reuses the SAME hub URL + key as the pusher;
|
||
// no new config keys. Gated on the trigger existing (hub reporting enabled). The ACK from the
|
||
// fired report delivers everything through the unchanged machinery.
|
||
waiter := report.NewWaiter(cfg.Hub.URL, cfg.Hub.APIKey, reportTrigger.Fire, logger)
|
||
go waiter.Run(ctx)
|
||
}
|
||
|
||
// v0.280.0 (`09` §3 decision 46): the setup gate's loop — every closed gate's traefik file exists, a probe
|
||
// that says "set up" opens its gate, and a gate file nobody owns is removed.
|
||
go stackMgr.RunSetupGateLoop(ctx, 20*time.Second)
|
||
// decision 53 (R-736): the one-time image clean-up for a box older than the rule — after the first catalog sync
|
||
// has had time to land (it reads the catalog's image names), once per box (a marker file).
|
||
go func() {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-time.After(3 * time.Minute):
|
||
}
|
||
stackMgr.RunImageRetentionOnce()
|
||
// decision 56 (R-745): the controller's own images — keep the running one and the one before it. At every
|
||
// start, 3 minutes in: a swap has ended by then (the agent's verify window is 90 s), so the image the agent
|
||
// would roll back to is the running one, and the one before it is in the swap record read here.
|
||
prevCtl := ""
|
||
if st, err := selfupdate.LoadState(cfg.Paths.DataDir); err == nil {
|
||
prevCtl = st.RecordedPrevious(Version)
|
||
}
|
||
_, _ = stackMgr.RetainControllerImages(stacks.ControllerImageRecord{Repo: cfg.SelfUpdate.Image, Running: Version, Previous: prevCtl})
|
||
}()
|
||
|
||
// --- Initialize API router ---
|
||
apiRouter := api.NewRouter(cfg, *configPath, sett, stackMgr, syncer, cpuCollector, backupMgr, metricsStore, updater, notifier, logger)
|
||
if reportTrigger != nil {
|
||
// Out-of-cycle, non-blocking hub report push (geo changes + app deploy/remove) so
|
||
// the hub reflects the new state immediately instead of after the next ~15-min cycle.
|
||
apiRouter.SetReportPushTrigger(reportTrigger.Fire)
|
||
}
|
||
if assetsSyncer != nil {
|
||
apiRouter.SetAssetsSyncer(assetsSyncer)
|
||
}
|
||
|
||
// --- Initialize Cloudflare geo-restriction ---
|
||
var geoSync *cf.GeoSyncManager
|
||
if cfg.Infrastructure.CFAPIToken != "" {
|
||
cfClient := cf.New(cfg.Infrastructure.CFAPIToken, logger, cfg.Logging.Level == "debug")
|
||
geoStacks := &geoStackAdapter{mgr: stackMgr, domain: cfg.Customer.Domain}
|
||
geoSync = cf.NewGeoSyncManager(cfClient, sett, cfg.Customer.Domain, geoStacks, logger)
|
||
geoSync.SetDebug(cfg.Logging.Level == "debug")
|
||
apiRouter.SetGeoSync(geoSync)
|
||
|
||
// Re-sync geo rules when apps are deployed/removed
|
||
apiRouter.OnGeoRelevantChange = func() {
|
||
geo := sett.GetGeoRestriction()
|
||
if geo != nil && geo.Enabled {
|
||
if err := geoSync.Sync(context.Background()); err != nil {
|
||
logger.Printf("[WARN] Geo sync after app change failed: %v", err)
|
||
}
|
||
}
|
||
}
|
||
|
||
// Periodic verification every 6 hours
|
||
sched.Every("geo-verify", 6*time.Hour, func(ctx context.Context) error {
|
||
geo := sett.GetGeoRestriction()
|
||
if geo == nil || !geo.Enabled {
|
||
return nil
|
||
}
|
||
return geoSync.Sync(ctx)
|
||
})
|
||
|
||
// Initial sync (delayed, non-blocking)
|
||
go func() {
|
||
time.Sleep(15 * time.Second)
|
||
if geo := sett.GetGeoRestriction(); geo != nil && geo.Enabled {
|
||
if err := geoSync.Sync(context.Background()); err != nil {
|
||
logger.Printf("[WARN] Initial geo sync failed: %v", err)
|
||
}
|
||
}
|
||
}()
|
||
|
||
logger.Printf("[INFO] Geo-restriction support enabled (CF API token configured)")
|
||
}
|
||
|
||
// --- Initialize integration manager ---
|
||
integrationStacks := &integrationStackAdapter{mgr: stackMgr}
|
||
integrationMgr := integrations.NewManager(sett, integrationStacks, cfg.Customer.Domain, cfg.Paths.StacksDir, encKey, logger)
|
||
integrationMgr.SetDebug(cfg.Logging.Level == "debug")
|
||
apiRouter.SetIntegrationManager(integrationMgr)
|
||
|
||
// --- Initialize app exporter ---
|
||
exportProv := &exportAdapter{mgr: stackMgr, encKey: encKey}
|
||
appExporter := appexport.NewExporter(exportProv, logger, Version)
|
||
appExporter.SetDebug(cfg.Logging.Level == "debug")
|
||
// R-166: the exporter stops apps too (export with "stop the app first"), so it shares the backup
|
||
// manager's ONE marker file rather than opening a second one — one file, one recovery. Without
|
||
// this the export path would be the uncovered sibling of two covered ones, which is how a reader
|
||
// concludes the whole class is handled (§2.2).
|
||
appExporter.SetStopGuard(exportStopGuard{g: appStopGuard})
|
||
apiRouter.SetDebug(cfg.Logging.Level == "debug")
|
||
|
||
// --- Initialize web server ---
|
||
webServer := web.NewServer(cfg, stackMgr, cpuCollector, backupMgr, sched, sett, alertMgr, notifier, updater, logger, Version)
|
||
// Migration done-hook: a decommission-initiated migration finalizes the source decommission on
|
||
// success (soft-mark + agent). Wire it before RecoverMigration so a resumed one still finalizes.
|
||
stackMgr.SetMigrationDoneHook(webServer.OnMigrationDone)
|
||
// R-536: „Alkalmazás telepítve" is sent when the async deploy ACTUALLY finishes, and a deploy
|
||
// that ends badly now says so instead of being silence. The accept-time event is
|
||
// `app_deploy_started`, emitted by the API router beside its 202.
|
||
// v0.279.0: the hook is set on EVERY box — after_install (decision 45) must run on a box with no hub too;
|
||
// only the events need the notifier.
|
||
stackMgr.SetDeployDoneHook(func(name string, ok bool, detail string) {
|
||
display := name
|
||
if s, found := stackMgr.GetStack(name); found && s.Meta.DisplayName != "" {
|
||
display = s.Meta.DisplayName
|
||
}
|
||
if ok {
|
||
if notifier != nil {
|
||
notifier.NotifyAppDeployed(name, display)
|
||
}
|
||
// v0.283.0 (decision 50, R-720): a FRESH install on a box whose customer has off-site is in the
|
||
// off-site copy by default. An app with an earlier backup choice keeps it (settings rule).
|
||
if on, why := sett.DefaultOffboxOnForNewApp(name); on {
|
||
log.Printf("[INFO] [backup] %s: off-site copy switched ON at install (decision 50)", name)
|
||
} else {
|
||
log.Printf("[DEBUG] [backup] %s: off-site default not applied: %s", name, why)
|
||
}
|
||
// v0.279.0 (decision 45): a FRESH install replaces a known default login with its generated one,
|
||
// when the template declares after_install. Never on a restore or a kept-data load — those do not
|
||
// pass through the deploy-done hook.
|
||
go func() {
|
||
if ran, err := stackMgr.RunAfterInstall(name, 10*time.Minute); ran && err != nil {
|
||
log.Printf("[WARN] [stacks] after_install %s: %v", name, err)
|
||
}
|
||
}()
|
||
return
|
||
}
|
||
if notifier != nil {
|
||
notifier.NotifyAppDeployFailed(name, display, detail)
|
||
}
|
||
})
|
||
// R-681 (v0.270.0): an install a restart cut off is finished through the failure path and reported —
|
||
// AFTER the deploy-done hook exists, so its event is sent.
|
||
if names := stackMgr.RecoverInterruptedInstalls(); len(names) > 0 {
|
||
log.Printf("[WARN] [stacks] %d install(s) interrupted by the restart were resolved and reported: %v", len(names), names)
|
||
}
|
||
stackMgr.RecoverMigration(ctx)
|
||
webServer.SetEncryptionKey(encKey)
|
||
webServer.SetAppExporter(appExporter)
|
||
// Disk-health degradation check (v0.169.0; hourly + persisted state v0.215.0): compare each
|
||
// physical disk's SMART verdict against the PERSISTED per-disk record (disk_health_state.go) and
|
||
// emit disk_health_degraded per the v0.215.0 decision rules — escalation against the last ALERTED
|
||
// verdict, plus a re-alert for a disk already at Hiba that keeps worsening. A disk's first verdict
|
||
// baselines silently; recovery and UNKNOWN never notify. Only on a provisioned guest (an agent to
|
||
// read /disks from); the check no-ops gracefully if the agent is unreachable.
|
||
//
|
||
// WHY HOURLY, not the original 6h: the one real failing drive's benign excursion lasted about ONE
|
||
// HOUR (11 Aug 12:28 -> 13:28, cleared completely). A 6-hourly sampler can land either side of an
|
||
// excursion like that and see nothing, then catch the terminal run half a day late. Measured on
|
||
// demo-hp 2026-08-14, the real /disks fetch costs min 0.805s / median 0.821s / max 0.841s over 10
|
||
// calls (3 physical disk rows), so an hourly poll is ~0.02% duty — far under the 5s bar that would
|
||
// have kept this at 6h.
|
||
if cfg.LocalAPI.Endpoint != "" {
|
||
sched.Every("disk-health-check", 1*time.Hour, webServer.RunDiskHealthCheck)
|
||
}
|
||
// Browser .fab upload (v0.128.0): upload state is in-memory, so a restart strands the .part —
|
||
// GC stray part files in every registered drive's exports dir at startup.
|
||
webServer.CleanupStaleUploadParts()
|
||
// Escrow wizard (v0.127.0): the Scenario-F stale-blob flag feeds the Távoli mentés card.
|
||
if escrowConfirmer != nil {
|
||
webServer.SetEscrowStale(escrowConfirmer.StaleBlob)
|
||
// R-193: the recovery screen may state WHEN the hub sealed the package, and nothing more,
|
||
// before a code is entered. A timestamp, never a secret.
|
||
webServer.SetEscrowSealedAt(escrowConfirmer.SealedAt)
|
||
}
|
||
// R-219 / R-218: the unlock brings the off-site tier up between placing the recovered key and
|
||
// reading the repository. Without it the listing the screen promises can never render on the
|
||
// pristine rebuilt box — no key ⇒ no target ⇒ no inventory — and the apply-bridge's only other
|
||
// trigger is a config refresh or a restart, neither of which the customer can cause.
|
||
if offsiteBridge != nil {
|
||
webServer.SetRecoveryTierUp(offsiteBridge.Reconcile)
|
||
}
|
||
webServer.SetIntegrationManager(integrationMgr)
|
||
if reportTrigger != nil {
|
||
// Out-of-cycle report push after hub-relevant user actions (escrow claim, settings
|
||
// save, offsite config/toggle, customer claim) — same debounced trigger as the api
|
||
// router's; nil (hub reporting off) leaves the seam a strict no-op.
|
||
webServer.SetReportTrigger(reportTrigger.Fire)
|
||
}
|
||
if quiesceLoop != nil {
|
||
webServer.SetBackupTrigger(quiesceLoop) // "Mentés most" → app-consistent backup via the quiesce loop
|
||
}
|
||
if mailShim != nil {
|
||
webServer.SetMailShim(mailShim) // settings toggle starts/stops the app-email shim at runtime
|
||
}
|
||
if assetsSyncer != nil {
|
||
webServer.SetAssetsSyncer(assetsSyncer)
|
||
}
|
||
if hubPusher != nil {
|
||
webServer.SetHubPushStatus(func() web.HubPushStatusData {
|
||
s := hubPusher.GetStatus()
|
||
return web.HubPushStatusData{
|
||
LastAttempt: s.LastAttempt,
|
||
LastSuccess: s.LastSuccess,
|
||
LastError: s.LastError,
|
||
Consecutive: s.Consecutive,
|
||
}
|
||
})
|
||
}
|
||
if logBuffer != nil {
|
||
webServer.SetLogBuffer(logBuffer)
|
||
}
|
||
webServer.SetStartTime(startTime)
|
||
|
||
// Controller→agent channel health (self-health slice). A ~60s probe of the local-API channel via
|
||
// the PRODUCTION memoized client (webServer.ProbeAgentChannel), classified + debounced, alerting the
|
||
// operator + dashboard on a transition. Only on a provisioned guest (endpoint set) — mirrors
|
||
// probeLocalAPI's guard. Registered after webServer (it owns the memoized client); sched.Every
|
||
// launches the job immediately since the scheduler is already started.
|
||
if cfg.LocalAPI.Endpoint != "" {
|
||
chSink := channelSink{notifier: notifier, alertMgr: alertMgr}
|
||
chChecker := channelhealth.New(webServer.ProbeAgentChannel, chSink, logger)
|
||
sched.Every("agent-channel-health", 60*time.Second, chChecker.Check)
|
||
}
|
||
|
||
// local_api endpoint drift (R-77, from the 2026-07-25 outage): controller.yaml and bootstrap.json
|
||
// can disagree indefinitely and silently — the island migration rewrote the latter and the
|
||
// controller kept dialling the former for 17.5 h, alerting only "agent unreachable". This NAMES
|
||
// the fault; it deliberately does not reconcile the files (R-78 owns which one wins).
|
||
//
|
||
// Startup-only is sufficient and correct: both files are read at boot and neither changes under a
|
||
// running controller, so a periodic re-check would add noise without adding signal.
|
||
if d := bootstrap.DetectEndpointDrift(*configPath, cfg, logger); d != nil {
|
||
alertMgr.SetEndpointDriftAlert(true, "alert.endpoint_drift", d.HungarianMessage())
|
||
if notifier != nil {
|
||
notifier.NotifyEndpointDrift(d.EnglishMessage(), d.FingerprintAgrees)
|
||
}
|
||
}
|
||
|
||
// Wire debug callbacks (only in debug mode)
|
||
if cfg.Logging.Level == "debug" {
|
||
dc := &web.DebugCallbacks{}
|
||
if hubPusher != nil {
|
||
dc.TriggerHubReportPush = func() error {
|
||
r := report.BuildReport(cfg, *configPath, stackMgr, backupMgr, cpuCollector, metricsStore, Version, sett.GetStoragePaths(), sett.GetGeoRestriction(), sett.GetSMBSettings(), logger)
|
||
r.Claimed = sett.GetClaimed() // v0.122.0 (F-4): set-only claim flag for the hub
|
||
r.Language = sett.GetLanguage() // v0.247.0 (i18n): the hub e-mails follow it (slice 3)
|
||
return hubPusher.Push(r)
|
||
}
|
||
}
|
||
// R-359/R-397: the debug button's caller. Same function as the scheduled job, `force=true` the
|
||
// only difference — see runOffsiteIntegrityCheck.
|
||
dc.RunIntegrityCheck = func(force bool) backup.IntegrityResult {
|
||
ctx, cancel := context.WithTimeout(context.Background(), 35*time.Minute)
|
||
defer cancel()
|
||
return runOffsiteIntegrityCheck(ctx, backupMgr, notifier, logger, force)
|
||
}
|
||
// R-87: the proof button's caller. Same function as the scheduled job — no second path.
|
||
dc.RunOffsiteProof = func() backup.ProofResult {
|
||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Minute)
|
||
defer cancel()
|
||
return runOffsiteProof(ctx, backupMgr, notifier, logger)
|
||
}
|
||
dc.HubConnectivityTest = func() (int, int64, error) {
|
||
start := time.Now()
|
||
client := &http.Client{Timeout: 10 * time.Second}
|
||
resp, err := client.Get(cfg.Hub.URL + "/healthz")
|
||
latency := time.Since(start).Milliseconds()
|
||
if err != nil {
|
||
return 0, latency, err
|
||
}
|
||
resp.Body.Close()
|
||
return resp.StatusCode, latency, nil
|
||
}
|
||
if cfg.Git.RepoURL != "" {
|
||
dc.GiteaConnectivityTest = func() (int, int64, error) {
|
||
start := time.Now()
|
||
client := &http.Client{Timeout: 5 * time.Second}
|
||
resp, err := client.Head(cfg.Git.RepoURL)
|
||
latency := time.Since(start).Milliseconds()
|
||
if err != nil {
|
||
return 0, latency, err
|
||
}
|
||
resp.Body.Close()
|
||
return resp.StatusCode, latency, nil
|
||
}
|
||
}
|
||
dc.GetTelemetryPreview = func() ([]report.AppTelemetry, error) {
|
||
return report.BuildAppTelemetryForDebug(stackMgr, metricsStore, logger), nil
|
||
}
|
||
webServer.SetDebugCallbacks(dc)
|
||
}
|
||
|
||
// Drive migration (full-drive move) has moved to the host agent (slice 8C);
|
||
// the controller no longer runs a DriveMigrator.
|
||
|
||
// --- Build HTTP mux ---
|
||
mux := http.NewServeMux()
|
||
|
||
// API routes (no auth for health endpoint, auth for everything else)
|
||
mux.HandleFunc("/api/health", apiRouter.HealthHandler)
|
||
// Disk management API — thin proxy to the host agent (slice 8C). The agent owns
|
||
// disk execution; the controller forwards list/assign/eject/format.
|
||
mux.Handle("/api/disks", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeDiskAPI))))
|
||
mux.Handle("/api/disks/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeDiskAPI))))
|
||
// Guided storage provisioning (init/attach/eject orchestration over the agent disk API + registry).
|
||
mux.Handle("/api/storage/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeStorageAPI))))
|
||
// LAN network-sharing folder picker (R-7 slice 1). Must live here, not in the web ServeHTTP
|
||
// switch: the /api/ subtree is routed on this mux, so a case there is shadowed and 401s.
|
||
mux.Handle("/api/sharing/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeSharingAPI))))
|
||
// Guest RAM resize (v0.143.0, R-24): read current allocation/bounds + apply a bounded resize.
|
||
mux.Handle("/api/system/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeSystemAPI))))
|
||
// R-490 (v0.242.0): the web layer's /api/system/ prefix knows only the memory routes and answered
|
||
// 404 to /api/system/info, so the monitoring page's memory-distribution card never rendered. An
|
||
// exact pattern wins over the prefix in ServeMux. Pinned by TestR490_SystemInfoIsMountedAheadOfTheWebPrefix.
|
||
mux.Handle("/api/system/info", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(apiRouter.ServeHTTP))))
|
||
// Standalone full-server (guest) restart — the "Kiszolgáló újraindítása" maintenance affordance,
|
||
// a sibling to the controller-only /api/selfrestart. Reuses the agent GuestReboot primitive.
|
||
mux.Handle("/api/server/reboot", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.HandleServerReboot))))
|
||
// Whole-guest (appliance) backup visibility + manual trigger. Distinct prefix from apiRouter's
|
||
// app-data /api/backup/{run,status} (DB dumps) to avoid shadowing the /api/ catch-all subtree.
|
||
mux.Handle("/api/guest-backup/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeBackupAPI))))
|
||
// Host metrics API — thin proxy to the host agent (slice 9). Read-only host-wide health +
|
||
// per-storage capacity for the monitoring view; the de-privileged controller can't read the
|
||
// host itself. GET only, so no CSRF wrapper needed.
|
||
mux.Handle("/api/host-metrics", webServer.RequireAuth(http.HandlerFunc(webServer.ServeHostMetricsAPI)))
|
||
// App export/import API routes handled by web server
|
||
mux.Handle("/api/export/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeExportAPI))))
|
||
// Debug API routes handled by web server (debug-mode gating inside handler)
|
||
mux.Handle("/api/debug/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeDebugAPI))))
|
||
// Escrow ceremony wizard API (v0.127.0) — session auth + CSRF on POSTs; the claim response is
|
||
// the ONLY surface the recovery code R ever crosses (no-store, never logged).
|
||
mux.Handle("/api/escrow/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeEscrowAPI))))
|
||
// Self-update API — accepts session auth OR hub API key (for external triggering)
|
||
// CsrfProtect exempts Bearer-token requests automatically.
|
||
mux.Handle("/api/selfupdate/", selfUpdateAuthMiddleware(cfg, webServer, webServer.CsrfProtect(http.HandlerFunc(apiRouter.ServeHTTP))))
|
||
// Config API — accepts session auth OR hub API key (for Hub config push)
|
||
mux.Handle("/api/config/", selfUpdateAuthMiddleware(cfg, webServer, webServer.CsrfProtect(http.HandlerFunc(apiRouter.ServeHTTP))))
|
||
// Geo API — accepts session auth OR hub API key (for Hub geo-disable)
|
||
mux.Handle("/api/geo/", selfUpdateAuthMiddleware(cfg, webServer, webServer.CsrfProtect(http.HandlerFunc(apiRouter.ServeHTTP))))
|
||
mux.Handle("/api/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(apiRouter.ServeHTTP))))
|
||
|
||
// Web UI routes (auth required)
|
||
mux.Handle("/", webServer.RequireAuth(webServer.CsrfProtect(http.HandlerFunc(webServer.ServeHTTP))))
|
||
|
||
// --- Start HTTP server ---
|
||
server := &http.Server{
|
||
Addr: cfg.Web.Listen,
|
||
Handler: webServer.CatchAllMiddleware(mux),
|
||
ReadTimeout: 30 * time.Second,
|
||
WriteTimeout: 60 * time.Second,
|
||
IdleTimeout: 120 * time.Second,
|
||
}
|
||
|
||
sigCh := make(chan os.Signal, 1)
|
||
signal.Notify(sigCh, syscall.SIGINT, syscall.SIGTERM)
|
||
|
||
go func() {
|
||
sig := <-sigCh
|
||
logger.Printf("[INFO] Received signal %v, shutting down...", sig)
|
||
// fix-6: spill the debug ring on a clean shutdown so a `systemctl restart` / container recreate
|
||
// preserves the pre-restart window (the periodic spill already covers a hard crash to ≤60 s).
|
||
if err := logBuffer.SpillTo(ringSpillPath); err != nil {
|
||
logger.Printf("[WARN] debug-ring spill on shutdown failed: %v", err)
|
||
}
|
||
cancel()
|
||
if mailShim != nil {
|
||
mailShim.Close()
|
||
}
|
||
|
||
shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||
defer shutdownCancel()
|
||
|
||
if err := server.Shutdown(shutdownCtx); err != nil {
|
||
logger.Printf("[ERROR] HTTP server shutdown error: %v", err)
|
||
}
|
||
}()
|
||
|
||
logger.Printf("[INFO] Web UI listening on %s", cfg.Web.Listen)
|
||
if err := server.ListenAndServe(); err != http.ErrServerClosed {
|
||
logger.Fatalf("[FATAL] HTTP server error: %v", err)
|
||
}
|
||
|
||
logger.Println("[INFO] felhom-controller stopped")
|
||
}
|
||
|
||
// selfUpdateAuthMiddleware allows access via session auth (normal UI) OR hub API key bearer token (external).
|
||
// deadAppBootGrace is the startup settle window before fix-3 evaluates deployed-app run states — apps
|
||
// legitimately take 30–60 s to come up, so a shorter grace would false-alarm during the controller's
|
||
// own boot. After the grace, an app that still isn't running alerts (the F11 dead-at-boot case).
|
||
const deadAppBootGrace = 90 * time.Second
|
||
|
||
// deadAppHeartbeatEvery is how many 30 s scans pass between deadapp heartbeat lines (F-OBS).
|
||
// 20 scans = one line per ~10 minutes: frequent enough that a stalled detector is obvious well
|
||
// inside the 180 s alarm grace it feeds, and 144 lines/day instead of 2880.
|
||
const deadAppHeartbeatEvery = 20
|
||
|
||
// deadAppScans counts completed deadapp scans since boot. Single-goroutine (the scheduler runs
|
||
// each job serially), so it needs no lock.
|
||
var deadAppScans int
|
||
|
||
// noteDeadAppScan emits the F-OBS heartbeat every deadAppHeartbeatEvery-th scan, at [INFO] so it
|
||
// survives a default `logging.level: info` box. Extracted from the job closure so a test can assert
|
||
// the OBSERVABLE (the line, at info) rather than merely that the scan function ran — which is the
|
||
// distinction F-OBS is about.
|
||
func noteDeadAppScan(logger *log.Logger, scans, evaluated, down int) {
|
||
if logger == nil || deadAppHeartbeatEvery <= 0 || scans%deadAppHeartbeatEvery != 0 {
|
||
return
|
||
}
|
||
logger.Printf("[INFO] [deadapp] check alive: %d scans since boot, %d deployed app(s) evaluated, %d currently down",
|
||
scans, evaluated, down)
|
||
}
|
||
|
||
// noteUnknownIntentSuppressions reports every app whose dead-app alarm was suppressed ONLY because
|
||
// no customer intent was ever recorded (R-386, §4's ruling).
|
||
//
|
||
// **A rule without a mechanism is not a rule.** §4 chose to keep today's behaviour for the unknown
|
||
// case, which is the safe choice — but a safe choice that is invisible is indistinguishable from the
|
||
// defect it replaced. This line is what makes the gap BOUNDED rather than silent: an operator can
|
||
// grep one word and answer "how many apps am I blind to, and which?".
|
||
//
|
||
// It rides the same cadence as the F-OBS heartbeat rather than firing every 30 s, for the reason
|
||
// noteDeadAppScan records: a per-scan line is 2880 lines/day of noise, and noise is what made the
|
||
// original author choose silence. The population only changes when someone starts or stops an app.
|
||
func noteUnknownIntentSuppressions(logger *log.Logger, states []notify.AppRunState) {
|
||
if logger == nil || deadAppHeartbeatEvery <= 0 || deadAppScans%deadAppHeartbeatEvery != 0 {
|
||
return
|
||
}
|
||
var names []string
|
||
for _, s := range states {
|
||
if s.IntentUnknown {
|
||
names = append(names, s.Name)
|
||
}
|
||
}
|
||
if len(names) == 0 {
|
||
return
|
||
}
|
||
sort.Strings(names)
|
||
logger.Printf("[INFO] [deadapp] %d stopped app(s) have NO recorded customer intent, so their "+
|
||
"dead-app alarm is suppressed by the unknown-intent fallback (R-386): %s. This closes itself "+
|
||
"as each app is started or stopped through the interface.", len(names), strings.Join(names, ", "))
|
||
}
|
||
|
||
// bootReconcileSettle is the delay before the FIRST observation. It lets the initial scan, the first
|
||
// status refresh and the two crash recoveries land before the R-52 sweep decides what "down" means.
|
||
var bootReconcileSettle = 5 * time.Second
|
||
|
||
// ── R-157 mechanism A: the window, and why these three numbers ───────────────────────────────────
|
||
//
|
||
// Until v0.190.0 the sweep looked exactly ONCE, at T+5 s, and returned. At T+5 s docker is still
|
||
// restoring containers after a hard reset, so an app that has not yet settled into a down state is
|
||
// not a candidate — and because the sweep never looked again, it stayed down. Measured failing on
|
||
// THREE OF SIX hard resets (CAMPAIGN-10). The predicate was never the problem; the single
|
||
// observation was.
|
||
//
|
||
// The fix is a window that SETTLES rather than a timer that runs forever (§5: an unbounded loop
|
||
// papers over a genuinely broken app and hammers docker). Three constants, each chosen against the
|
||
// 90 s dead-app boot grace:
|
||
//
|
||
// - bootReconcileSample = 5 s. Fine enough that a container settling at T+40 s is seen within one
|
||
// sample, coarse enough that a quiet boot costs ~12 cheap GetStacks() calls, not hundreds.
|
||
// - bootReconcileStableFor = 3 consecutive identical samples (15 s of no change) before the fleet
|
||
// is called settled. One sample cannot distinguish "settled" from "sampled between two docker
|
||
// events"; three spans the gap between a container exiting and its restart policy re-creating it.
|
||
// - bootReconcileBudget = 50 s. THE BINDING CONSTRAINT, and it is arithmetic, not taste:
|
||
// bootReconcileSettle (5 s) + budget (50 s) + ONE DefaultRetryDelay (30 s) inside the final
|
||
// sweep = 85 s, which must stay under deadAppBootGrace (90 s) so a recovery that works is
|
||
// SILENT. 60 s was the first choice and TestBootWindow_CommonCaseFitsInsideTheDeadAppGrace
|
||
// rejected it at 95 s — the test is the reason this number is 50 and not a round 60.
|
||
// Extending the grace to make a bigger budget fit was rejected (§8.3): that hides a late
|
||
// recovery rather than reporting it. A window that genuinely overruns is reported instead —
|
||
// see recordLateRecovery below.
|
||
//
|
||
// The window ends on WHICHEVER COMES FIRST — settled, or budget exhausted — and the log says which,
|
||
// because "settled and found nothing" and "ran out of time still churning" are different facts about
|
||
// the box and must not read the same (the v0.91.2 lesson).
|
||
var (
|
||
bootReconcileSample = 5 * time.Second
|
||
bootReconcileStableFor = 3
|
||
bootReconcileBudget = 50 * time.Second
|
||
)
|
||
|
||
// bootReconcileFn is the R-52 sweep, a package var purely so the wiring below is testable from
|
||
// package main (the v0.154.0 / v0.91.0 lesson: a seam proven only through injection proves the
|
||
// component and not the caller).
|
||
var bootReconcileFn = func(ctx context.Context, mgr bootrecon.StackProvider, logger *log.Logger) bootrecon.Result {
|
||
r := bootrecon.New(mgr, logger)
|
||
// R-171: the sweep must not start an app whose data drive is absent. Wired HERE, at the one
|
||
// place the sweep is constructed, so there is no path that builds an ungated reconciler.
|
||
if sm, ok := mgr.(*stacks.Manager); ok {
|
||
r.SetDriveGate(bootDriveGate{drive: driveStartGate{mgr: sm, sett: bootDriveSettings}})
|
||
}
|
||
return r.Run(ctx)
|
||
}
|
||
|
||
// bootDriveSettings / bootStartHolders are what the boot start gate reads. Init-only, set in main()
|
||
// before the reconcile goroutine is launched; both are nil-safe (see MayStart).
|
||
var (
|
||
bootDriveSettings *settings.Settings
|
||
bootQuiesceLoop *quiesce.Loop
|
||
bootAppStopGuard *backup.AppStopGuard
|
||
bootStackMgr *stacks.Manager
|
||
)
|
||
|
||
// bootDriveGate answers bootrecon.StartGate for the real controller. It enforces §8.2: an app that
|
||
// something else is deliberately holding must NOT be started by the boot sweep.
|
||
//
|
||
// THE THREE HOLDERS, in the order they are checked. The first two only became reachable when R-157
|
||
// mechanism A widened the window — the old T+5 s single sweep never overlapped a quiesce or a
|
||
// running app-data operation, and that is exactly why widening it needed these:
|
||
//
|
||
// 1. QUIESCE — the whole-guest backup loop stops app stacks and restarts exactly the ones it
|
||
// stopped. Starting one mid-backup would put a running app inside a snapshot that is supposed to
|
||
// be clean-shutdown-consistent, which is the entire point of quiescing. `SuppressedStacks` is
|
||
// the set it already publishes for precisely this "an app WE stopped is not a fault" question,
|
||
// so reusing it means the two cannot drift.
|
||
// 2. THE APP-STOP GUARD — a volume dump / offbox reconstitute / .fab export that is CURRENTLY
|
||
// holding an app down. Its own Recover already ran to completion before this goroutine started,
|
||
// so the marker seen here belongs to an operation running NOW, not to a crashed one.
|
||
// 3. THE DRIVE — R-171. Two questions, because they fail in opposite directions and neither alone
|
||
// is sufficient at boot:
|
||
// • the settings flags (`Disconnected`/`Decommissioned`) — the SAME signal the API's own
|
||
// `startGatedByMissingDrive` uses, so the customer's path and the sweep cannot disagree about
|
||
// whether an app may start. But they are the drive gate's bookkeeping, and inside the boot
|
||
// window that gate may not have ticked yet, so a genuinely absent drive can still read as
|
||
// connected.
|
||
// • `Manager.DriveLive` — the live mountpoint check, the SAME `isMountPoint` seam the userdata
|
||
// belt already uses. It answers immediately and needs no gate tick.
|
||
//
|
||
// Fail-safe per bootrecon.StartGate's contract: anything that cannot be determined returns false.
|
||
type bootDriveGate struct {
|
||
drive driveStartGate
|
||
}
|
||
|
||
func (g bootDriveGate) MayStart(stackName string) (bool, string) {
|
||
// 1. quiesce (nil-safe on an unprovisioned guest: SuppressedStacks returns nil)
|
||
if bootQuiesceLoop.SuppressedStacks()[stackName] {
|
||
return false, "a whole-guest backup (quiesce) is holding it — the quiesce loop restarts its own stacks"
|
||
}
|
||
// 1b. slice 4: a guarded update is moving this app (including one RecoverUpdates marked for
|
||
// resumption). The update job owns the bring-up and ends in healthy or held.
|
||
if bootStackMgr != nil && bootStackMgr.IsUpdating(stackName) {
|
||
return false, "a guarded update is in progress — the update brings it up and verifies it"
|
||
}
|
||
// 2. an app-data operation in flight
|
||
for _, held := range bootAppStopGuard.HeldStacks() {
|
||
if held == stackName {
|
||
return false, "an app-data operation is holding it — the app-stop guard restarts it when the operation ends"
|
||
}
|
||
}
|
||
// 3. the drive — the SHARED predicate, so this gate and the app-stop guard's crash recovery
|
||
// (R-174) cannot disagree about whether an app's drive is available.
|
||
return g.drive.MayStart(stackName)
|
||
}
|
||
|
||
// driveStartGate is holder #3 of bootDriveGate, EXTRACTED so it has two callers and one
|
||
// implementation (R-174).
|
||
//
|
||
// WHY IT IS SEPARATE FROM bootDriveGate RATHER THAN REUSED WHOLE. The app-stop guard's Recover needs
|
||
// exactly this question and NOT the other two holders:
|
||
//
|
||
// - Holder #2 reads `bootAppStopGuard.HeldStacks()`, which during Recover is the guard's OWN
|
||
// marker — the very stacks being recovered. Reusing bootDriveGate there would refuse every
|
||
// recovery it was meant to perform, self-referentially.
|
||
// - Holders #1 and #2 read package-level vars assigned in main() at the `bootDriveSettings` block,
|
||
// which runs AFTER `appStopGuard.Recover()`. They are nil at recovery time, so a whole-gate
|
||
// reuse would be correct only by accident of nil-safety — and would silently invert the moment
|
||
// anyone moved either line. This project has shipped that class of accident before.
|
||
//
|
||
// Fail-safe per bootrecon.StartGate's contract: anything that cannot be determined returns false.
|
||
type driveStartGate struct {
|
||
mgr *stacks.Manager
|
||
sett *settings.Settings
|
||
}
|
||
|
||
func (g driveStartGate) MayStart(stackName string) (bool, string) {
|
||
// R-379/R-380 HOLDER, checked FIRST and deliberately ABOVE the `HDD_PATH == ""` early return
|
||
// below. The apps this hold exists for are DRIVELESS — a failed database restore is the case,
|
||
// and 40 of the 53 catalogue apps have no drive at all — so a hold placed after that return
|
||
// would never be consulted for the exact class it was built for.
|
||
//
|
||
// This is holder #4, and putting it here rather than in bootDriveGate is what gives it two
|
||
// callers with one implementation: the boot sweep reaches it through bootDriveGate, and the
|
||
// app-stop guard's Recover reaches it directly. A hold only one path honours is not a hold.
|
||
if g.sett != nil {
|
||
if h, ok := g.sett.GetRestoreHold(stackName); ok {
|
||
if h.Reason == settings.HoldReasonUpdateFailed {
|
||
return false, "held after a failed update (" + h.At + ") — restore it from its backup to start it"
|
||
}
|
||
return false, "held after a failed restore whose rollback also failed (" + h.At + ") — clear the hold to start it"
|
||
}
|
||
}
|
||
if g.mgr == nil {
|
||
return false, "no stack manager wired — the drive cannot be determined"
|
||
}
|
||
cfg := g.mgr.LoadAppConfigByName(stackName)
|
||
if cfg == nil {
|
||
// CANNOT DETERMINE. A deployed app whose app.yaml will not load cannot have its drive
|
||
// resolved, so the fail-safe direction applies rather than a hopeful start.
|
||
return false, "app.yaml could not be read"
|
||
}
|
||
hdd := cfg.Env["HDD_PATH"]
|
||
if hdd == "" {
|
||
return true, "" // SSD-resident: there is no external drive to be absent (mirrors the API gate)
|
||
}
|
||
if g.sett != nil {
|
||
for _, sp := range g.sett.GetStoragePaths() {
|
||
if sp.Path != hdd {
|
||
continue
|
||
}
|
||
if sp.Decommissioned {
|
||
return false, "drive " + hdd + " is decommissioned"
|
||
}
|
||
if sp.Disconnected {
|
||
return false, "drive " + hdd + " is flagged disconnected"
|
||
}
|
||
}
|
||
}
|
||
if !g.mgr.DriveLive(hdd) {
|
||
return false, "drive " + hdd + " is not a live mountpoint"
|
||
}
|
||
return true, ""
|
||
}
|
||
|
||
// gatedAppStopStarter is the app-stop guard's starter, wrapped in the drive gate (R-174).
|
||
//
|
||
// THE DEFECT IT CLOSES, found by review on 2026-08-02 in code shipped 2026-08-01 (v0.189.0):
|
||
// `appStopGuard.SetStarter(stackMgr)` handed Recover the RAW stack manager, whose `StartStack` has
|
||
// no drive gate. Recover runs at startup — precisely when an external drive may not have come back —
|
||
// so a backup that stopped an app, followed by a power cut and a drive that did not remount, ended
|
||
// with the app started onto a missing drive. That is R-171 one path over, and the rule is not new:
|
||
// the API's own `startGatedByMissingDrive` already refuses this to the customer.
|
||
//
|
||
// The check stays HERE, per-caller, and is deliberately NOT pushed into `Manager.StartStack` — it has
|
||
// fourteen callers and most of them legitimately start apps outside this concern.
|
||
type gatedAppStopStarter struct {
|
||
inner backup.AppStopStarter
|
||
gate driveStartGate
|
||
logger *log.Logger
|
||
}
|
||
|
||
// WantsStopped (R-721, v0.283.1) forwards the household's intent to the crash recovery (see Recover).
|
||
func (s gatedAppStopStarter) WantsStopped(name string) bool {
|
||
if w, ok := s.inner.(interface{ WantsStopped(string) bool }); ok {
|
||
return w.WantsStopped(name)
|
||
}
|
||
return false
|
||
}
|
||
|
||
func (s gatedAppStopStarter) StartStack(name string) error {
|
||
if ok, why := s.gate.MayStart(name); !ok {
|
||
s.logger.Printf("[WARN] [appstop] refusing to restart %q after an interrupted operation: %s", name, why)
|
||
return fmt.Errorf("%w: %s", backup.ErrStartRefused, why)
|
||
}
|
||
return s.inner.StartStack(name)
|
||
}
|
||
|
||
// fillWatchStartupDelay is how long the startup fill check waits before its single run. Long enough
|
||
// for the drive gate's first reconcile and for mounts to settle — a drive still coming back must read
|
||
// as unreadable (and be skipped, §8.4), never as a filesystem worth warning about.
|
||
const fillWatchStartupDelay = 90 * time.Second
|
||
|
||
// fillTargets is §8.1's watch list: the app-data volume, the system-data volume, and every
|
||
// registered drive. Resolved at CHECK time, not at startup, so a drive added or decommissioned
|
||
// between checks is picked up without a controller restart.
|
||
//
|
||
// WHY THESE THREE AND NOT ONLY THE REGISTERED DRIVES — which is all the healthcheck ever looked at:
|
||
//
|
||
// - the APP-DATA volume (`mp0`, `/var/lib/docker`) holds every app's live named volumes;
|
||
// - the SYSTEM-DATA volume (`mp1`, `/mnt/sys_drive`) holds the retained recovery unit of every
|
||
// DRIVELESS app (architecture/07-backup-architecture.md §7.5) and is the smaller of the two at
|
||
// 20 G against 50 G — the very mismatch R-165 exists to remove. It is also the filesystem whose
|
||
// silent exhaustion R-158 measured;
|
||
// - the registered DRIVES hold the customer's own data.
|
||
//
|
||
// De-duplicated by path: on a box where a drive is not a separate mount these collapse, and warning
|
||
// twice about one filesystem is exactly the noise the per-filesystem rule exists to prevent. A
|
||
// DECOMMISSIONED drive is dropped — it is deliberately out of service, and its fill is not news. A
|
||
// DISCONNECTED one is kept in the list but will read as unreadable and be skipped by §8.4, which is
|
||
// the correct outcome: its absence already has its own alert.
|
||
func fillTargets(cfg *config.Config, sett *settings.Settings) []fillwatch.Target {
|
||
seen := make(map[string]bool)
|
||
var out []fillwatch.Target
|
||
add := func(path, label string) {
|
||
if path == "" || seen[path] {
|
||
return
|
||
}
|
||
seen[path] = true
|
||
out = append(out, fillwatch.Target{Path: path, Label: label})
|
||
}
|
||
|
||
// The docker data-root as seen from inside the guest, and the system-data area.
|
||
add(system.DockerVolumePath, "Alkalmazások területe")
|
||
if cfg != nil {
|
||
add(cfg.Paths.SystemDataPath, "Rendszer- és mentési terület")
|
||
}
|
||
if sett != nil {
|
||
for _, sp := range sett.GetStoragePaths() {
|
||
if sp.Decommissioned {
|
||
continue
|
||
}
|
||
add(sp.Path, sp.Label)
|
||
}
|
||
}
|
||
return fillwatch.SortTargets(out)
|
||
}
|
||
|
||
// bootFleetSample is a comparable snapshot of one deployed app — name, state and container count,
|
||
// per §8.1. Container count is in it deliberately: a stack can go from 3 containers to 0 without its
|
||
// aggregate state changing, and that IS the boot still moving.
|
||
type bootFleetSample struct {
|
||
name string
|
||
state string
|
||
containers int
|
||
}
|
||
|
||
// sampleBootFleet REFRESHES, then returns the fleet snapshot, sorted, so two samples compare by
|
||
// equality.
|
||
//
|
||
// THE REFRESH IS LOAD-BEARING, and it was found by live validation, not by review. `GetStacks()`
|
||
// returns the Manager's IN-MEMORY map, which the scheduler refreshes on its own 10 s cadence
|
||
// (`status-refresh`, main.go). Sampling it every 5 s without refreshing means two consecutive samples
|
||
// can straddle one refresh and be identical because THE CACHE DID NOT UPDATE — not because the fleet
|
||
// stopped moving. The window would then declare "settled" on stale data and sweep on a picture of the
|
||
// box from up to 10 s ago, which is a quieter version of the exact defect R-157 mechanism A is.
|
||
//
|
||
// Observed on guest 9201 on 2026-08-02: a container removed ~5 s before the window closed was still
|
||
// present in the sampled fleet, so the sweep logged "no boot-orphaned apps" for an app that had none.
|
||
//
|
||
// `RefreshStatus` is a cheap `docker ps`-based refresh of the in-memory map (the same one the
|
||
// scheduler runs every 10 s), so at most ~10 extra calls per boot. A refresh error is logged and the
|
||
// sample proceeds on whatever the map holds — degraded, but never silently: a boot window that cannot
|
||
// see docker is exactly when a stale verdict is most dangerous.
|
||
func sampleBootFleet(mgr bootrecon.StackProvider) []bootFleetSample {
|
||
_ = mgr.RefreshStatus()
|
||
stacksNow := mgr.GetStacks()
|
||
out := make([]bootFleetSample, 0, len(stacksNow))
|
||
for _, s := range stacksNow {
|
||
if !s.Deployed {
|
||
continue
|
||
}
|
||
out = append(out, bootFleetSample{name: s.Name, state: string(s.State), containers: len(s.Containers)})
|
||
}
|
||
sort.Slice(out, func(i, j int) bool { return out[i].name < out[j].name })
|
||
return out
|
||
}
|
||
|
||
func sameBootFleet(a, b []bootFleetSample) bool {
|
||
if len(a) != len(b) {
|
||
return false
|
||
}
|
||
for i := range a {
|
||
if a[i] != b[i] {
|
||
return false
|
||
}
|
||
}
|
||
return true
|
||
}
|
||
|
||
// runBootReconcile waits out the settle delay, then samples the fleet until it stops changing (or
|
||
// the budget runs out) and performs the bounded sweep ONCE, at the end.
|
||
//
|
||
// Sweeping on every sample was rejected: the sweep's own StartStack changes the fleet, so a
|
||
// sweep-per-sample would never observe a settled fleet and would race docker's restore. Sampling is
|
||
// read-only; exactly one sweep runs, and it re-derives its candidate set from a settled fleet —
|
||
// which is the whole point, since the pre-v0.190.0 bug was a candidate set derived too early.
|
||
//
|
||
// Called from main() in a goroutine.
|
||
func runBootReconcile(ctx context.Context, mgr bootrecon.StackProvider, logger *log.Logger) {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-time.After(bootReconcileSettle):
|
||
}
|
||
|
||
started := time.Now()
|
||
prev := sampleBootFleet(mgr)
|
||
stable := 1
|
||
settled := false
|
||
|
||
for time.Since(started) < bootReconcileBudget {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-time.After(bootReconcileSample):
|
||
}
|
||
cur := sampleBootFleet(mgr)
|
||
if sameBootFleet(prev, cur) {
|
||
stable++
|
||
} else {
|
||
// Not settled — the boot is still moving. Log at DEBUG: at 5 s cadence an INFO line per
|
||
// sample would bury the one line that matters, which is the verdict below.
|
||
logger.Printf("[DEBUG] [bootrecon] boot window: fleet still changing (%d app(s)) — resampling", len(cur))
|
||
stable = 1
|
||
}
|
||
prev = cur
|
||
if stable >= bootReconcileStableFor {
|
||
settled = true
|
||
break
|
||
}
|
||
}
|
||
|
||
if settled {
|
||
logger.Printf("[INFO] [bootrecon] boot window: fleet settled after %.0fs (%d identical samples %s apart) — sweeping",
|
||
time.Since(started).Seconds(), bootReconcileStableFor, bootReconcileSample)
|
||
} else {
|
||
// NOT a failure — a box whose apps are still churning at the budget is exactly the box that
|
||
// most needs the sweep. But it is a different fact from "settled", and saying so is what makes
|
||
// a stuck boot visible instead of looking like a quiet one.
|
||
logger.Printf("[INFO] [bootrecon] boot window: budget %s exhausted while the fleet was still changing — sweeping anyway",
|
||
bootReconcileBudget)
|
||
}
|
||
|
||
res := bootReconcileFn(ctx, mgr, logger)
|
||
recordLateRecovery(logger, started, res)
|
||
}
|
||
|
||
// recordLateRecovery keeps §8.3 honest. The window can finish AFTER deadAppBootGrace, and when it
|
||
// does the dead-app alarm has already fired for an app this sweep then recovered. Extending the
|
||
// grace to hide that was rejected; reporting it is the alternative, so the record is truthful and
|
||
// the operator is not left with a stale alarm and no counter-evidence.
|
||
//
|
||
// Measured from controller start, which is what the grace is measured from.
|
||
func recordLateRecovery(logger *log.Logger, started time.Time, res bootrecon.Result) {
|
||
if len(res.Recovered) == 0 {
|
||
return
|
||
}
|
||
elapsed := bootReconcileSettle + time.Since(started)
|
||
if elapsed <= deadAppBootGrace {
|
||
return
|
||
}
|
||
logger.Printf("[WARN] [bootrecon] LATE RECOVERY: %d app(s) recovered %.0fs after start, past the %s dead-app grace — an alert may already have fired for: %v",
|
||
len(res.Recovered), elapsed.Seconds(), deadAppBootGrace, res.Recovered)
|
||
}
|
||
|
||
// scanDeployedAppRunStates returns the fix-3 view of the deployed apps: the DEAD ones (for the
|
||
// state-based dashboard banner) and EVERY deployed app's run state (for the notifier's one-event-per-
|
||
// transition tracking). Deploying apps are skipped (mid-deploy is not a fault). Pure over GetStacks()
|
||
// — the derivation itself lives in classifyRunStates so it is testable without a live Manager.
|
||
func scanDeployedAppRunStates(mgr *stacks.Manager, q *quiesce.Loop, g *backup.AppStopGuard, held updateHeldLister) ([]web.DeadApp, []notify.AppRunState) {
|
||
// R-97b: a stack THIS controller stopped for a backup is not a fault. q may be nil (unprovisioned
|
||
// guest) — SuppressedStacks is nil-safe and returns nothing, i.e. suppress nothing.
|
||
//
|
||
// R-330: there are TWO mechanisms in this product that stop a customer's app on purpose, and until
|
||
// v0.224.0 only one of them told the alarm. `q` covers the WHOLE-GUEST quiesce loop (vzdump/PBS);
|
||
// `g` covers the per-app operations — the nightly volume dump, an off-site reconstitution and a
|
||
// .fab export. Both are nil-safe, and the union is taken here rather than inside classifyRunStates
|
||
// so that pure function keeps its single `quiesced` parameter and its existing tests.
|
||
// Slice 4: a THIRD mechanism moves an app on purpose — the guarded update recreates it and waits
|
||
// for health, and it ends in healthy or HELD. Counting it as dead mid-update would be R-330's false
|
||
// alarm one mechanism over.
|
||
suppressed := unionSuppressed(unionSuppressed(q.SuppressedStacks(), g.SuppressedStacks()), mgr.UpdatingStacks())
|
||
// R-660 (v0.268.0): a FOURTH — an app HELD after a failed update is stopped by the product and has
|
||
// its own event (`app_update_held`). Without this each hold was followed by an `app_start_failed`
|
||
// for the same app. Pinned by TestR660_UpdateHeldAppIsNotReportedDown + the wiring test beside it.
|
||
suppressed = unionSuppressed(suppressed, updateHeldSet(held))
|
||
return classifyRunStates(mgr.GetStacks(), suppressed, q.FailedRestarts(), time.Now())
|
||
}
|
||
|
||
// updateHeldLister is the backup manager's UpdateHeldStacks, as a seam for the test.
|
||
type updateHeldLister interface {
|
||
UpdateHeldStacks() map[string]bool
|
||
}
|
||
|
||
// updateHeldSet is nil-safe over a nil interface (a box with backup disabled has no holds).
|
||
func updateHeldSet(l updateHeldLister) map[string]bool {
|
||
if l == nil {
|
||
return nil
|
||
}
|
||
return l.UpdateHeldStacks()
|
||
}
|
||
|
||
// unionSuppressed merges the suppression sets of the two mechanisms that stop apps on purpose.
|
||
// Returns nil when both are empty so the common case allocates nothing.
|
||
func unionSuppressed(a, b map[string]bool) map[string]bool {
|
||
if len(a) == 0 {
|
||
return b
|
||
}
|
||
if len(b) == 0 {
|
||
return a
|
||
}
|
||
out := make(map[string]bool, len(a)+len(b))
|
||
for n := range a {
|
||
out[n] = true
|
||
}
|
||
for n := range b {
|
||
out[n] = true
|
||
}
|
||
return out
|
||
}
|
||
|
||
// classifyRunStates is the pure fix-3 derivation over a plain stack slice. It splits the deployed
|
||
// apps into the DEAD list (dashboard banner) and the per-app run states (notifier transition tracker).
|
||
//
|
||
// v0.164.0: a deliberate user stop is NOT a fault and must not alarm anywhere (banner OR email). The
|
||
// down predicate therefore excludes StateStopped — but NOT unconditionally. That exclusion rested on
|
||
// two invariants, and F-CRIT-1 (Campaign 8) showed the first of them had already become false:
|
||
//
|
||
// - I1 (AS WRITTEN, AND WRONG): "the UI stop path Manager.StopStack runs `docker compose down` →
|
||
// containers are removed → a deployed stack with zero containers aggregates to StateStopped, so
|
||
// StateStopped means 'deployed, deliberately stopped by the user'."
|
||
// **The quiesce loop stops stacks by the SAME path.** So a stack that a backup quiesce stopped and
|
||
// then FAILED to restart is also StateStopped — byte-identical to a user stop on the Docker side —
|
||
// and the whitelist swallowed it. Campaign 8 observed a customer app dead indefinitely with no
|
||
// banner, no event and no email, while the dead-app scanner ran 11 times over it.
|
||
// - I2 (still true): the P2 restart-policy census (2026-07-21, 53 templates / 78 services) found
|
||
// every catalog service on `unless-stopped`, so a CRASHING app never comes to rest at `stopped` —
|
||
// faults surface as StateExited / StateDegraded (and restarting/unhealthy).
|
||
//
|
||
// I1 RESTATED, TRUE AS OF v0.179.0: StateStopped means "deployed with zero containers", which is
|
||
// EITHER a deliberate user stop OR a quiesce restart that failed. No state test can tell them apart,
|
||
// because they are the same state. The distinguishing fact is not in the state at all — it is that
|
||
// the quiesce loop TRIED to restart the stack and could not, which it now reports via
|
||
// Loop.FailedRestarts(). `failedRestart` is that set, and it is the ONLY thing that lifts the
|
||
// StateStopped whitelist.
|
||
//
|
||
// (An out-of-band `docker compose stop` leaves the containers present → StateExited → still alerts,
|
||
// which is correct: out-of-band tampering IS reportable.) IsDownState is intentionally left unchanged
|
||
// — other callers rely on stopped counting as down; the suppression is a filter at this single
|
||
// derivation point only.
|
||
// `now` is injected (C9-F2) so the crash-loop threshold is a unit-testable contract rather than a
|
||
// property of the wall clock.
|
||
func classifyRunStates(sts []stacks.Stack, quiesced map[string]bool, failedRestart map[string]bool, now time.Time) ([]web.DeadApp, []notify.AppRunState) {
|
||
var dead []web.DeadApp
|
||
var states []notify.AppRunState
|
||
for _, st := range sts {
|
||
if !st.Deployed || st.Deploying {
|
||
continue
|
||
}
|
||
// R-97b: a stack a quiesce cycle stopped (or restarted within the grace window) is NOT down —
|
||
// we stopped it. This is a CYCLE-keyed suppression, not a state test, because an app caught
|
||
// mid-restart is `starting`/`unhealthy`, not StateStopped, so v0.164.0's state filter above
|
||
// cannot see it. The window EXPIRES (quiesceAlarmGrace): an app that genuinely fails to come
|
||
// back still alarms on the first scan after it closes.
|
||
// F-CRIT-1: StateStopped is whitelisted as a deliberate user stop UNLESS the quiesce loop
|
||
// reports that it stopped this stack and could not restart it. That single term is what turns
|
||
// an indefinitely-silent dead app back into an alarm, without re-alarming genuine user stops.
|
||
// C9-F2: a SUSTAINED restarting is a crash loop, and a crash loop is a dead app. Deliberately
|
||
// NOT folded into IsDownState — that would alarm on every deploy and update fleet-wide, which
|
||
// is the over-correction F-A1 nearly cost us. The threshold (stacks.crashLoopAfter, 5 min) sits
|
||
// above the deploy health timeout, the slowest catalog start_period AND R-97b's grace, so a
|
||
// brief restart never reaches it and the two suppression windows compose into one bounded
|
||
// delay. Quiesce suppression below still wins inside its own window.
|
||
crashLooping := st.CrashLooping(now)
|
||
|
||
// R-386: ASK THE FIELD THAT KNOWS, do not guess from the state.
|
||
//
|
||
// Until v0.223.0 this read `st.State == StateStopped && !failedRestart[...]` — i.e. EVERY
|
||
// stopped stack was assumed to be a deliberate customer stop. Measured on `demo-hp`
|
||
// 2026-08-23: `privatebin` stopped out of band, nine dead-app scans, zero events, zero
|
||
// banner. The comment above claimed an out-of-band stop "still alerts"; it did not, because
|
||
// `aggregateState` folds StateExited into the stopped counter and StateExited never survives
|
||
// aggregation.
|
||
//
|
||
// `DesiredState` is what the customer actually asked for, it has exactly ONE writer (the
|
||
// customer's own action — see the field's comment in internal/stacks/deploy.go), and it is
|
||
// TRI-state. The three-way ruling, operator-approved 2026-08-23:
|
||
//
|
||
// Stopped → the customer asked → no alarm (unchanged)
|
||
// Running → nobody asked → ALARM (the R-386 fix)
|
||
// absent → UNKNOWN, never "running" → no alarm, AND say so (see IntentUnknown)
|
||
//
|
||
// The absent case keeps today's behaviour deliberately. Reading unknown as "nobody asked"
|
||
// would, on the first cycle after this ships, email about every app any owner has ever
|
||
// deliberately stopped — fleet-wide, from a field that predates the intent it is being asked
|
||
// about. That is the same over-correction the tri-state exists to prevent, and the backfill
|
||
// cannot help: it seeds Running only from an observed-UP reading, so anything stopped at
|
||
// upgrade time stays unknown — which is exactly the ambiguous population.
|
||
//
|
||
// The gap is BOUNDED AND NAMED, not silent: IntentUnknown is set, the caller logs it at INFO
|
||
// with the app name, and an operator can therefore answer "how many apps am I blind to?".
|
||
// It closes itself as apps are started and stopped through the interface.
|
||
//
|
||
// `failedRestart` still lifts a Stopped intent, and that ordering is load-bearing: the
|
||
// quiesce loop stops stacks by the same path a customer does, so a stack it stopped and could
|
||
// NOT restart must alarm whatever the intent says. Removing that term re-opens F-CRIT-1's
|
||
// indefinitely-silent dead app.
|
||
intentUnknown := false
|
||
userStopped := false
|
||
if st.State == stacks.StateStopped && !failedRestart[st.Name] {
|
||
switch stacks.DesiredStateOf(st) {
|
||
case stacks.DesiredStateStopped:
|
||
userStopped = true
|
||
case stacks.DesiredStateRunning:
|
||
userStopped = false
|
||
default: // DesiredStateUnknown — the §4 fallback
|
||
userStopped = true
|
||
intentUnknown = true
|
||
}
|
||
}
|
||
|
||
down := (stacks.IsDownState(st.State) || crashLooping) && !userStopped && !quiesced[st.Name]
|
||
states = append(states, notify.AppRunState{
|
||
Name: st.Name, DisplayName: st.Meta.DisplayName, Down: down,
|
||
IntentUnknown: intentUnknown && !down,
|
||
})
|
||
if down {
|
||
dead = append(dead, web.DeadApp{Name: st.Name, DisplayName: st.Meta.DisplayName, State: string(st.State)})
|
||
}
|
||
}
|
||
return dead, states
|
||
}
|
||
|
||
func selfUpdateAuthMiddleware(cfg *config.Config, webServer *web.Server, next http.Handler) http.Handler {
|
||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||
// Check bearer token first (for external API calls: hub, build scripts)
|
||
if auth := r.Header.Get("Authorization"); strings.HasPrefix(auth, "Bearer ") {
|
||
token := strings.TrimPrefix(auth, "Bearer ")
|
||
if token != "" && cfg.Hub.APIKey != "" && subtle.ConstantTimeCompare([]byte(token), []byte(cfg.Hub.APIKey)) == 1 {
|
||
next.ServeHTTP(w, r)
|
||
return
|
||
}
|
||
}
|
||
// Fall back to session auth
|
||
webServer.RequireAuth(next).ServeHTTP(w, r)
|
||
})
|
||
}
|
||
|
||
// setupLogger builds the v0.116.0 capture layer: the LogBuffer ring ALWAYS exists
|
||
// and captures every line (DEBUG included — remote diagnostics need the detail to
|
||
// exist without a config flip), while stdout (docker logs) keeps respecting
|
||
// logging.level via the level filter. Lshortfile stays debug-only (unchanged).
|
||
func setupLogger(cfg *config.Config) (*log.Logger, *web.LogBuffer) {
|
||
// fix-6 (CAMPAIGN-3): 1000 wrapped in ~6.5 min under the campaign's ~2.6 entries/s load — the exact
|
||
// post-incident window an operator needs was the first thing lost. 5000 gives ≈32 min at that raw
|
||
// rate, and ≈50+ min now that the periodic-scheduler TRACE noise no longer enters the ring (2b).
|
||
logBuffer := web.NewLogBuffer(5000)
|
||
flags := log.LstdFlags
|
||
if cfg.Logging.Level == "debug" {
|
||
flags |= log.Lshortfile
|
||
}
|
||
stdout := web.NewLevelFilterWriter(os.Stdout, cfg.Logging.Level)
|
||
return log.New(io.MultiWriter(stdout, logBuffer), "", flags), logBuffer
|
||
}
|
||
|
||
// stackAdapter implements backup.StackDataProvider using stacks.Manager.
|
||
type stackAdapter struct {
|
||
mgr *stacks.Manager
|
||
getStoragePaths func() []settings.StoragePath
|
||
encKey []byte // for decrypting live app.yaml secrets during restore-from-unit
|
||
}
|
||
|
||
func (a *stackAdapter) GetStackComposePath(name string) (string, bool) {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return "", false
|
||
}
|
||
return s.ComposePath, true
|
||
}
|
||
|
||
// GetStackClassifiedBinds delegates to the stacks manager's backup-classification resolver (Task 2).
|
||
// INERT — no backup tier calls this yet; wired so Task 3 consumes a tested seam.
|
||
func (a *stackAdapter) GetStackClassifiedBinds(name string) ([]backup.ClassifiedBind, bool) {
|
||
return a.mgr.ClassifiedBinds(name)
|
||
}
|
||
|
||
func (a *stackAdapter) ListDeployedStacks() []backup.StackSummary {
|
||
var result []backup.StackSummary
|
||
for _, s := range a.mgr.GetStacks() {
|
||
// R-634 (v0.265.0): `Deployed` is set true in memory the moment a deploy is ACCEPTED (the
|
||
// no-stale-button UX), so without the second test a deploy still pulling was in every backup
|
||
// leg's list — and the volume leg stopped and restarted it in the middle of its own `up`.
|
||
if !s.Deployed || s.Deploying {
|
||
continue
|
||
}
|
||
result = append(result, backup.StackSummary{
|
||
Name: s.Name,
|
||
DisplayName: s.Meta.DisplayName,
|
||
ComposePath: s.ComposePath,
|
||
NeedsHDD: s.Meta.Resources.NeedsHDD,
|
||
HasVolumes: len(backup.ParseComposeNamedVolumes(s.ComposePath)) > 0,
|
||
})
|
||
}
|
||
return result
|
||
}
|
||
|
||
func (a *stackAdapter) StopStack(name string) error {
|
||
return a.mgr.StopStack(name)
|
||
}
|
||
|
||
// IsDeploying answers backup's optional deployingReporter (R-634): the volume leg asks it again
|
||
// immediately before it stops an app, because its list is taken once at the start of the run.
|
||
func (a *stackAdapter) IsDeploying(name string) bool {
|
||
s, ok := a.mgr.GetStack(name)
|
||
return ok && s.Deploying
|
||
}
|
||
|
||
// WantsStopped (R-721, v0.283.1): the backup package asks this before restarting an app it stopped for a
|
||
// volume dump. Measured live on 9202 2026-09-30 with v0.283.0: the method existed on the stacks manager but
|
||
// NOT on this adapter, so the dump's type assertion failed and a household Stop was undone 8 s later.
|
||
// Pinned by TestR721_TheBackupsStackProviderAnswersWantsStopped.
|
||
func (a *stackAdapter) WantsStopped(name string) bool { return a.mgr.WantsStopped(name) }
|
||
|
||
func (a *stackAdapter) StartStack(name string) error {
|
||
return a.mgr.StartStack(name)
|
||
}
|
||
|
||
func (a *stackAdapter) GetStackHDDMounts(name string) []string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return nil
|
||
}
|
||
|
||
// Priority 1: Read the app's own HDD_PATH from its app.yaml
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
appCfg := stacks.LoadAppConfig(stackDir)
|
||
if appCfg != nil && appCfg.Env["HDD_PATH"] != "" {
|
||
return stacks.ParseComposeHDDMounts(s.ComposePath, appCfg.Env["HDD_PATH"])
|
||
}
|
||
|
||
// Priority 2: Try all registered storage paths (fallback)
|
||
var allMounts []string
|
||
seen := make(map[string]bool)
|
||
for _, sp := range a.getStoragePaths() {
|
||
mounts := stacks.ParseComposeHDDMounts(s.ComposePath, sp.Path)
|
||
for _, m := range mounts {
|
||
if !seen[m] {
|
||
seen[m] = true
|
||
allMounts = append(allMounts, m)
|
||
}
|
||
}
|
||
}
|
||
return allMounts
|
||
}
|
||
|
||
func (a *stackAdapter) GetDockerVolumes(name string) []string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return nil
|
||
}
|
||
return backup.ResolveDockerVolumeNames(s.ComposePath)
|
||
}
|
||
|
||
// GetImportRoot delegates to the stack manager's canonical drop-zone root (R-75). Both the backup
|
||
// and the export provider need it: ${IMPORT_PATH} binds live on the SYSTEM drive and can NOT be
|
||
// resolved from an app's HDD_PATH.
|
||
func (a *stackAdapter) GetImportRoot() string { return a.mgr.GetImportRoot() }
|
||
|
||
func (a *stackAdapter) GetStackHDDPath(name string) string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return ""
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
appCfg := stacks.LoadAppConfig(stackDir)
|
||
if appCfg != nil && appCfg.Env["HDD_PATH"] != "" {
|
||
return filepath.Clean(appCfg.Env["HDD_PATH"])
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// GetStackRecoveryInfo gathers the inputs for an app's recovery unit: the stack dir, pinned image
|
||
// tags, the non-secret env, the NAMES of every secret/data-key env var, and — since D5 — the decrypted
|
||
// VALUES of the PORTABLE secret class.
|
||
//
|
||
// The non-secret env still excludes every named secret (plus a defensive crypto.IsEncrypted guard), so
|
||
// the two sets are disjoint by construction and a secret can only reach the unit by being in the
|
||
// portable class. The EXCLUDED class (`type: password`, and the nonPortableSecrets register) is
|
||
// name-only here and is recovered from the guest — or regenerated (O4) — exactly as before.
|
||
func (a *stackAdapter) GetStackRecoveryInfo(name string) (backup.RecoveryInfo, bool) {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return backup.RecoveryInfo{}, false
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
meta := stacks.LoadMetadata(stackDir)
|
||
|
||
// Secret set = all secret/password fields ∪ any data_key fields (in deterministic metadata order).
|
||
secretSet := make(map[string]bool)
|
||
var secretNames []string
|
||
add := func(v string) {
|
||
if !secretSet[v] {
|
||
secretSet[v] = true
|
||
secretNames = append(secretNames, v)
|
||
}
|
||
}
|
||
for _, v := range stacks.SensitiveEnvVars(&meta) {
|
||
add(v)
|
||
}
|
||
dataKeys := meta.DataKeyEnvVars()
|
||
for _, v := range dataKeys {
|
||
add(v)
|
||
}
|
||
|
||
// Non-secret env: raw app.yaml values that are neither named-secret nor (defensively) encrypted.
|
||
nonSecret := make(map[string]string)
|
||
if appCfg := stacks.LoadAppConfig(stackDir); appCfg != nil {
|
||
for k, v := range appCfg.Env {
|
||
if secretSet[k] || crypto.IsEncrypted(v) {
|
||
continue
|
||
}
|
||
nonSecret[k] = v
|
||
}
|
||
}
|
||
|
||
// D5: decrypt the portable class so it can travel in the unit. Reuses the SAME decrypt path as the
|
||
// restore side (LoadAppConfigDecrypted), so there is one way to turn app.yaml into plaintext, not
|
||
// two. A name whose value is absent/empty is simply omitted — the restore's fail-closed data-key
|
||
// gate is what decides whether that is survivable.
|
||
portableNames := stacks.PortableSecretEnvVars(&meta)
|
||
portable := make(map[string]string, len(portableNames))
|
||
if len(portableNames) > 0 {
|
||
if dec := stacks.LoadAppConfigDecrypted(stackDir, a.encKey); dec != nil {
|
||
for _, n := range portableNames {
|
||
if v, ok := dec.Env[n]; ok && v != "" {
|
||
portable[n] = v
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
return backup.RecoveryInfo{
|
||
StackDir: stackDir,
|
||
DisplayName: s.Meta.DisplayName,
|
||
ImagePins: backup.ParseComposeImages(s.ComposePath),
|
||
NonSecretEnv: nonSecret,
|
||
SecretEnvVars: secretNames,
|
||
DataKeyEnvVars: dataKeys,
|
||
PortableSecretEnvVars: portableNames,
|
||
PortableSecrets: portable,
|
||
InstalledImages: installedRefs(s.AppConfig),
|
||
}, true
|
||
}
|
||
|
||
// installedRefs renders app.yaml's `installed_images` observation as service -> `ref@digest` (v0.275.0,
|
||
// R-696) for the data stamps. Nil when nothing was observed — absent means UNKNOWN (deploy.go).
|
||
func installedRefs(cfg *stacks.AppConfig) map[string]string {
|
||
if cfg == nil || len(cfg.InstalledImages) == 0 {
|
||
return nil
|
||
}
|
||
out := make(map[string]string, len(cfg.InstalledImages))
|
||
for svc, im := range cfg.InstalledImages {
|
||
ref := im.Ref
|
||
if im.Digest != "" {
|
||
ref += "@" + im.Digest
|
||
}
|
||
out[svc] = ref
|
||
}
|
||
return out
|
||
}
|
||
|
||
// RecoverStackSecrets returns the live decrypted values for the named secret env vars present in the
|
||
// stack's app.yaml (the guest's own — live rootfs or PBS-restored). Absent/empty names are omitted;
|
||
// the caller's fail-closed gate decides. Secrets come from the guest, never from the recovery unit.
|
||
func (a *stackAdapter) RecoverStackSecrets(name string, names []string) map[string]string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return nil
|
||
}
|
||
cfg := stacks.LoadAppConfigDecrypted(filepath.Dir(s.ComposePath), a.encKey)
|
||
if cfg == nil {
|
||
return nil
|
||
}
|
||
out := make(map[string]string)
|
||
for _, n := range names {
|
||
if v, ok := cfg.Env[n]; ok && v != "" {
|
||
out[n] = v
|
||
}
|
||
}
|
||
return out
|
||
}
|
||
|
||
// RecreateStackDefinitionFromUnit restores the app definition from the unit's compose dir into the
|
||
// stack dir and persists app.yaml from the reconstructed full env. Secrets in fullEnv were recovered
|
||
// from the guest, never regenerated. It starts NOTHING — the restore flow brings the database service
|
||
// up alone for the dump replay and only then starts the whole stack (R-47).
|
||
func (a *stackAdapter) RecreateStackDefinitionFromUnit(name, composeSrcDir string, fullEnv map[string]string) error {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return fmt.Errorf("stack %q not found", name)
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
// Recover the app definition from the unit (compose + .felhom.yml) into the stack dir.
|
||
for _, fname := range []string{"docker-compose.yml", ".felhom.yml"} {
|
||
data, err := os.ReadFile(filepath.Join(composeSrcDir, fname))
|
||
if err != nil {
|
||
continue // capture whichever existed
|
||
}
|
||
if err := os.WriteFile(filepath.Join(stackDir, fname), data, 0644); err != nil {
|
||
return fmt.Errorf("restoring %s from unit: %w", fname, err)
|
||
}
|
||
}
|
||
if err := a.mgr.PersistUnitRedeployConfig(name, fullEnv); err != nil {
|
||
return err
|
||
}
|
||
|
||
// v0.235.0 — PIN TO WHAT THE UNIT CAPTURED. THIS IS WHAT CLOSES R-441.
|
||
//
|
||
// Measured before this line existed: this function writes the recovery unit's compose — carrying
|
||
// the OLD image pin — into the live stack dir, and `Syncer.copyIfChanged` then overwrote it from
|
||
// the catalog on the next 15-minute tick. So a restore's image-level recovery had a <=15-minute
|
||
// half-life, and the next `compose up -d` from any of the thirteen call sites re-applied the
|
||
// catalog's version. Pinning here makes the render obey the restored definition instead.
|
||
//
|
||
// Pinned from the file just written, so the pin and the stored definition cannot disagree.
|
||
// A failure is loud and does NOT fail the restore: an unpinned app is exactly today's behaviour,
|
||
// and refusing a completed restore over a bookkeeping write would be the worse trade.
|
||
if pin, data, err := stacks.PinFromCompose(stacks.ComposePathIn(stackDir)); err != nil {
|
||
log.Printf("[WARN] [stacks] pin %s: cannot pin from the restored definition: %v", name, err)
|
||
} else if err := a.mgr.SetPin(name, stackDir, pin, data); err != nil {
|
||
log.Printf("[ERROR] [stacks] pin %s: %v", name, err)
|
||
}
|
||
// R-669 (v0.270.0): the restored .felhom.yml IS the pinned version's record now.
|
||
a.mgr.RecordRestoredAppliedMeta(name, stackDir)
|
||
return nil
|
||
}
|
||
|
||
// StartStackServices brings up only the named compose services (the DB-only replay window, R-47).
|
||
func (a *stackAdapter) StartStackServices(name string, services []string) error {
|
||
return a.mgr.StartStackServices(name, services)
|
||
}
|
||
|
||
// RefreshAndIsRunning forces a docker ps scan before checking state.
|
||
// Called during post-restore health check (~every 5s for up to 90s).
|
||
// Full refresh is acceptable here since restores are rare operations.
|
||
func (a *stackAdapter) RefreshAndIsRunning(name string) bool {
|
||
a.mgr.RefreshStatus()
|
||
s, ok := a.mgr.GetStack(name)
|
||
return ok && s.State == stacks.StateRunning
|
||
}
|
||
|
||
// integrationStackAdapter implements integrations.StackProvider using stacks.Manager.
|
||
type integrationStackAdapter struct {
|
||
mgr *stacks.Manager
|
||
}
|
||
|
||
func (a *integrationStackAdapter) GetStack(name string) (*stacks.Stack, bool) {
|
||
return a.mgr.GetStack(name)
|
||
}
|
||
|
||
func (a *integrationStackAdapter) GetStacks() []stacks.Stack {
|
||
return a.mgr.GetStacks()
|
||
}
|
||
|
||
func (a *integrationStackAdapter) RestartStack(name string) error {
|
||
return a.mgr.RestartStack(name)
|
||
}
|
||
|
||
// geoStackAdapter implements cloudflare.StackLister for geo-restriction sync.
|
||
type geoStackAdapter struct {
|
||
mgr *stacks.Manager
|
||
domain string
|
||
}
|
||
|
||
func (a *geoStackAdapter) GetDeployedHostnames() map[string]string {
|
||
result := make(map[string]string)
|
||
for _, stack := range a.mgr.GetStacks() {
|
||
if !stack.Deployed {
|
||
continue
|
||
}
|
||
subdomain := stack.Meta.Subdomain
|
||
// Check for custom subdomain in app.yaml
|
||
if appCfg := a.mgr.LoadAppConfigByName(stack.Name); appCfg != nil {
|
||
if sd, ok := appCfg.Env["SUBDOMAIN"]; ok && sd != "" {
|
||
subdomain = sd
|
||
}
|
||
}
|
||
if subdomain != "" {
|
||
result[stack.Name] = subdomain + "." + a.domain
|
||
}
|
||
}
|
||
return result
|
||
}
|
||
|
||
// exportAdapter implements appexport.ExportStackProvider using stacks.Manager.
|
||
type exportAdapter struct {
|
||
mgr *stacks.Manager
|
||
encKey []byte
|
||
}
|
||
|
||
func (a *exportAdapter) GetStackDir(name string) (string, bool) {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return "", false
|
||
}
|
||
return filepath.Dir(s.ComposePath), true
|
||
}
|
||
|
||
// GetStackClassifiedBinds delegates to the stacks manager's backup-classification resolver (Task 2),
|
||
// used by the `.fab` class-scoped export plan (Task 4).
|
||
func (a *exportAdapter) GetStackClassifiedBinds(name string) ([]appbackup.ClassifiedBind, bool) {
|
||
return a.mgr.ClassifiedBinds(name)
|
||
}
|
||
|
||
func (a *exportAdapter) GetStackComposePath(name string) (string, bool) {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return "", false
|
||
}
|
||
return s.ComposePath, true
|
||
}
|
||
|
||
func (a *exportAdapter) GetStackHDDMounts(name string) []string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return nil
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
appCfg := stacks.LoadAppConfig(stackDir)
|
||
if appCfg != nil && appCfg.Env["HDD_PATH"] != "" {
|
||
// C6B-F1 (v0.130.0): union ${HDD_PATH} binds + the ${USERDATA_PATH} root. The old
|
||
// ParseComposeHDDMounts-only call was blind to the standard userdata convention, so
|
||
// 12/13 needs_hdd catalog apps exported hollow (config-only) bundles. The backup-side
|
||
// stackAdapter is intentionally NOT changed here — the scheduled/tier-2 path copies the
|
||
// recovery unit + the app's resolved appdata/<name> dir(s) only (NOT the userdata tree —
|
||
// F-S1; NOT the namespace wholesale), and derives that dir name from the compose binds
|
||
// (F-S2, see backup.tier2AppDataName). Spec:
|
||
// felhom.eu/documentation/audits/SPIKE-backup-classification-2026-07-14.md.
|
||
return stacks.ExportDataMounts(s.ComposePath, appCfg.Env["HDD_PATH"], a.mgr.StackNamespaceRoot(name))
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// GetImportRoot delegates to the stack manager's canonical drop-zone root (R-75). Both the backup
|
||
// and the export provider need it: ${IMPORT_PATH} binds live on the SYSTEM drive and can NOT be
|
||
// resolved from an app's HDD_PATH.
|
||
func (a *exportAdapter) GetImportRoot() string { return a.mgr.GetImportRoot() }
|
||
|
||
// GetStackNamespaceRoot (R-203) — the felhom-data namespace root, NOT the drive path. The appbackup
|
||
// path helpers all take this; passing HDD_PATH straight in is what made the export plan and the
|
||
// off-site capture set describe different directories on the system-data fallback.
|
||
func (a *exportAdapter) GetStackNamespaceRoot(name string) string {
|
||
return a.mgr.StackNamespaceRoot(name)
|
||
}
|
||
|
||
func (a *exportAdapter) GetStackHDDPath(name string) string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return ""
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
appCfg := stacks.LoadAppConfig(stackDir)
|
||
if appCfg != nil && appCfg.Env["HDD_PATH"] != "" {
|
||
return filepath.Clean(appCfg.Env["HDD_PATH"])
|
||
}
|
||
return ""
|
||
}
|
||
|
||
func (a *exportAdapter) IsStackRunning(name string) bool {
|
||
s, ok := a.mgr.GetStack(name)
|
||
// StateDegraded (R-51) counts as running: the export must stop the still-live members before
|
||
// reading their volumes, exactly as it would for a fully running stack.
|
||
return ok && (s.State == stacks.StateRunning || s.State == stacks.StateDegraded)
|
||
}
|
||
|
||
func (a *exportAdapter) StopStack(name string) error {
|
||
return a.mgr.StopStack(name)
|
||
}
|
||
|
||
func (a *exportAdapter) StartStack(name string) error {
|
||
return a.mgr.StartStack(name)
|
||
}
|
||
|
||
func (a *exportAdapter) GetStackDisplayName(name string) string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return name
|
||
}
|
||
return s.Meta.DisplayName
|
||
}
|
||
|
||
func (a *exportAdapter) GetStackNeedsHDD(name string) bool {
|
||
s, ok := a.mgr.GetStack(name)
|
||
return ok && s.Meta.Resources.NeedsHDD
|
||
}
|
||
|
||
func (a *exportAdapter) GetDockerVolumes(name string) []string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return nil
|
||
}
|
||
return backup.ResolveDockerVolumeNames(s.ComposePath)
|
||
}
|
||
|
||
func (a *exportAdapter) IsStackDeployed(name string) bool {
|
||
s, ok := a.mgr.GetStack(name)
|
||
return ok && s.Deployed
|
||
}
|
||
|
||
func (a *exportAdapter) GetDecryptedEnv(name string) map[string]string {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return nil
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
cfg := stacks.LoadAppConfigDecrypted(stackDir, a.encKey)
|
||
if cfg == nil {
|
||
return nil
|
||
}
|
||
return cfg.Env
|
||
}
|
||
|
||
func (a *exportAdapter) GetStacksBaseDir() string {
|
||
return a.mgr.GetStacksBaseDir()
|
||
}
|
||
|
||
// exportStopGuard adapts *backup.AppStopGuard to the exporter's reason-free seam (R-166). The reason
|
||
// is supplied HERE rather than passed in, so backup.ReasonAppExport's value exists in exactly one
|
||
// place and the two packages cannot drift apart.
|
||
type exportStopGuard struct{ g *backup.AppStopGuard }
|
||
|
||
func (a exportStopGuard) Begin(opID string, stacks []string) error {
|
||
return a.g.Begin(opID, backup.ReasonAppExport, stacks)
|
||
}
|
||
|
||
func (a exportStopGuard) End() { a.g.End() }
|
||
|
||
func (a exportStopGuard) ReleaseFailed(stacks ...string) { a.g.ReleaseFailed(stacks...) }
|
||
|
||
func (a *exportAdapter) SaveEncryptedAppConfig(stackDir string, env map[string]string) error {
|
||
meta := stacks.LoadMetadata(stackDir)
|
||
sensitiveVars := stacks.SensitiveEnvVars(&meta)
|
||
cfg := &stacks.AppConfig{
|
||
Deployed: true,
|
||
DeployedAt: time.Now().Format(time.RFC3339),
|
||
Env: env,
|
||
// R-166 — a CUSTOMER-INTENT POINT, and the one that is not the API action switch. Importing
|
||
// a `.fab` bundle is the customer installing that app on this box, and the import path starts
|
||
// it (appexport/restore.go). Without this the app would come back from a restore with NO
|
||
// recorded intent and fall to legacy boot behaviour — meaning a power cut days later would
|
||
// strand it, which is exactly the failure this release exists to remove.
|
||
DesiredState: stacks.DesiredStateRunning,
|
||
}
|
||
return stacks.SaveAppConfig(stackDir, cfg, a.encKey, sensitiveVars)
|
||
}
|
||
|
||
func (a *exportAdapter) RefreshStacks() error {
|
||
return a.mgr.RefreshStatus()
|
||
}
|
||
|
||
func (a *exportAdapter) RemoveStackVolumes(name string) error {
|
||
s, ok := a.mgr.GetStack(name)
|
||
if !ok {
|
||
return fmt.Errorf("stack %q not found", name)
|
||
}
|
||
stackDir := filepath.Dir(s.ComposePath)
|
||
|
||
// Build env from decrypted app config
|
||
cmdEnv := os.Environ()
|
||
appCfg := stacks.LoadAppConfigDecrypted(stackDir, a.encKey)
|
||
if appCfg != nil {
|
||
for k, v := range appCfg.Env {
|
||
cmdEnv = append(cmdEnv, k+"="+v)
|
||
}
|
||
}
|
||
|
||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||
defer cancel()
|
||
cmd := dockerexec.CommandContext(ctx, "docker", "compose", "down", "--volumes")
|
||
cmd.Dir = stackDir
|
||
cmd.Env = cmdEnv
|
||
out, err := cmd.CombinedOutput()
|
||
if err != nil {
|
||
return fmt.Errorf("compose down --volumes: %s — %w", strings.TrimSpace(string(out)), err)
|
||
}
|
||
return nil
|
||
}
|
||
|
||
// fileExists returns true if the path exists (file or directory).
|
||
func fileExists(path string) bool {
|
||
_, err := os.Stat(path)
|
||
return err == nil
|
||
}
|
||
|
||
// quiesceBackend adapts *agentapi.Client to quiesce.Backend (bool/string, decoupled from the
|
||
// agentapi response structs).
|
||
type quiesceBackend struct{ c *agentapi.Client }
|
||
|
||
func (b quiesceBackend) Due(ctx context.Context) (bool, *int64, error) {
|
||
r, err := b.c.BackupDue(ctx)
|
||
return r.Due, r.AgeSecs, err
|
||
}
|
||
func (b quiesceBackend) StartBackup(ctx context.Context) (string, error) {
|
||
r, err := b.c.StartBackup(ctx)
|
||
return r.JobID, mapBusy(err) // F-A1: the untargeted (pre-R-82) path can 409 identically
|
||
}
|
||
func (b quiesceBackend) BackupStatus(ctx context.Context) (string, error) {
|
||
r, err := b.c.BackupStatus(ctx)
|
||
return r.Phase, err
|
||
}
|
||
|
||
// ---- R-82: the tiered surface (quiesce.TieredBackend) ------------------------------------
|
||
//
|
||
// quiesceBackend satisfies quiesce.TieredBackend as well, so the loop schedules per tier when the
|
||
// agent supports it. Against a PRE-R-82 agent, Tiers returns quiesce.ErrTiersUnsupported and the
|
||
// loop degrades to the untargeted methods above — still taking a backup, never skipping one.
|
||
|
||
func (b quiesceBackend) Tiers(ctx context.Context) ([]quiesce.BackupTier, error) {
|
||
r, err := b.c.BackupTiers(ctx)
|
||
if errors.Is(err, agentapi.ErrTiersUnsupported) {
|
||
// Translate the transport-layer probe into the loop's vocabulary; the loop keys on this.
|
||
return nil, quiesce.ErrTiersUnsupported
|
||
}
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
out := make([]quiesce.BackupTier, 0, len(r.Tiers))
|
||
for _, t := range r.Tiers {
|
||
out = append(out, quiesce.BackupTier{Target: t.Target, Primary: t.Primary, StorageAbsent: t.Storage == "absent"})
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
func (b quiesceBackend) DueFor(ctx context.Context, target string) (bool, *int64, string, error) {
|
||
r, err := b.c.BackupDueFor(ctx, target)
|
||
// AgeState is passed through RAW; quiesce.ageStateFromWire owns the mapping, including the
|
||
// legacy-vs-unknown distinction. An empty string here means a pre-v0.105.0 agent.
|
||
return r.Due, r.AgeSecs, r.AgeState, err
|
||
}
|
||
func (b quiesceBackend) StartBackupFor(ctx context.Context, target string) (string, error) {
|
||
r, err := b.c.StartBackupFor(ctx, target)
|
||
// F-A1: translate the agent's HTTP 409 into the loop's vocabulary, exactly as Tiers translates a
|
||
// 404 into ErrTiersUnsupported. 409 means the agent's R-85 single-flight gate refused because a
|
||
// restore-test (or another backup) holds it — CONTENTION, not failure. `internal/quiesce` keeps
|
||
// no dependency on `internal/agentapi`, so the mapping belongs here at the seam.
|
||
return r.JobID, mapBusy(err)
|
||
}
|
||
|
||
// mapBusy converts a 409 from the agent into quiesce.ErrTierBusy, preserving the original error for
|
||
// diagnosis. Any other status passes through untouched — treating a real 5xx as contention would
|
||
// swallow genuine failures, which is the over-correction F-A1's fix must not make.
|
||
func mapBusy(err error) error {
|
||
if err == nil {
|
||
return nil
|
||
}
|
||
var se *agentapi.StatusError
|
||
if errors.As(err, &se) && se.Code == http.StatusConflict {
|
||
return fmt.Errorf("%w: %v", quiesce.ErrTierBusy, err)
|
||
}
|
||
return err
|
||
}
|
||
func (b quiesceBackend) BackupStatusFor(ctx context.Context, target string) (string, error) {
|
||
r, err := b.c.BackupStatusFor(ctx, target)
|
||
return r.Phase, err
|
||
}
|
||
|
||
// quiesceTierNotifier adapts *notify.Notifier to quiesce.TierNotifier (R-97a).
|
||
//
|
||
// The whole-guest tier had NO route to the hub at all — `internal/quiesce` did not import
|
||
// `internal/notify`, so on 2026-07-27 three failed backups and twelve app-stack stop/starts produced
|
||
// ZERO `backup_failed` events. The event type was already in the hub's allowlist and
|
||
// `NotifyBackupFailed` already existed; only this adapter and its wiring were missing.
|
||
//
|
||
// The tier is carried in the MESSAGE rather than a new event type, because the hub gates on
|
||
// `allowedEventTypes` and a per-tier type would need a hub-side change to be deliverable at all.
|
||
// See the cooldown note in REPORT.md: the hub's operator cooldown is keyed
|
||
// `customerID + ":" + eventType`, so two tiers failing within an hour share one key — both events
|
||
// are STORED, but only the first sends an operator email.
|
||
type quiesceTierNotifier struct{ n *notify.Notifier }
|
||
|
||
func (q quiesceTierNotifier) BackupFailed(tier, message, errMsg string) {
|
||
q.n.NotifyWholeGuestBackupFailed(tier, message, errMsg)
|
||
}
|
||
|
||
func (q quiesceTierNotifier) BackupRecovered(tier, message string) {
|
||
q.n.NotifyWholeGuestBackupRecovered(tier, message)
|
||
}
|
||
|
||
// BackupTierSkipped satisfies quiesce.TierSkipNotifier (R-518).
|
||
func (q quiesceTierNotifier) BackupTierSkipped(tier, message string) {
|
||
q.n.NotifyBackupTierSkipped(tier, message)
|
||
}
|
||
|
||
// startQuiesceLoop wires + starts the slice-8B quiesce loop when the local API is configured and
|
||
// quiesce is enabled. It Recovers (restarts stacks left stopped by a mid-quiesce crash) before
|
||
// starting the loop goroutine. Non-fatal: any misconfig disables the loop with a log line.
|
||
func startQuiesceLoop(ctx context.Context, cfg *config.Config, sett *settings.Settings, stackMgr *stacks.Manager, logger *log.Logger) *quiesce.Loop {
|
||
if cfg.LocalAPI.Endpoint == "" || cfg.LocalAPI.Token == "" {
|
||
return nil // not a provisioned guest — no agent to back up against
|
||
}
|
||
if !cfg.Quiesce.QuiesceEnabled() {
|
||
logger.Printf("[INFO] [quiesce] disabled by config")
|
||
return nil
|
||
}
|
||
client, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint)
|
||
if err != nil {
|
||
logger.Printf("[WARN] [quiesce] disabled (agent client init failed): %v", err)
|
||
return nil
|
||
}
|
||
poll := parseDurationOr(cfg.Quiesce.PollInterval, 5*time.Minute)
|
||
statusPoll := parseDurationOr(cfg.Quiesce.StatusPoll, 10*time.Second)
|
||
maxQuiesce := parseDurationOr(cfg.Quiesce.MaxQuiesce, 30*time.Minute)
|
||
loop := quiesce.New(quiesce.Options{
|
||
Backend: quiesceBackend{c: client},
|
||
Stacks: stackMgr,
|
||
MarkerPath: filepath.Join(cfg.Paths.DataDir, "quiesce-state.json"),
|
||
Poll: poll,
|
||
StatusPoll: statusPoll,
|
||
MaxQuiesce: maxQuiesce,
|
||
Logger: logger,
|
||
// Window gate (v0.168.0): read the effective window fresh each poll so a customer change takes
|
||
// effect without restart. Scheduled cycles run only inside [W+2h, W+6h), with the safety valve.
|
||
WindowStartFn: func() string {
|
||
return backupwindow.EffectiveWindow(sett.GetBackupWindowStart(), cfg.Backup.DBDumpSchedule)
|
||
},
|
||
// v0.271.0 (`09` decision 20): the full-system backup waits for the automatic update leg.
|
||
UpdateLegFn: stackMgr.UpdateLegState,
|
||
})
|
||
loop.Recover() // crash-safety: restart any stacks stranded-down by a mid-quiesce crash
|
||
go loop.Run(ctx)
|
||
return loop
|
||
}
|
||
|
||
// parseDurationOr parses a duration string, falling back to def on empty/invalid input.
|
||
func parseDurationOr(s string, def time.Duration) time.Duration {
|
||
if s == "" {
|
||
return def
|
||
}
|
||
d, err := time.ParseDuration(s)
|
||
if err != nil || d <= 0 {
|
||
return def
|
||
}
|
||
return d
|
||
}
|
||
|
||
// channelSink adapts the notify.Notifier + web.AlertManager to the channelhealth.Sink seam (kept in
|
||
// main.go so the channelhealth package imports neither — no cycle). SetDashboard reflects the current
|
||
// state each probe (idempotent); NotifyDown/NotifyRecovered fire only on a transition.
|
||
type channelSink struct {
|
||
notifier *notify.Notifier
|
||
alertMgr *web.AlertManager
|
||
}
|
||
|
||
// R-573: the checker already CLASSIFIES the fault, so the banner is keyed by that classification
|
||
// rather than by the sentence it happened to compose. An unmapped reason passes an empty key and the
|
||
// AlertManager falls back to the composed sentence — never a blank banner.
|
||
func (s channelSink) SetDashboard(down bool, reason channelhealth.Reason, msg string) {
|
||
key := ""
|
||
if down && reason != "" {
|
||
key = "alert.agent_channel." + string(reason)
|
||
}
|
||
s.alertMgr.SetAgentChannelAlert(down, key, msg)
|
||
}
|
||
func (s channelSink) NotifyDown(reason channelhealth.Reason, eventType, severity, msg string) {
|
||
s.notifier.NotifyAgentChannelDown(string(reason), eventType, severity, msg)
|
||
}
|
||
func (s channelSink) NotifyRecovered() { s.notifier.NotifyAgentChannelRecovered() }
|
||
|
||
// probeLocalAPI proves the controller↔agent local-API channel at startup and logs this guest's
|
||
// mounts (slice 8A). Non-fatal: it only runs when a local-API endpoint is configured, and any
|
||
// error is logged for diagnosis without affecting the controller's boot. The leaf SHA-256 from
|
||
// the bootstrap is pinned by the client (fails closed on mismatch).
|
||
func probeLocalAPI(cfg *config.Config, logger *log.Logger) {
|
||
if cfg.LocalAPI.Endpoint == "" || cfg.LocalAPI.Token == "" {
|
||
return
|
||
}
|
||
client, err := agentapi.New(cfg.LocalAPI.Endpoint, cfg.LocalAPI.Token, cfg.LocalAPI.Fingerprint)
|
||
if err != nil {
|
||
logger.Printf("[WARN] local-api: client init failed (%v) — channel not verified", err)
|
||
return
|
||
}
|
||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||
defer cancel()
|
||
resp, err := client.Storage(ctx)
|
||
if err != nil {
|
||
logger.Printf("[WARN] local-api: GET /storage failed (%v) — channel not verified", err)
|
||
return
|
||
}
|
||
logger.Printf("[INFO] local-api: channel up (agent %s) — guest %d, %d mount(s) visible",
|
||
cfg.LocalAPI.Endpoint, resp.VMID, len(resp.Mounts))
|
||
for _, m := range resp.Mounts {
|
||
logger.Printf("[INFO] local-api: mount %s → %s (storage=%s, class=%s, backup=%v)",
|
||
m.Key, m.MountPoint, m.Storage, m.Class, m.Backup)
|
||
}
|
||
}
|
||
|
||
// runSetupMode starts the setup wizard on dual listeners and blocks until signal.
|
||
func runSetupMode(cfg *config.Config, logger *log.Logger) {
|
||
ips := setup.DetectLocalIPs()
|
||
setup.LogSetupMode(cfg.Customer.Domain, ips, cfg.Web.SetupListen, logger)
|
||
|
||
setupSrv := setup.NewServer(cfg, cfg.Paths.DataDir, logger, Version)
|
||
handler := setupSrv.Handler()
|
||
|
||
// Health endpoint wrapper (returns setup_mode: true)
|
||
healthHandler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||
w.Header().Set("Content-Type", "application/json")
|
||
json.NewEncoder(w).Encode(map[string]interface{}{
|
||
"ok": true, "message": "felhom-controller is healthy",
|
||
"setup_mode": true, "version": Version,
|
||
})
|
||
})
|
||
|
||
// Mux for both listeners
|
||
mux := http.NewServeMux()
|
||
mux.HandleFunc("/api/health", healthHandler)
|
||
mux.Handle("/", handler)
|
||
|
||
// Start main listener (:8080, behind Traefik for domain access)
|
||
mainServer := &http.Server{
|
||
Addr: cfg.Web.Listen,
|
||
Handler: mux,
|
||
ReadTimeout: 30 * time.Second,
|
||
WriteTimeout: 60 * time.Second,
|
||
IdleTimeout: 120 * time.Second,
|
||
}
|
||
go func() {
|
||
logger.Printf("[INFO] Setup wizard (main) listening on %s", cfg.Web.Listen)
|
||
if err := mainServer.ListenAndServe(); err != http.ErrServerClosed {
|
||
logger.Printf("[ERROR] Main HTTP server error: %v", err)
|
||
}
|
||
}()
|
||
|
||
// Start setup-only listener (:8081, direct HTTP for LAN access)
|
||
setupServer := &http.Server{
|
||
Addr: cfg.Web.SetupListen,
|
||
Handler: mux,
|
||
ReadTimeout: 30 * time.Second,
|
||
WriteTimeout: 60 * time.Second,
|
||
IdleTimeout: 120 * time.Second,
|
||
}
|
||
go func() {
|
||
logger.Printf("[INFO] Setup wizard (LAN) listening on %s", cfg.Web.SetupListen)
|
||
if err := setupServer.ListenAndServe(); err != http.ErrServerClosed {
|
||
logger.Printf("[ERROR] Setup HTTP server error: %v", err)
|
||
}
|
||
}()
|
||
|
||
// Wait for signal
|
||
sigCh := make(chan os.Signal, 1)
|
||
signal.Notify(sigCh, syscall.SIGINT, syscall.SIGTERM)
|
||
sig := <-sigCh
|
||
logger.Printf("[INFO] Received signal %v, shutting down setup wizard...", sig)
|
||
|
||
shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||
defer shutdownCancel()
|
||
mainServer.Shutdown(shutdownCtx)
|
||
setupServer.Shutdown(shutdownCtx)
|
||
logger.Println("[INFO] Setup wizard stopped")
|
||
}
|
||
|
||
// discoverHDDPaths scans deployed apps' app.yaml for HDD_PATH env values.
|
||
func discoverHDDPaths(stacksDir string, logger *log.Logger) []string {
|
||
entries, err := os.ReadDir(stacksDir)
|
||
if err != nil {
|
||
logger.Printf("[WARN] Cannot read stacks dir for HDD path discovery: %v", err)
|
||
return nil
|
||
}
|
||
seen := make(map[string]bool)
|
||
var paths []string
|
||
for _, e := range entries {
|
||
if !e.IsDir() {
|
||
continue
|
||
}
|
||
appCfg := stacks.LoadAppConfig(filepath.Join(stacksDir, e.Name()))
|
||
if appCfg == nil || !appCfg.Deployed {
|
||
continue
|
||
}
|
||
if hddPath, ok := appCfg.Env["HDD_PATH"]; ok && hddPath != "" {
|
||
cleaned := filepath.Clean(hddPath)
|
||
if !seen[cleaned] {
|
||
seen[cleaned] = true
|
||
paths = append(paths, cleaned)
|
||
}
|
||
}
|
||
}
|
||
return paths
|
||
}
|
||
|
||
// ── R-359 / R-397 — the integrity check's one caller, shared by the scheduler and the debug button ──
|
||
//
|
||
// ONE function for both so the hand-run cannot drift from the scheduled run. The only difference
|
||
// between them is `force`, which skips the due-ness question; every other guard — the single-writer
|
||
// flag above all — applies identically. A hand-run that bypassed the flag "because the operator asked
|
||
// for it" is precisely the hazard this whole feature is shaped around.
|
||
//
|
||
// R-397: `NotifyIntegrityOK` and `NotifyIntegrityFailed` have existed in internal/notify since the
|
||
// notifier did, the hub allowlists both event types, and the hub carries the Hungarian customer text
|
||
// for both. Everything was built except the caller. That is the SIXTH time this project has found that
|
||
// shape, and it is worth naming as a pattern rather than treating as a novelty each time.
|
||
func runOffsiteIntegrityCheck(ctx context.Context, mgr *backup.Manager, n *notify.Notifier, logger *log.Logger, force bool) backup.IntegrityResult {
|
||
if mgr == nil {
|
||
return backup.IntegrityResult{Skipped: true, SkipReason: "backup manager not configured"}
|
||
}
|
||
if !force {
|
||
due, last := mgr.IntegrityDue(time.Now())
|
||
if !due {
|
||
logger.Printf("[INFO] [offbox] integrity: not due (last successful check %s ago, at %s) — nothing was run",
|
||
time.Since(last).Round(time.Minute), last.Format("2006-01-02 15:04"))
|
||
return backup.IntegrityResult{Skipped: true, SkipReason: "not due"}
|
||
}
|
||
}
|
||
|
||
res := mgr.CheckOffboxIntegrity(ctx)
|
||
|
||
switch {
|
||
case res.Skipped:
|
||
// Not a failure and not an alarm. The log already said why, inside the check.
|
||
return res
|
||
case res.Unreachable:
|
||
// "I could not look" is not "I looked and it is broken". Deliberately NO notifier: R-339 owns
|
||
// reachability, and a second alarm for the same fact would train the operator to discount the
|
||
// one alarm that means the customer's backups are damaged.
|
||
return res
|
||
case res.OK:
|
||
mgr.RecordIntegrityVerdict(res)
|
||
// severity `info`, which severityNotifies DROPS before either leg — so this mails NOBODY, by
|
||
// design (08 §6.1). A weekly success e-mail is how people stop reading their alerts. It is
|
||
// still pushed, because the hub stores it and the event stream is where "was it checked?" is
|
||
// answered.
|
||
n.NotifyIntegrityOK(integrityOKMsg(res))
|
||
return res
|
||
default:
|
||
mgr.RecordIntegrityVerdict(res)
|
||
// A FAILING store advances due-ness too: re-checking a broken repository every night is load
|
||
// with no new information, and the hourly operator cooldown already governs the mail.
|
||
//
|
||
// The customer gets a SENTENCE. restic's own words go to the log, truncated, where the operator
|
||
// can diagnose without a rebuild. R-379 is the reason that split exists: 615 bytes of raw
|
||
// database text reached a customer once.
|
||
n.NotifyIntegrityFailed(integrityFailedMsg, "restic check reported repository errors")
|
||
return res
|
||
}
|
||
}
|
||
|
||
// R-359 customer-facing strings. Named constants because tests assert them verbatim and because a
|
||
// silent edit is how an honest message drifts into a comforting one.
|
||
const (
|
||
// integrityFailedMsg names what to do and what NOT to do. "Ne törölj semmit" is load-bearing: a
|
||
// customer who believes their backups are broken may try to "start fresh", which destroys the one
|
||
// copy that might still be partly recoverable.
|
||
integrityFailedMsg = "A távoli mentés ellenőrzése hibát talált a tárolóban. A mentések egy része sérült lehet. Ne törölj semmit, és vedd fel velünk a kapcsolatot."
|
||
|
||
integrityOKBase = "A távoli mentés ellenőrzése rendben lezajlott."
|
||
)
|
||
|
||
// ── R-87 — the nightly off-site PROOF's one caller ─────────────────────────────────────────────
|
||
//
|
||
// ONE function, like runOffsiteIntegrityCheck above, so a future debug button cannot drift from the
|
||
// scheduled run. Every guard lives inside `ProveOffboxUnit`; this owns only what happens to the
|
||
// verdict afterwards, because only the caller knows whether a verdict counts.
|
||
//
|
||
// THE FOUR NON-ALARMING OUTCOMES ARE NOT FAILURES and none of them notifies:
|
||
// - Skipped — it yielded to a running backup. Correct behaviour; alarming would punish it.
|
||
// - NoSnapshot — nothing is due. Not a fact about any backup.
|
||
// - Err — the restore did not finish, so it SAW NOTHING and may not claim anything about
|
||
// the backup. This is the same separation `IntegrityResult` draws between a failed
|
||
// check and an unreachable store, and it is why a download error can never become
|
||
// the "intact but empty" alarm.
|
||
// - CannotJudge — recorded, never alarmed. Alarming on our own blind spot trains the operator to
|
||
// discount the one alarm that means the customer's backup holds nothing.
|
||
func runOffsiteProof(ctx context.Context, mgr *backup.Manager, n *notify.Notifier, logger *log.Logger) backup.ProofResult {
|
||
if mgr == nil {
|
||
return backup.ProofResult{Skipped: true, SkipReason: "backup manager not configured"}
|
||
}
|
||
res := mgr.ProveOffboxUnit(ctx)
|
||
if res.Verdict() == "" {
|
||
return res // skipped / nothing due / restore error — no verdict was reached
|
||
}
|
||
mgr.RecordProofVerdict(res)
|
||
if res.Judgement.Verdict == backup.UnitProofFail {
|
||
// ONE event. The customer-grade sentence goes in the message; the machine detail goes in the
|
||
// detail field, where the operator can diagnose without a rebuild — the R-379 split.
|
||
n.NotifyOffsiteProofEmpty(proofEmptyMsg(res.Stack), proofEmptyDetail(res))
|
||
}
|
||
if logger != nil && res.Judgement.Verdict == backup.UnitProofCannotJudge {
|
||
logger.Printf("[WARN] [offbox] proof: %s could not be judged (%s) — recorded, deliberately NOT alarmed",
|
||
res.Stack, res.Judgement.Reason)
|
||
}
|
||
return res
|
||
}
|
||
|
||
// proofEmptyMsg is the R-87 customer-grade sentence. A named function because tests assert it
|
||
// verbatim and because a silent edit is how an honest message drifts into a comforting one.
|
||
//
|
||
// IT MUST SAY THE TWO THINGS THAT MATTER AND NOT A THIRD. The backup is READABLE — the store is not
|
||
// damaged, and saying so is load-bearing, because "sérült" (damaged) sends the operator and the
|
||
// customer down the R-359 path, which is a different fault with a different remedy and a standing
|
||
// instruction not to delete anything. And it names the ONE action that helps: the backup has to be
|
||
// made again, which is a capture-side fix.
|
||
func proofEmptyMsg(stack string) string {
|
||
return "A(z) " + stack + " legutóbbi távoli mentése olvasható, de nem tartalmazza az alkalmazás adatait. " +
|
||
"A tároló nem sérült — a mentés készült el üresen. A mentést újra el kell készíteni; addig ebből a mentésből nem lehet visszaállítani."
|
||
}
|
||
|
||
// proofEmptyDetail is the OPERATOR half: the machine reason code, the snapshot it was proved on, and
|
||
// what was expected. Never a file's content and never a path outside the unit — units carry portable
|
||
// secrets.
|
||
func proofEmptyDetail(res backup.ProofResult) string {
|
||
d := "snapshot " + res.SnapshotID + ", reason " + string(res.Judgement.Reason)
|
||
if len(res.Judgement.Missing) > 0 {
|
||
d += ", expected: " + strings.Join(res.Judgement.Missing, " ")
|
||
}
|
||
return d
|
||
}
|
||
|
||
// integrityOKMsg states what was actually checked, so a structure-only pass is never read as a
|
||
// full data verification. The depth is a fact the customer's sentence has to carry: "checked" means
|
||
// two different things depending on it.
|
||
func integrityOKMsg(res backup.IntegrityResult) string {
|
||
msg := integrityOKBase + " (" + res.Duration.Round(time.Second).String()
|
||
if res.ReadDataSubset != "" {
|
||
msg += ", a mentett adatok " + res.ReadDataSubset + "-át újraolvasva"
|
||
}
|
||
return msg + ")"
|
||
}
|
||
|
||
// updateGuardsAdapter implements stacks.UpdateGuards over the backup manager (slice 4). The stacks
|
||
// package cannot import backup, so this is the one place the two meet. Nil-safe on b: a box with
|
||
// backup disabled has no restore point, and the update is refused for that true reason.
|
||
type updateGuardsAdapter struct {
|
||
b *backup.Manager
|
||
q *quiesce.Loop
|
||
}
|
||
|
||
// DumpStamps (v0.275.0, A4) — the app's own unit's database dumps with the versions that wrote them;
|
||
// the conversion-copy release's second condition (stacks.DumpStampSource).
|
||
// The release reads the stamps only through this optional interface — a method nobody satisfies would
|
||
// silently keep every conversion copy forever; this line fails the build instead (v0.275.0).
|
||
var _ stacks.DumpStampSource = (*updateGuardsAdapter)(nil)
|
||
|
||
func (a *updateGuardsAdapter) DumpStamps(name string) []stacks.DataDumpStamp {
|
||
if a.b == nil {
|
||
return nil
|
||
}
|
||
var out []stacks.DataDumpStamp
|
||
for _, d := range a.b.UnitDumpStamps(name) {
|
||
out = append(out, stacks.DataDumpStamp{File: d.File, At: d.At, Images: d.Images})
|
||
}
|
||
return out
|
||
}
|
||
|
||
func (a *updateGuardsAdapter) HoldFor(name string) (bool, string) {
|
||
if a.b == nil {
|
||
return false, ""
|
||
}
|
||
return a.b.RestoreHoldFor(name)
|
||
}
|
||
|
||
// HoldForLang renders the hold sentence for one reader (R-647, v0.265.0).
|
||
func (a *updateGuardsAdapter) HoldForLang(name, lang string) (bool, string) {
|
||
if a.b == nil {
|
||
return false, ""
|
||
}
|
||
return a.b.RestoreHoldForLang(name, lang)
|
||
}
|
||
|
||
func (a *updateGuardsAdapter) Busy(name string) (bool, string) {
|
||
if a.q.SuppressedStacks()[name] {
|
||
return true, "a whole-guest backup (quiesce) is holding it"
|
||
}
|
||
if a.b == nil {
|
||
return false, ""
|
||
}
|
||
return a.b.UpdateBusy(name)
|
||
}
|
||
|
||
// RestorePoints (R-475) reads EVERY tier through backup.UpdateRestorePoints. It must never go back to
|
||
// Tier2UnitRestorePoint alone — pinned by TestR475_AdapterReadsEveryTier.
|
||
func (a *updateGuardsAdapter) RestorePoints(ctx context.Context, name string, accept func(stacks.UpdateRestorePoint) bool) (stacks.UpdateRestorePoint, bool, []stacks.UpdateRestorePoint) {
|
||
if a.b == nil {
|
||
return stacks.UpdateRestorePoint{}, false, nil
|
||
}
|
||
conv := func(p backup.UpdateTierPoint) stacks.UpdateRestorePoint {
|
||
return stacks.UpdateRestorePoint{Tier: p.Tier, ProvenAt: p.At}
|
||
}
|
||
var acc func(backup.UpdateTierPoint) bool
|
||
if accept != nil {
|
||
acc = func(p backup.UpdateTierPoint) bool { return accept(conv(p)) }
|
||
}
|
||
chosen, ok, seen := a.b.UpdateRestorePoints(ctx, name, acc)
|
||
out := make([]stacks.UpdateRestorePoint, 0, len(seen))
|
||
for _, p := range seen {
|
||
out = append(out, conv(p))
|
||
}
|
||
return conv(chosen), ok, out
|
||
}
|
||
|
||
func (a *updateGuardsAdapter) CanBackUp(name string) (bool, string) {
|
||
if a.b == nil {
|
||
return false, "backup is not enabled on this box"
|
||
}
|
||
return a.b.CanBackUpApp(name)
|
||
}
|
||
|
||
func (a *updateGuardsAdapter) BackupNow(ctx context.Context, name string) error {
|
||
if a.b == nil {
|
||
return fmt.Errorf("backup is not enabled on this box")
|
||
}
|
||
return a.b.RunAppBackupNow(ctx, name)
|
||
}
|
||
|
||
func (a *updateGuardsAdapter) SafetyDump(ctx context.Context, name string) ([]string, error) {
|
||
if a.b == nil {
|
||
return nil, fmt.Errorf("backup is not enabled on this box")
|
||
}
|
||
return a.b.WriteUpdateSafetyDump(ctx, name)
|
||
}
|
||
|
||
func (a *updateGuardsAdapter) HoldAfterFailedUpdate(name string, at time.Time, rp stacks.UpdateRestorePoint, undoState string) error {
|
||
if a.b == nil {
|
||
return fmt.Errorf("backup is not enabled on this box — the hold cannot be recorded")
|
||
}
|
||
// R-659 (v0.268.0): the hold names the newest copy that brings the app back WHOLE — not the
|
||
// precondition copy `rp`, which may be one the restore refuses — and none when there is none.
|
||
// R-479 stands inside it: the sentence still says what that copy holds.
|
||
_, err := a.b.HoldAfterFailedUpdateWhole(context.Background(), name, at, undoState)
|
||
return err
|
||
}
|
||
|
||
// HoldKind names the hold in force (v0.269.0) — the page offers Start for an unhealthy stop.
|
||
func (a *updateGuardsAdapter) HoldKind(name string) string {
|
||
if a.b == nil {
|
||
return ""
|
||
}
|
||
return a.b.HoldKind(name)
|
||
}
|
||
|
||
// HoldNoWholeCopy is the page's half of R-659: a hold naming no copy gets no restore button.
|
||
func (a *updateGuardsAdapter) HoldNoWholeCopy(name string) bool {
|
||
if a.b == nil {
|
||
return false
|
||
}
|
||
return a.b.HoldNoWholeCopy(name)
|
||
}
|
||
|
||
// ── Decision 28 (v0.269.0): the box stops an app in a crash loop or an out-of-memory storm ──────────
|
||
|
||
// unhealthyDeps are the seams of stopUnhealthyApps — every one wired in main to the real component.
|
||
type unhealthyDeps struct {
|
||
verdicts func(now time.Time, ooms []stacks.OOMContainer) []stacks.UnhealthyVerdict
|
||
suppressed func() map[string]bool
|
||
busy func(name string) (bool, string)
|
||
holdKind func(name string) string
|
||
stop func(name string) error
|
||
hold func(name, kind string, at time.Time) (int, error)
|
||
notify func(v stacks.UnhealthyVerdict, trip int)
|
||
}
|
||
|
||
// stopUnhealthyApps acts on the detector's verdicts. NEVER during a deploy or an update (the detector
|
||
// skips those), a backup's or a restore's own stop, an app-data operation or a whole-guest quiesce (the
|
||
// suppression sets, `08` §5), or on an app already held. Order: stop, then the hold (so no start path
|
||
// revives it), then the events. Returns the apps stopped.
|
||
func stopUnhealthyApps(logger *log.Logger, d unhealthyDeps, ooms []stacks.OOMContainer, now time.Time) []string {
|
||
var stopped []string
|
||
vs := d.verdicts(now, ooms)
|
||
if len(vs) == 0 {
|
||
return nil
|
||
}
|
||
sup := d.suppressed()
|
||
for _, v := range vs {
|
||
switch {
|
||
case sup[v.Stack]:
|
||
logger.Printf("[INFO] [deadapp] %s crossed the %s threshold (%d in %s) while a backup/quiesce/update holds it — NOT stopped", v.Stack, v.Kind, v.Count, v.Window)
|
||
continue
|
||
case d.holdKind(v.Stack) != "":
|
||
continue
|
||
}
|
||
if busy, why := d.busy(v.Stack); busy {
|
||
logger.Printf("[INFO] [deadapp] %s crossed the %s threshold but %s — NOT stopped", v.Stack, v.Kind, why)
|
||
continue
|
||
}
|
||
logger.Printf("[WARN] [deadapp] %s: %s — %d in %s; STOPPING it (decision 28)", v.Stack, v.Kind, v.Count, v.Window)
|
||
if err := d.stop(v.Stack); err != nil {
|
||
logger.Printf("[ERROR] [deadapp] stopping %s failed: %v — not held, not announced", v.Stack, err)
|
||
continue
|
||
}
|
||
trip, err := d.hold(v.Stack, v.Kind, now)
|
||
if err != nil {
|
||
logger.Printf("[ERROR] [deadapp] %s: %v", v.Stack, err)
|
||
}
|
||
d.notify(v, trip)
|
||
stopped = append(stopped, v.Stack)
|
||
}
|
||
return stopped
|
||
}
|
||
|
||
// chainUpdateLeg runs the off-site half, then the update leg — ALWAYS, whatever the off-site half did:
|
||
// returned nil (ran, or had nothing to do), returned an error, or panicked. The off-site half's error is
|
||
// returned (the scheduler logs it); a panic becomes an error. `09` §6.4.2 point 2; pinned by
|
||
// TestChainUpdateLeg_EveryPath.
|
||
func chainUpdateLeg(ctx context.Context, offsite func(context.Context) error, leg func(context.Context)) (err error) {
|
||
defer func() {
|
||
if r := recover(); r != nil {
|
||
err = fmt.Errorf("the off-site leg panicked: %v", r)
|
||
}
|
||
leg(ctx)
|
||
}()
|
||
return offsite(ctx)
|
||
}
|
||
|
||
var (
|
||
budapestLocOnce sync.Once
|
||
budapestLocVal *time.Location
|
||
)
|
||
|
||
// budapestLoc is the wall clock the backup window is read in (the scheduler's and the quiesce gate's).
|
||
func budapestLoc() *time.Location {
|
||
budapestLocOnce.Do(func() {
|
||
loc, err := time.LoadLocation("Europe/Budapest")
|
||
if err != nil {
|
||
loc = time.Local
|
||
}
|
||
budapestLocVal = loc
|
||
})
|
||
return budapestLocVal
|
||
}
|
||
|
||
// keptSpaceSentence is the drive-full warning's kept-data sentence for one drive (Hungarian, like the
|
||
// warning it extends): each kept folder by app name and size. "" when the drive holds none.
|
||
func keptSpaceSentence(sm *stacks.Manager, drive string) string {
|
||
items := sm.ListKept([]string{drive})
|
||
if len(items) == 0 {
|
||
return ""
|
||
}
|
||
parts := make([]string, 0, len(items))
|
||
for _, it := range items {
|
||
parts = append(parts, fmt.Sprintf("%s (%s)", it.DisplayName, appbackup.HumanizeBytes(it.SizeBytes)))
|
||
}
|
||
return util.Text(i18n.Default, "kept.diskwarn", strings.Join(parts, ", "))
|
||
}
|