5270bad76e
gates / gates (push) Successful in 23s
Slice 1 translated the dashboard's markup. The sentences the program BUILDS were still Hungarian literals in Go, so an English household clicked an English button and was answered in Hungarian. 226 of them move into the bundle here. A flash was the hard part: it travels inside the redirect URL and is rendered by a DIFFERENT request, so it now carries a bundle key plus its parameters. A link minted by an older controller carries prose and is shown verbatim — never a raw key, never dropped. Also converted: page data and view-model text, the internal/api JSON answers, the alert banners (Alert.MessageKey, rendered on the way out of GetAlerts), 237 country names at display, and the four page titles built around an app name (R-566 closed). Hungarian is byte-identical, and that is measured rather than read: scripts/i18n_go_parity.py freezes every Go literal at the base commit (7 467) and refuses a key whose Hungarian is not that text, byte for byte. Three decoys, each seen to convict. Its own first version filtered the capture through an ASCII-Hungarian word list and missed seven real literals — the R-565 class. The filter is gone. Nothing on the wire moved, and wire goldens now hold it there: the report's health warnings and every notify event message stay Hungarian, because the hub MAILS the controller's sentence when it has no entry of its own. Slice 3 (R-558) owns those. MinAgent: 0.131.0 (unchanged). No hub release needed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
398 lines
16 KiB
Go
398 lines
16 KiB
Go
package monitor
|
|
|
|
import (
|
|
"fmt"
|
|
"log"
|
|
"os"
|
|
"os/exec"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/infra"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
|
)
|
|
|
|
// HealthReport contains the results of a system health check.
|
|
type HealthReport struct {
|
|
Status string // "ok", "warn", "fail"
|
|
Issues []string // critical problems
|
|
Warnings []string // non-critical warnings
|
|
// WarningKinds is parallel to Warnings: one entry per warning, "" when the warning has no kind.
|
|
// R-553 — it exists so the dashboard can decide WHERE a warning is shown without reading the
|
|
// warning's Hungarian words (internal/web/alerts.go used to match „meghajtó"/„adattároló", which
|
|
// localisation slice 2 translates). It is INTERNAL: internal/report/builder.go copies Status,
|
|
// Issues and Warnings only, so the hub report is unchanged — pinned by
|
|
// TestR553_HubReportWarningsAreUnchangedOnTheWire.
|
|
WarningKinds []string
|
|
Info []string // informational items
|
|
Timestamp time.Time
|
|
}
|
|
|
|
// Warning kinds (R-553). A kind names WHAT the warning is about; the text stays the only thing shown.
|
|
const (
|
|
WarnKindStorageNotSeparate = "storage-not-separate" // app data sits on the system drive
|
|
WarnKindStorageDisconnected = "storage-disconnected" // a registered drive is unplugged
|
|
WarnKindStorageUnavailable = "storage-unavailable" // the path cannot be read
|
|
WarnKindStorageUsageHigh = "storage-usage-high" // a data drive is filling up
|
|
)
|
|
|
|
// Storage warning formats. Named (v0.252.0, R-557) so the one property localisation slice 2 must not
|
|
// break is testable: these sentences are ON THE WIRE (report `health.warnings`), so they stay
|
|
// Hungarian in every language and their bytes may not move. The dashboard renders its OWN sentence
|
|
// for the household's language from the kind above plus the parameters below — never from these.
|
|
// Pinned by TestStorageWarningFormatsAreFrozen.
|
|
const (
|
|
warnFmtStorageDisconnected = "Meghajtó leválasztva: %s (%s)"
|
|
warnFmtStorageUnavailable = "Adattároló nem elérhető: %s"
|
|
warnFmtStorageNotSeparate = "Az adattároló (%s) nem külön meghajtón van — az adatok a rendszermeghajtóra íródnak"
|
|
warnFmtStorageUsageHigh = "Adattároló használat magas: %s (%.0f%%)"
|
|
issueFmtStorageAlmostFull = "Adattároló majdnem megtelt: %s (%.0f%%)"
|
|
)
|
|
|
|
// addWarning appends a warning together with its kind, so the two slices cannot drift apart. Every
|
|
// warning goes through here; `WarningKindAt` reads them back.
|
|
func (r *HealthReport) addWarning(text, kind string) {
|
|
r.Warnings = append(r.Warnings, text)
|
|
r.WarningKinds = append(r.WarningKinds, kind)
|
|
}
|
|
|
|
// WarningKindAt returns the kind of Warnings[i], or "" when there is none (older callers, a report
|
|
// built by hand in a test, or a warning that simply has no kind).
|
|
func (r *HealthReport) WarningKindAt(i int) string {
|
|
if r == nil || i < 0 || i >= len(r.WarningKinds) {
|
|
return ""
|
|
}
|
|
return r.WarningKinds[i]
|
|
}
|
|
|
|
// RunHealthCheck runs system checks and returns a diagnostic report.
|
|
func RunHealthCheck(cfg *config.Config, cpuCollector *system.CPUCollector, storagePaths []settings.StoragePath, smb settings.SMBSettings, logger *log.Logger) *HealthReport {
|
|
report := &HealthReport{
|
|
Status: "ok",
|
|
Timestamp: time.Now(),
|
|
}
|
|
|
|
debug := cfg.Logging.Level == "debug" && logger != nil
|
|
|
|
hddPath := cfg.Paths.HDDPath
|
|
if len(storagePaths) > 0 {
|
|
hddPath = storagePaths[0].Path
|
|
}
|
|
sysInfo := system.GetInfo(hddPath, cpuCollector)
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Raw values: disk=%.1f%%, hdd=%.1f%% (configured=%v), mem=%.1f%% (%dMB/%dMB), cpu=%.1f%%, temp=%.1f°C (%s)",
|
|
sysInfo.DiskPercent, sysInfo.HDDPercent, sysInfo.HDDConfigured,
|
|
sysInfo.MemPercent, sysInfo.UsedMemMB, sysInfo.TotalMemMB,
|
|
sysInfo.CPUPercent, sysInfo.TemperatureCelsius, sysInfo.TemperatureSource)
|
|
}
|
|
|
|
// 1. Disk usage (SSD). NOTE (storage-split): sysInfo.DiskPercent statfs's the controller
|
|
// container's "/", whose overlay upperdir lives on the guest's /var/lib/docker volume — so this
|
|
// IS the Docker-data volume guard (post-split it's the dedicated data volume; pre-split it's the
|
|
// rootfs — either way it's wherever Docker's data-root lives). Warn at 80% / crit at 90% used
|
|
// trips ABOVE the prevention layer's 10%-free reserved buffer, so the customer is warned before
|
|
// the deploy gate even engages.
|
|
if sysInfo.DiskPercent > 0 {
|
|
if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("SSD disk usage critical: %.0f%%", sysInfo.DiskPercent))
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] Disk (SSD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
|
|
}
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] SSD disk: CRITICAL (%.0f%% >= %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
|
|
}
|
|
} else if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
|
|
report.addWarning(fmt.Sprintf("SSD disk usage high: %.0f%%", sysInfo.DiskPercent), "")
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] Disk (SSD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
|
|
}
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] SSD disk: WARN (%.0f%% >= %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
|
|
}
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("SSD: %.0f%% used", sysInfo.DiskPercent))
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] SSD disk: OK (%.0f%%)", sysInfo.DiskPercent)
|
|
}
|
|
}
|
|
}
|
|
|
|
// HDD disk usage
|
|
if sysInfo.HDDConfigured && sysInfo.HDDPercent > 0 {
|
|
if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("HDD disk usage critical: %.0f%%", sysInfo.HDDPercent))
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] Disk (HDD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.HDDPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
|
|
}
|
|
} else if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
|
|
report.addWarning(fmt.Sprintf("HDD disk usage high: %.0f%%", sysInfo.HDDPercent), "")
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] Disk (HDD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.HDDPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
|
|
}
|
|
}
|
|
}
|
|
|
|
// 2. Memory usage
|
|
if sysInfo.MemPercent > 0 {
|
|
if sysInfo.MemPercent >= float64(cfg.Monitoring.Thresholds.MemoryWarnPercent) {
|
|
report.addWarning(fmt.Sprintf("Memory usage high: %.0f%%", sysInfo.MemPercent), "")
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] Memory threshold breached: %.0f%% (limit: %d%%)", sysInfo.MemPercent, cfg.Monitoring.Thresholds.MemoryWarnPercent)
|
|
}
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Memory: WARN (%.0f%% >= %d%%)", sysInfo.MemPercent, cfg.Monitoring.Thresholds.MemoryWarnPercent)
|
|
}
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("Memory: %.0f%% used", sysInfo.MemPercent))
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Memory: OK (%.0f%%)", sysInfo.MemPercent)
|
|
}
|
|
}
|
|
}
|
|
|
|
// 3. CPU usage
|
|
if sysInfo.CPUPercent > 0 {
|
|
if sysInfo.CPUPercent >= float64(cfg.Monitoring.Thresholds.CPUWarnPercent) {
|
|
report.addWarning(fmt.Sprintf("CPU usage high: %.0f%%", sysInfo.CPUPercent), "")
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] CPU threshold breached: %.0f%% (limit: %d%%)", sysInfo.CPUPercent, cfg.Monitoring.Thresholds.CPUWarnPercent)
|
|
}
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] CPU: WARN (%.0f%% >= %d%%)", sysInfo.CPUPercent, cfg.Monitoring.Thresholds.CPUWarnPercent)
|
|
}
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("CPU: %.0f%%", sysInfo.CPUPercent))
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] CPU: OK (%.0f%%)", sysInfo.CPUPercent)
|
|
}
|
|
}
|
|
}
|
|
|
|
// 4. Temperature
|
|
if sysInfo.TemperatureCelsius > 0 {
|
|
if sysInfo.TemperatureCelsius >= float64(cfg.Monitoring.Thresholds.TemperatureWarnCelsius) {
|
|
report.addWarning(fmt.Sprintf("Temperature high: %.0f°C (%s)", sysInfo.TemperatureCelsius, sysInfo.TemperatureSource), "")
|
|
if logger != nil {
|
|
logger.Printf("[WARN] [monitor] Temperature threshold breached: %.0f°C (limit: %d°C)", sysInfo.TemperatureCelsius, cfg.Monitoring.Thresholds.TemperatureWarnCelsius)
|
|
}
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Temperature: WARN (%.0f°C >= %d°C)", sysInfo.TemperatureCelsius, cfg.Monitoring.Thresholds.TemperatureWarnCelsius)
|
|
}
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("Temperature: %.0f°C", sysInfo.TemperatureCelsius))
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Temperature: OK (%.0f°C)", sysInfo.TemperatureCelsius)
|
|
}
|
|
}
|
|
}
|
|
|
|
// 5. Docker health
|
|
if err := checkDocker(); err != nil {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("Docker: %v", err))
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Docker daemon: FAIL (%v)", err)
|
|
}
|
|
} else {
|
|
report.Info = append(report.Info, "Docker: reachable")
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Docker daemon: OK")
|
|
}
|
|
}
|
|
|
|
// 6. Protected containers (effective set: cloudflared only counts when a tunnel token is
|
|
// configured, so a LAN-only node doesn't report FAIL forever for a stack it intentionally skips).
|
|
protected := EffectiveProtected(cfg, smb)
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Checking %d protected containers: %v", len(protected), protected)
|
|
}
|
|
missingProtected := checkProtectedContainers(protected)
|
|
for _, name := range missingProtected {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("Protected container not running: %s", name))
|
|
}
|
|
if debug {
|
|
if len(missingProtected) > 0 {
|
|
logger.Printf("[DEBUG] [monitor] Protected containers missing: %v", missingProtected)
|
|
} else {
|
|
logger.Printf("[DEBUG] [monitor] All protected containers running")
|
|
}
|
|
}
|
|
|
|
// 7. Storage paths
|
|
storageIssues, storageWarnings, storageKinds := checkStoragePaths(storagePaths)
|
|
report.Issues = append(report.Issues, storageIssues...)
|
|
for i, w := range storageWarnings {
|
|
report.addWarning(w, storageKinds[i])
|
|
}
|
|
|
|
// Determine status
|
|
if len(report.Issues) > 0 {
|
|
report.Status = "fail"
|
|
} else if len(report.Warnings) > 0 {
|
|
report.Status = "warn"
|
|
}
|
|
|
|
if logger != nil {
|
|
logger.Printf("[INFO] [monitor] Health check: status=%s", report.Status)
|
|
}
|
|
|
|
if debug {
|
|
logger.Printf("[DEBUG] [monitor] Final status: %s (issues=%d, warnings=%d, info=%d)",
|
|
report.Status, len(report.Issues), len(report.Warnings), len(report.Info))
|
|
}
|
|
|
|
return report
|
|
}
|
|
|
|
// FormatMessage returns a human-readable summary for healthcheck ping body.
|
|
func (r *HealthReport) FormatMessage() string {
|
|
var sb strings.Builder
|
|
|
|
sb.WriteString(fmt.Sprintf("Status: %s\n", strings.ToUpper(r.Status)))
|
|
sb.WriteString(fmt.Sprintf("Time: %s\n\n", r.Timestamp.Format("2006-01-02 15:04:05")))
|
|
|
|
if len(r.Issues) > 0 {
|
|
sb.WriteString("ISSUES:\n")
|
|
for _, issue := range r.Issues {
|
|
sb.WriteString(" - " + issue + "\n")
|
|
}
|
|
sb.WriteString("\n")
|
|
}
|
|
|
|
if len(r.Warnings) > 0 {
|
|
sb.WriteString("WARNINGS:\n")
|
|
for _, w := range r.Warnings {
|
|
sb.WriteString(" - " + w + "\n")
|
|
}
|
|
sb.WriteString("\n")
|
|
}
|
|
|
|
if len(r.Info) > 0 {
|
|
sb.WriteString("INFO:\n")
|
|
for _, info := range r.Info {
|
|
sb.WriteString(" - " + info + "\n")
|
|
}
|
|
}
|
|
|
|
return sb.String()
|
|
}
|
|
|
|
func checkDocker() error {
|
|
cmd := exec.Command("docker", "info", "--format", "{{.ServerVersion}}")
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
return fmt.Errorf("docker not reachable: %v", err)
|
|
}
|
|
if len(strings.TrimSpace(string(out))) == 0 {
|
|
return fmt.Errorf("docker returned empty version")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// EffectiveProtected returns the protected-container set that actually applies to this node. It is
|
|
// the configured cfg.Stacks.Protected minus stacks that are intentionally not deployed here, plus
|
|
// the DYNAMIC extras whose deployment depends on customer state rather than config:
|
|
//
|
|
// - cloudflared is dropped when no tunnel token is configured (a LAN-only node legitimately runs
|
|
// without it, so it must not be reported as a missing protected container forever);
|
|
// - the samba container is ADDED only when network sharing is switched on AND the household
|
|
// password has been set (R-7b, tightened by R-77). Sharing is a customer-toggled feature, so it
|
|
// can never appear in the golden controller.yaml — but once it is actually RUNNING, a dead
|
|
// sharing service is exactly as customer-visible as a dead traefik and must raise the same
|
|
// protected-container issue → alert → Hungarian degradation e-mail.
|
|
//
|
|
// THE COUPLING, and why it is spelled out: this set must mirror EVERY early return in
|
|
// stacks.reconcileSambaAt, because that function decides whether the container exists at all. It has
|
|
// TWO:
|
|
//
|
|
// if !smb.Enabled { return } // feature off
|
|
// if !smb.UserSet { return } // on, but no household password yet → deliberately NOT deployed
|
|
//
|
|
// R-77 exists because this comment previously claimed "detection and deployment agree in both
|
|
// directions" while citing only the first. The second was added later and never mirrored here, so
|
|
// enabling sharing without setting a password made the box report health=fail forever for a state
|
|
// the controller had deliberately chosen (observed live on demo-hp, 2026-07-26). A THIRD early
|
|
// return in reconcileSambaAt would need the same mirror — and this comment must be updated with it,
|
|
// because a comment asserting a guarantee the code no longer provides is how the bug came back.
|
|
//
|
|
// Deliberately NOT over-suppressed: sharing on WITH a password and a dead container still raises the
|
|
// issue. That is the case the protected set exists for.
|
|
//
|
|
// NOTE: the entries are CONTAINER names (checkProtectedContainers docker-inspects them). For the
|
|
// base stacks the container name happens to equal the stack name; for samba it does NOT — the stack
|
|
// is „samba" but the container is infra.SambaContainerName — which is why the constant is read here
|
|
// rather than the stack name assumed.
|
|
func EffectiveProtected(cfg *config.Config, smb settings.SMBSettings) []string {
|
|
out := make([]string, 0, len(cfg.Stacks.Protected)+1)
|
|
for _, name := range cfg.Stacks.Protected {
|
|
if name == "cloudflared" && cfg.Infrastructure.CFTunnelToken == "" {
|
|
continue
|
|
}
|
|
out = append(out, name)
|
|
}
|
|
// Mirrors reconcileSambaAt's two early returns — see the coupling note above.
|
|
if smb.Enabled && smb.UserSet {
|
|
out = append(out, infra.SambaContainerName)
|
|
}
|
|
return out
|
|
}
|
|
|
|
func checkProtectedContainers(protected []string) []string {
|
|
var missing []string
|
|
for _, name := range protected {
|
|
cmd := exec.Command("docker", "inspect", "--format", "{{.State.Running}}", name)
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
missing = append(missing, name)
|
|
continue
|
|
}
|
|
if strings.TrimSpace(string(out)) != "true" {
|
|
missing = append(missing, name)
|
|
}
|
|
}
|
|
return missing
|
|
}
|
|
|
|
// checkStoragePaths returns the storage issues and, beside each warning, its KIND (R-553) — the
|
|
// dashboard places the "not on a separate drive" warning inline under the storage bars, and it must
|
|
// find it by kind rather than by the words the sentence happens to contain today.
|
|
func checkStoragePaths(paths []settings.StoragePath) (issues, warnings, kinds []string) {
|
|
for _, sp := range paths {
|
|
// Skip decommissioned paths — no longer in active use
|
|
if sp.Decommissioned {
|
|
continue
|
|
}
|
|
|
|
// Skip disconnected paths — handled by the storage watchdog
|
|
if sp.Disconnected {
|
|
warnings, kinds = append(warnings, fmt.Sprintf(warnFmtStorageDisconnected, sp.Label, sp.Path)), append(kinds, WarnKindStorageDisconnected)
|
|
continue
|
|
}
|
|
|
|
// Path accessible?
|
|
if _, err := os.Stat(sp.Path); err != nil {
|
|
warnings, kinds = append(warnings, fmt.Sprintf(warnFmtStorageUnavailable, sp.Path)), append(kinds, WarnKindStorageUnavailable)
|
|
continue
|
|
}
|
|
|
|
// Mount point check — warning, not issue (avoids false FAIL on demo/test environments)
|
|
if !system.IsMountPoint(sp.Path) {
|
|
warnings = append(warnings, fmt.Sprintf(
|
|
warnFmtStorageNotSeparate, sp.Path))
|
|
kinds = append(kinds, WarnKindStorageNotSeparate)
|
|
}
|
|
|
|
// Disk usage
|
|
if di := system.GetDiskUsage(sp.Path); di != nil {
|
|
if di.UsedPercent >= 95 {
|
|
issues = append(issues, fmt.Sprintf(issueFmtStorageAlmostFull, sp.Path, di.UsedPercent))
|
|
} else if di.UsedPercent >= 90 {
|
|
warnings, kinds = append(warnings, fmt.Sprintf(warnFmtStorageUsageHigh, sp.Path, di.UsedPercent)), append(kinds, WarnKindStorageUsageHigh)
|
|
}
|
|
}
|
|
}
|
|
return
|
|
}
|