R-516 items 7-10: one banner per drive, local times, te-form, health banners in the household's language

- item 7: the storage page shows the disconnect time in local time (fmtTimeStr), not the raw RFC3339 UTC.
- item 8: a disconnected drive no longer has two banners - the health check's warning for it is dropped
  when the dedicated alert.storage.disconnected banner was built (it stays on the wire).
- item 9: alert.deadapp.group says "nezd meg" (te-form); formal ceiling 14 -> 13.
- item 10: the disk/memory/CPU/temperature health banners show the dashboard's own sentence
  (health.* keys, hu + en) via HealthReport.WarningMsgs/IssueMsgs; the wire text is unchanged.
  checkResources split out of RunHealthCheck as the test seam.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-06 01:42:16 +02:00
parent 0f2eab7a3e
commit e4774e6a06
10 changed files with 388 additions and 115 deletions
+162 -104
View File
@@ -26,8 +26,15 @@ type HealthReport struct {
// Issues and Warnings only, so the hub report is unchanged — pinned by
// TestR553_HubReportWarningsAreUnchangedOnTheWire.
WarningKinds []string
Info []string // informational items
Timestamp time.Time
// WarningMsgs / IssueMsgs are parallel to Warnings / Issues (R-516 item 10): the bundle key and
// arguments of the dashboard's OWN sentence for that entry, in the household's language. The
// Warnings/Issues text stays exactly what it was — it is ON THE WIRE (report health.*), and
// internal/report/builder.go copies Status, Issues and Warnings only (pinned by
// TestR553_HubReportWarningsAreUnchangedOnTheWire). An entry with no key shows its text verbatim.
WarningMsgs []MsgRef
IssueMsgs []MsgRef
Info []string // informational items
Timestamp time.Time
}
// Warning kinds (R-553). A kind names WHAT the warning is about; the text stays the only thing shown.
@@ -54,8 +61,43 @@ const (
// addWarning appends a warning together with its kind, so the two slices cannot drift apart. Every
// warning goes through here; `WarningKindAt` reads them back.
func (r *HealthReport) addWarning(text, kind string) {
r.addWarningMsg(text, kind, MsgRef{})
}
// MsgRef names a dashboard sentence by bundle key; the zero value means "show the text verbatim".
type MsgRef struct {
Key string
Args []interface{}
}
// addWarningMsg is addWarning with the dashboard's own sentence beside the wire text (R-516 item 10).
func (r *HealthReport) addWarningMsg(text, kind string, msg MsgRef) {
r.Warnings = append(r.Warnings, text)
r.WarningKinds = append(r.WarningKinds, kind)
r.WarningMsgs = append(r.WarningMsgs, msg)
}
// addIssue appends a critical issue with its dashboard sentence (zero MsgRef = verbatim), so the two
// slices cannot drift apart. Every issue goes through here; IssueMsgAt reads them back.
func (r *HealthReport) addIssue(text string, msg MsgRef) {
r.Issues = append(r.Issues, text)
r.IssueMsgs = append(r.IssueMsgs, msg)
}
// WarningMsgAt / IssueMsgAt return the dashboard sentence of entry i, or the zero MsgRef when there is
// none (a report built by hand in a test, or an entry with no key).
func (r *HealthReport) WarningMsgAt(i int) MsgRef {
if r == nil || i < 0 || i >= len(r.WarningMsgs) {
return MsgRef{}
}
return r.WarningMsgs[i]
}
func (r *HealthReport) IssueMsgAt(i int) MsgRef {
if r == nil || i < 0 || i >= len(r.IssueMsgs) {
return MsgRef{}
}
return r.IssueMsgs[i]
}
// WarningKindAt returns the kind of Warnings[i], or "" when there is none (older callers, a report
@@ -89,109 +131,11 @@ func RunHealthCheck(cfg *config.Config, cpuCollector *system.CPUCollector, stora
sysInfo.CPUPercent, sysInfo.TemperatureCelsius, sysInfo.TemperatureSource)
}
// 1. Disk usage (SSD). NOTE (storage-split): sysInfo.DiskPercent statfs's the controller
// container's "/", whose overlay upperdir lives on the guest's /var/lib/docker volume — so this
// IS the Docker-data volume guard (post-split it's the dedicated data volume; pre-split it's the
// rootfs — either way it's wherever Docker's data-root lives). Warn at 80% / crit at 90% used
// trips ABOVE the prevention layer's 10%-free reserved buffer, so the customer is warned before
// the deploy gate even engages.
if sysInfo.DiskPercent > 0 {
if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
report.Issues = append(report.Issues, fmt.Sprintf("SSD disk usage critical: %.0f%%", sysInfo.DiskPercent))
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (SSD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] SSD disk: CRITICAL (%.0f%% >= %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
}
} else if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
report.addWarning(fmt.Sprintf("SSD disk usage high: %.0f%%", sysInfo.DiskPercent), "")
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (SSD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] SSD disk: WARN (%.0f%% >= %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("SSD: %.0f%% used", sysInfo.DiskPercent))
if debug {
logger.Printf("[DEBUG] [monitor] SSD disk: OK (%.0f%%)", sysInfo.DiskPercent)
}
}
}
// HDD disk usage
if sysInfo.HDDConfigured && sysInfo.HDDPercent > 0 {
if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
report.Issues = append(report.Issues, fmt.Sprintf("HDD disk usage critical: %.0f%%", sysInfo.HDDPercent))
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (HDD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.HDDPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
}
} else if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
report.addWarning(fmt.Sprintf("HDD disk usage high: %.0f%%", sysInfo.HDDPercent), "")
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (HDD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.HDDPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
}
}
}
// 2. Memory usage
if sysInfo.MemPercent > 0 {
if sysInfo.MemPercent >= float64(cfg.Monitoring.Thresholds.MemoryWarnPercent) {
report.addWarning(fmt.Sprintf("Memory usage high: %.0f%%", sysInfo.MemPercent), "")
if logger != nil {
logger.Printf("[WARN] [monitor] Memory threshold breached: %.0f%% (limit: %d%%)", sysInfo.MemPercent, cfg.Monitoring.Thresholds.MemoryWarnPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] Memory: WARN (%.0f%% >= %d%%)", sysInfo.MemPercent, cfg.Monitoring.Thresholds.MemoryWarnPercent)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("Memory: %.0f%% used", sysInfo.MemPercent))
if debug {
logger.Printf("[DEBUG] [monitor] Memory: OK (%.0f%%)", sysInfo.MemPercent)
}
}
}
// 3. CPU usage
if sysInfo.CPUPercent > 0 {
if sysInfo.CPUPercent >= float64(cfg.Monitoring.Thresholds.CPUWarnPercent) {
report.addWarning(fmt.Sprintf("CPU usage high: %.0f%%", sysInfo.CPUPercent), "")
if logger != nil {
logger.Printf("[WARN] [monitor] CPU threshold breached: %.0f%% (limit: %d%%)", sysInfo.CPUPercent, cfg.Monitoring.Thresholds.CPUWarnPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] CPU: WARN (%.0f%% >= %d%%)", sysInfo.CPUPercent, cfg.Monitoring.Thresholds.CPUWarnPercent)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("CPU: %.0f%%", sysInfo.CPUPercent))
if debug {
logger.Printf("[DEBUG] [monitor] CPU: OK (%.0f%%)", sysInfo.CPUPercent)
}
}
}
// 4. Temperature
if sysInfo.TemperatureCelsius > 0 {
if sysInfo.TemperatureCelsius >= float64(cfg.Monitoring.Thresholds.TemperatureWarnCelsius) {
report.addWarning(fmt.Sprintf("Temperature high: %.0f°C (%s)", sysInfo.TemperatureCelsius, sysInfo.TemperatureSource), "")
if logger != nil {
logger.Printf("[WARN] [monitor] Temperature threshold breached: %.0f°C (limit: %d°C)", sysInfo.TemperatureCelsius, cfg.Monitoring.Thresholds.TemperatureWarnCelsius)
}
if debug {
logger.Printf("[DEBUG] [monitor] Temperature: WARN (%.0f°C >= %d°C)", sysInfo.TemperatureCelsius, cfg.Monitoring.Thresholds.TemperatureWarnCelsius)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("Temperature: %.0f°C", sysInfo.TemperatureCelsius))
if debug {
logger.Printf("[DEBUG] [monitor] Temperature: OK (%.0f°C)", sysInfo.TemperatureCelsius)
}
}
}
checkResources(report, sysInfo, cfg, logger, debug)
// 5. Docker health
if err := checkDocker(); err != nil {
report.Issues = append(report.Issues, fmt.Sprintf("Docker: %v", err))
report.addIssue(fmt.Sprintf("Docker: %v", err), MsgRef{})
if debug {
logger.Printf("[DEBUG] [monitor] Docker daemon: FAIL (%v)", err)
}
@@ -210,7 +154,7 @@ func RunHealthCheck(cfg *config.Config, cpuCollector *system.CPUCollector, stora
}
missingProtected := checkProtectedContainers(protected)
for _, name := range missingProtected {
report.Issues = append(report.Issues, fmt.Sprintf("Protected container not running: %s", name))
report.addIssue(fmt.Sprintf("Protected container not running: %s", name), MsgRef{})
}
if debug {
if len(missingProtected) > 0 {
@@ -222,7 +166,9 @@ func RunHealthCheck(cfg *config.Config, cpuCollector *system.CPUCollector, stora
// 7. Storage paths
storageIssues, storageWarnings, storageKinds := checkStoragePaths(storagePaths)
report.Issues = append(report.Issues, storageIssues...)
for _, is := range storageIssues {
report.addIssue(is, MsgRef{})
}
for i, w := range storageWarnings {
report.addWarning(w, storageKinds[i])
}
@@ -395,3 +341,115 @@ func checkStoragePaths(paths []settings.StoragePath) (issues, warnings, kinds []
}
return
}
// checkResources is the threshold half of RunHealthCheck (disk, memory, CPU, temperature), split out
// so a test can drive it with a made-up SystemInfo instead of the box it runs on (R-516 item 10:
// TestR516_ResourceWarningsCarryTheirDashboardSentence).
func checkResources(report *HealthReport, sysInfo system.SystemInfo, cfg *config.Config, logger *log.Logger, debug bool) {
// 1. Disk usage (SSD). NOTE (storage-split): sysInfo.DiskPercent statfs's the controller
// container's "/", whose overlay upperdir lives on the guest's /var/lib/docker volume — so this
// IS the Docker-data volume guard (post-split it's the dedicated data volume; pre-split it's the
// rootfs — either way it's wherever Docker's data-root lives). Warn at 80% / crit at 90% used
// trips ABOVE the prevention layer's 10%-free reserved buffer, so the customer is warned before
// the deploy gate even engages.
if sysInfo.DiskPercent > 0 {
if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
report.addIssue(fmt.Sprintf("SSD disk usage critical: %.0f%%", sysInfo.DiskPercent),
MsgRef{"health.system_disk_critical", []interface{}{sysInfo.DiskPercent}})
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (SSD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] SSD disk: CRITICAL (%.0f%% >= %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
}
} else if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
report.addWarningMsg(fmt.Sprintf("SSD disk usage high: %.0f%%", sysInfo.DiskPercent), "",
MsgRef{"health.system_disk_high", []interface{}{sysInfo.DiskPercent}})
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (SSD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] SSD disk: WARN (%.0f%% >= %d%%)", sysInfo.DiskPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("SSD: %.0f%% used", sysInfo.DiskPercent))
if debug {
logger.Printf("[DEBUG] [monitor] SSD disk: OK (%.0f%%)", sysInfo.DiskPercent)
}
}
}
// HDD disk usage
if sysInfo.HDDConfigured && sysInfo.HDDPercent > 0 {
if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
report.addIssue(fmt.Sprintf("HDD disk usage critical: %.0f%%", sysInfo.HDDPercent),
MsgRef{"health.data_disk_critical", []interface{}{sysInfo.HDDPercent}})
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (HDD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.HDDPercent, cfg.Monitoring.Thresholds.DiskCritPercent)
}
} else if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
report.addWarningMsg(fmt.Sprintf("HDD disk usage high: %.0f%%", sysInfo.HDDPercent), "",
MsgRef{"health.data_disk_high", []interface{}{sysInfo.HDDPercent}})
if logger != nil {
logger.Printf("[WARN] [monitor] Disk (HDD) threshold breached: %.0f%% (limit: %d%%)", sysInfo.HDDPercent, cfg.Monitoring.Thresholds.DiskWarnPercent)
}
}
}
// 2. Memory usage
if sysInfo.MemPercent > 0 {
if sysInfo.MemPercent >= float64(cfg.Monitoring.Thresholds.MemoryWarnPercent) {
report.addWarningMsg(fmt.Sprintf("Memory usage high: %.0f%%", sysInfo.MemPercent), "",
MsgRef{"health.memory_high", []interface{}{sysInfo.MemPercent}})
if logger != nil {
logger.Printf("[WARN] [monitor] Memory threshold breached: %.0f%% (limit: %d%%)", sysInfo.MemPercent, cfg.Monitoring.Thresholds.MemoryWarnPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] Memory: WARN (%.0f%% >= %d%%)", sysInfo.MemPercent, cfg.Monitoring.Thresholds.MemoryWarnPercent)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("Memory: %.0f%% used", sysInfo.MemPercent))
if debug {
logger.Printf("[DEBUG] [monitor] Memory: OK (%.0f%%)", sysInfo.MemPercent)
}
}
}
// 3. CPU usage
if sysInfo.CPUPercent > 0 {
if sysInfo.CPUPercent >= float64(cfg.Monitoring.Thresholds.CPUWarnPercent) {
report.addWarningMsg(fmt.Sprintf("CPU usage high: %.0f%%", sysInfo.CPUPercent), "",
MsgRef{"health.cpu_high", []interface{}{sysInfo.CPUPercent}})
if logger != nil {
logger.Printf("[WARN] [monitor] CPU threshold breached: %.0f%% (limit: %d%%)", sysInfo.CPUPercent, cfg.Monitoring.Thresholds.CPUWarnPercent)
}
if debug {
logger.Printf("[DEBUG] [monitor] CPU: WARN (%.0f%% >= %d%%)", sysInfo.CPUPercent, cfg.Monitoring.Thresholds.CPUWarnPercent)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("CPU: %.0f%%", sysInfo.CPUPercent))
if debug {
logger.Printf("[DEBUG] [monitor] CPU: OK (%.0f%%)", sysInfo.CPUPercent)
}
}
}
// 4. Temperature
if sysInfo.TemperatureCelsius > 0 {
if sysInfo.TemperatureCelsius >= float64(cfg.Monitoring.Thresholds.TemperatureWarnCelsius) {
report.addWarningMsg(fmt.Sprintf("Temperature high: %.0f°C (%s)", sysInfo.TemperatureCelsius, sysInfo.TemperatureSource), "",
MsgRef{"health.temperature_high", []interface{}{sysInfo.TemperatureCelsius, sysInfo.TemperatureSource}})
if logger != nil {
logger.Printf("[WARN] [monitor] Temperature threshold breached: %.0f°C (limit: %d°C)", sysInfo.TemperatureCelsius, cfg.Monitoring.Thresholds.TemperatureWarnCelsius)
}
if debug {
logger.Printf("[DEBUG] [monitor] Temperature: WARN (%.0f°C >= %d°C)", sysInfo.TemperatureCelsius, cfg.Monitoring.Thresholds.TemperatureWarnCelsius)
}
} else {
report.Info = append(report.Info, fmt.Sprintf("Temperature: %.0f°C", sysInfo.TemperatureCelsius))
if debug {
logger.Printf("[DEBUG] [monitor] Temperature: OK (%.0f°C)", sysInfo.TemperatureCelsius)
}
}
}
}
@@ -0,0 +1,64 @@
package monitor
import (
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
)
// R-516 item 10 — the resource warnings („SSD disk usage high: 90%") reached the household's banner in
// English on every page. checkResources now attaches the dashboard's own sentence (a bundle key) beside
// each one, and the wire text is byte-for-byte what it was.
//
// COMPANION RED-PROOF: give the SSD warning a zero MsgRef (addWarning) — the "warn" case fails here
// and the banner test in internal/web shows the English wire text again.
func TestR516_ResourceWarningsCarryTheirDashboardSentence(t *testing.T) {
cfg := config.Default()
cases := []struct {
name string
info system.SystemInfo
wantIssue, wantKey []string
wantWire []string
}{
{"warn", system.SystemInfo{DiskPercent: 85, HDDConfigured: true, HDDPercent: 85, MemPercent: 90, CPUPercent: 95,
TemperatureCelsius: 80, TemperatureSource: "coretemp"}, nil,
[]string{"health.system_disk_high", "health.data_disk_high", "health.memory_high", "health.cpu_high", "health.temperature_high"},
[]string{"SSD disk usage high: 85%", "HDD disk usage high: 85%", "Memory usage high: 90%", "CPU usage high: 95%", "Temperature high: 80°C (coretemp)"}},
{"critical", system.SystemInfo{DiskPercent: 95, HDDConfigured: true, HDDPercent: 96},
[]string{"health.system_disk_critical", "health.data_disk_critical"}, nil,
[]string{"SSD disk usage critical: 95%", "HDD disk usage critical: 96%"}},
}
for _, c := range cases {
r := &HealthReport{}
checkResources(r, c.info, cfg, nil, false)
texts, keys := r.Warnings, []string{}
for i := range r.Warnings {
keys = append(keys, r.WarningMsgAt(i).Key)
}
if c.wantIssue != nil {
texts, keys = r.Issues, nil
for i := range r.Issues {
keys = append(keys, r.IssueMsgAt(i).Key)
}
}
want := c.wantKey
if c.wantIssue != nil {
want = c.wantIssue
}
if len(texts) != len(c.wantWire) {
t.Fatalf("%s: %d entries %q, want %d — the cases below would be compared against nothing", c.name, len(texts), texts, len(c.wantWire))
}
for i := range texts {
if texts[i] != c.wantWire[i] {
t.Errorf("%s: the wire text moved: %q, want %q", c.name, texts[i], c.wantWire[i])
}
if keys[i] != want[i] {
t.Errorf("%s: %q carries dashboard key %q, want %q", c.name, texts[i], keys[i], want[i])
}
}
if len(r.Warnings) != len(r.WarningKinds) || len(r.Warnings) != len(r.WarningMsgs) || len(r.Issues) != len(r.IssueMsgs) {
t.Errorf("%s: the parallel slices drifted apart", c.name)
}
}
}