aca3b8680a
- Fix backup toggles not appearing (read each app's own HDD_PATH from app.yaml) - Storage paths registry in settings.json with auto-discovery from deployed apps - Settings page "Adattárolók" section with disk usage, add/remove/default/schedulable - Deploy page path field as dropdown of registered storage paths - Health check storage monitoring (mount point, disk usage alerts) - Mount-point validation utilities (Linux syscall + cross-platform stubs) - Controller docker-compose mount changed to /mnt:/mnt:rw for multi-storage Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
197 lines
6.1 KiB
Go
197 lines
6.1 KiB
Go
package monitor
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"strings"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/system"
|
|
)
|
|
|
|
// HealthReport contains the results of a system health check.
|
|
type HealthReport struct {
|
|
Status string // "ok", "warn", "fail"
|
|
Issues []string // critical problems
|
|
Warnings []string // non-critical warnings
|
|
Info []string // informational items
|
|
Timestamp time.Time
|
|
}
|
|
|
|
// RunHealthCheck runs system checks and returns a diagnostic report.
|
|
func RunHealthCheck(cfg *config.Config, cpuCollector *system.CPUCollector, storagePaths []settings.StoragePath) *HealthReport {
|
|
report := &HealthReport{
|
|
Status: "ok",
|
|
Timestamp: time.Now(),
|
|
}
|
|
|
|
hddPath := cfg.Paths.HDDPath
|
|
if len(storagePaths) > 0 {
|
|
hddPath = storagePaths[0].Path
|
|
}
|
|
sysInfo := system.GetInfo(hddPath, cpuCollector)
|
|
|
|
// 1. Disk usage (SSD)
|
|
if sysInfo.DiskPercent > 0 {
|
|
if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("SSD disk usage critical: %.0f%%", sysInfo.DiskPercent))
|
|
} else if sysInfo.DiskPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
|
|
report.Warnings = append(report.Warnings, fmt.Sprintf("SSD disk usage high: %.0f%%", sysInfo.DiskPercent))
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("SSD: %.0f%% used", sysInfo.DiskPercent))
|
|
}
|
|
}
|
|
|
|
// HDD disk usage
|
|
if sysInfo.HDDConfigured && sysInfo.HDDPercent > 0 {
|
|
if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskCritPercent) {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("HDD disk usage critical: %.0f%%", sysInfo.HDDPercent))
|
|
} else if sysInfo.HDDPercent >= float64(cfg.Monitoring.Thresholds.DiskWarnPercent) {
|
|
report.Warnings = append(report.Warnings, fmt.Sprintf("HDD disk usage high: %.0f%%", sysInfo.HDDPercent))
|
|
}
|
|
}
|
|
|
|
// 2. Memory usage
|
|
if sysInfo.MemPercent > 0 {
|
|
if sysInfo.MemPercent >= float64(cfg.Monitoring.Thresholds.MemoryWarnPercent) {
|
|
report.Warnings = append(report.Warnings, fmt.Sprintf("Memory usage high: %.0f%%", sysInfo.MemPercent))
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("Memory: %.0f%% used", sysInfo.MemPercent))
|
|
}
|
|
}
|
|
|
|
// 3. CPU usage
|
|
if sysInfo.CPUPercent > 0 {
|
|
if sysInfo.CPUPercent >= float64(cfg.Monitoring.Thresholds.CPUWarnPercent) {
|
|
report.Warnings = append(report.Warnings, fmt.Sprintf("CPU usage high: %.0f%%", sysInfo.CPUPercent))
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("CPU: %.0f%%", sysInfo.CPUPercent))
|
|
}
|
|
}
|
|
|
|
// 4. Temperature
|
|
if sysInfo.TemperatureCelsius > 0 {
|
|
if sysInfo.TemperatureCelsius >= float64(cfg.Monitoring.Thresholds.TemperatureWarnCelsius) {
|
|
report.Warnings = append(report.Warnings, fmt.Sprintf("Temperature high: %.0f°C (%s)", sysInfo.TemperatureCelsius, sysInfo.TemperatureSource))
|
|
} else {
|
|
report.Info = append(report.Info, fmt.Sprintf("Temperature: %.0f°C", sysInfo.TemperatureCelsius))
|
|
}
|
|
}
|
|
|
|
// 5. Docker health
|
|
if err := checkDocker(); err != nil {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("Docker: %v", err))
|
|
} else {
|
|
report.Info = append(report.Info, "Docker: reachable")
|
|
}
|
|
|
|
// 6. Protected containers
|
|
missingProtected := checkProtectedContainers(cfg.Stacks.Protected)
|
|
for _, name := range missingProtected {
|
|
report.Issues = append(report.Issues, fmt.Sprintf("Protected container not running: %s", name))
|
|
}
|
|
|
|
// 7. Storage paths
|
|
storageIssues, storageWarnings := checkStoragePaths(storagePaths)
|
|
report.Issues = append(report.Issues, storageIssues...)
|
|
report.Warnings = append(report.Warnings, storageWarnings...)
|
|
|
|
// Determine status
|
|
if len(report.Issues) > 0 {
|
|
report.Status = "fail"
|
|
} else if len(report.Warnings) > 0 {
|
|
report.Status = "warn"
|
|
}
|
|
|
|
return report
|
|
}
|
|
|
|
// FormatMessage returns a human-readable summary for healthcheck ping body.
|
|
func (r *HealthReport) FormatMessage() string {
|
|
var sb strings.Builder
|
|
|
|
sb.WriteString(fmt.Sprintf("Status: %s\n", strings.ToUpper(r.Status)))
|
|
sb.WriteString(fmt.Sprintf("Time: %s\n\n", r.Timestamp.Format("2006-01-02 15:04:05")))
|
|
|
|
if len(r.Issues) > 0 {
|
|
sb.WriteString("ISSUES:\n")
|
|
for _, issue := range r.Issues {
|
|
sb.WriteString(" - " + issue + "\n")
|
|
}
|
|
sb.WriteString("\n")
|
|
}
|
|
|
|
if len(r.Warnings) > 0 {
|
|
sb.WriteString("WARNINGS:\n")
|
|
for _, w := range r.Warnings {
|
|
sb.WriteString(" - " + w + "\n")
|
|
}
|
|
sb.WriteString("\n")
|
|
}
|
|
|
|
if len(r.Info) > 0 {
|
|
sb.WriteString("INFO:\n")
|
|
for _, info := range r.Info {
|
|
sb.WriteString(" - " + info + "\n")
|
|
}
|
|
}
|
|
|
|
return sb.String()
|
|
}
|
|
|
|
func checkDocker() error {
|
|
cmd := exec.Command("docker", "info", "--format", "{{.ServerVersion}}")
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
return fmt.Errorf("docker not reachable: %v", err)
|
|
}
|
|
if len(strings.TrimSpace(string(out))) == 0 {
|
|
return fmt.Errorf("docker returned empty version")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func checkProtectedContainers(protected []string) []string {
|
|
var missing []string
|
|
for _, name := range protected {
|
|
cmd := exec.Command("docker", "inspect", "--format", "{{.State.Running}}", name)
|
|
out, err := cmd.Output()
|
|
if err != nil {
|
|
missing = append(missing, name)
|
|
continue
|
|
}
|
|
if strings.TrimSpace(string(out)) != "true" {
|
|
missing = append(missing, name)
|
|
}
|
|
}
|
|
return missing
|
|
}
|
|
|
|
func checkStoragePaths(paths []settings.StoragePath) (issues, warnings []string) {
|
|
for _, sp := range paths {
|
|
// Path accessible?
|
|
if _, err := os.Stat(sp.Path); err != nil {
|
|
warnings = append(warnings, fmt.Sprintf("Storage path not accessible: %s", sp.Path))
|
|
continue
|
|
}
|
|
|
|
// Mount point check
|
|
if !system.IsMountPoint(sp.Path) {
|
|
issues = append(issues, fmt.Sprintf("Storage path %s is NOT a mount point — data writes to SSD!", sp.Path))
|
|
}
|
|
|
|
// Disk usage
|
|
if di := system.GetDiskUsage(sp.Path); di != nil {
|
|
if di.UsedPercent >= 95 {
|
|
issues = append(issues, fmt.Sprintf("Storage %s nearly full: %.0f%%", sp.Path, di.UsedPercent))
|
|
} else if di.UsedPercent >= 90 {
|
|
warnings = append(warnings, fmt.Sprintf("Storage %s usage high: %.0f%%", sp.Path, di.UsedPercent))
|
|
}
|
|
}
|
|
}
|
|
return
|
|
}
|