package web import ( "fmt" "hash/crc32" "log" "sync" "time" "gitea.dooplex.hu/admin/felhom-controller/internal/backup" "gitea.dooplex.hu/admin/felhom-controller/internal/config" "gitea.dooplex.hu/admin/felhom-controller/internal/i18n" "gitea.dooplex.hu/admin/felhom-controller/internal/monitor" "gitea.dooplex.hu/admin/felhom-controller/internal/settings" ) // Alert represents a persistent dashboard alert banner. // // LOCALISATION (v0.252.0, R-557). An alert is BUILT by a background health cycle and READ minutes // later by whichever page the household opens, so its text cannot be rendered where it is made — // that is the same shape as a flash in a redirect URL, one layer in. An alert therefore carries a // bundle KEY plus its parameters, and GetAlerts renders it in the language the reader asked for. // // Two deliberate exceptions, both because the text is NOT ours to translate: // // - the health report's Issues and Warnings, which arrive as finished sentences and go ON THE WIRE // to the hub (report `health.warnings`, pinned by internal/monitor's wire golden). They keep // `Message` and stay Hungarian in every language until slice 3 gives the hub a language; // - the agent-channel and endpoint-drift lines, whose text is composed by the channel-health // checker. Filed as R-573. // // An alert with no MessageKey renders `Message` verbatim, which is what makes both exceptions work // without a second mechanism. type Alert struct { ID string // unique identifier for filtering Level string // "error", "warning", "info" Message string // display text; used verbatim when MessageKey is empty MessageKey string // bundle key for Message (preferred); empty = Message is already text MessageArgs []interface{} // the key's printf parameters, in order Link string // optional link to relevant page LinkText string // link display text; used verbatim when LinkTextKey is empty LinkTextKey string // bundle key for LinkText PageOnly []string // if non-empty, only show on these pages (e.g., ["dashboard", "monitoring"]) Inline bool // if true, rendered by page template inline, not in layout banner } // rendered returns a copy of the alert with its keys resolved in lang. A key the bundle does not // carry falls back to whatever Message/LinkText already held, so an alert can never render blank. func (a Alert) rendered(lang string) Alert { b, err := i18n.Shared() if err != nil { return a } if a.MessageKey != "" { if len(a.MessageArgs) > 0 { a.Message = b.Msgf(lang, a.MessageKey, a.MessageArgs...) } else { a.Message = b.Msg(lang, a.MessageKey) } } if a.LinkTextKey != "" { a.LinkText = b.Msg(lang, a.LinkTextKey) } return a } // AlertManager generates and stores dashboard alerts from health check results. // Alerts are state-based (not event-based) — they reflect current system state // and are regenerated after each health check cycle. type AlertManager struct { mu sync.RWMutex alerts []Alert logger *log.Logger hubPushStatusFn func() HubPushStatusData // agentChannelAlert is set/cleared by the channel-health checker (out-of-band from Refresh, which // is health-report-driven). nil = channel up. It is prepended in GetAlerts so a dead controller→ // agent link (disk/storage UI broken) is always visible regardless of the health-report cycle. agentChannelAlert *Alert // deadAppAlerts (fix-3, CAMPAIGN-3) is set each cycle by the health loop from the deployed-app // running-state view (which is NOT in the health report — it comes from the stack manager). Same // out-of-band, state-based, self-clearing model as agentChannelAlert: passing an empty slice when // every deployed app is running clears the banner with no manual dismissal. deadAppAlerts []Alert // endpointDriftAlert (R-77) is set/cleared at startup by the local_api drift check. Separate from // agentChannelAlert on purpose: drift is usually the CAUSE and "agent unreachable" the SYMPTOM, // and during the 2026-07-25 outage only the symptom was visible. endpointDriftAlert *Alert } // NewAlertManager creates a new AlertManager. func NewAlertManager(logger *log.Logger) *AlertManager { return &AlertManager{ logger: logger, } } // SetHubPushStatus sets the hub push status callback for generating hub alerts. func (am *AlertManager) SetHubPushStatus(fn func() HubPushStatusData) { am.mu.Lock() am.hubPushStatusFn = fn am.mu.Unlock() } // SetAgentChannelAlert sets (down=true) or clears (down=false) the controller→agent channel-down // dashboard banner. Called by the channel-health checker each probe (idempotent). msg is the short // Hungarian display line. func (am *AlertManager) SetAgentChannelAlert(down bool, msg string) { am.mu.Lock() defer am.mu.Unlock() if !down { am.agentChannelAlert = nil return } am.agentChannelAlert = &Alert{ ID: "agent-channel-down", Level: "error", Message: msg, // composed by the channel-health checker -- R-573 Link: "/settings", LinkTextKey: "alert.link.settings", } } // SetEndpointDriftAlert sets (drift=true) or clears the local_api endpoint-drift banner (R-77). // // It is deliberately a SEPARATE alert from SetAgentChannelAlert: during the 2026-07-25 outage the // generic "agent unreachable" banner was the ONLY signal, and it looked like a dead agent. The two // can also be true at once — a drifted endpoint usually CAUSES the channel to be down — so folding // them together would hide the actionable one behind the symptom. func (am *AlertManager) SetEndpointDriftAlert(drift bool, msg string) { am.mu.Lock() defer am.mu.Unlock() if !drift { am.endpointDriftAlert = nil return } am.endpointDriftAlert = &Alert{ ID: "local-api-endpoint-drift", Level: "error", Message: msg, // composed by the endpoint-drift checker -- R-573 Link: "/settings", LinkTextKey: "alert.link.settings", } } // DeadApp is a deployed app the health loop found not-running (fix-3). State is the container-state // string for the display (e.g. "stopped"/"exited"). type DeadApp struct { Name string DisplayName string State string } // deadAppGroupThreshold: above this many dead apps, collapse to ONE grouped alert (a reboot storm // with many NAS apps down should not paper the dashboard with a wall of banners — fix-3). const deadAppGroupThreshold = 3 // buildDeadAppAlerts turns the dead-app list into dashboard alerts (WARN). ≤ threshold → one per app; // more → a single grouped alert. Pure → unit-tested. Empty list → nil (clears the banner). func buildDeadAppAlerts(dead []DeadApp) []Alert { if len(dead) == 0 { return nil } if len(dead) > deadAppGroupThreshold { return []Alert{{ ID: "deadapp-group", Level: "warning", MessageKey: "alert.deadapp.group", MessageArgs: []interface{}{len(dead)}, Link: "/monitoring", LinkTextKey: "alert.link.monitoring", }} } alerts := make([]Alert, 0, len(dead)) for _, d := range dead { name := d.DisplayName if name == "" { name = d.Name } key, args := "alert.deadapp.single", []interface{}{name} if d.State != "" { key, args = "alert.deadapp.single_state", []interface{}{name, d.State} } alerts = append(alerts, Alert{ ID: "deadapp-" + simpleHash(d.Name), Level: "warning", MessageKey: key, MessageArgs: args, Link: "/monitoring", LinkTextKey: "alert.link.monitoring", }) } return alerts } // SetDeadAppAlerts stores the current deployed-not-running banner set (fix-3). State-based: called // each health cycle with the live dead-app list (empty clears it). Included in GetAlerts. func (am *AlertManager) SetDeadAppAlerts(dead []DeadApp) { alerts := buildDeadAppAlerts(dead) am.mu.Lock() am.deadAppAlerts = alerts am.mu.Unlock() } // Refresh regenerates alerts from the latest health check report and config state. // Called after each health check cycle (every 5 minutes) and on storage state changes. func (am *AlertManager) Refresh(report *monitor.HealthReport, cfg *config.Config, backupMgr *backup.Manager, updateAvailable bool, latestVersion string, storagePaths ...[]settings.StoragePath) { var alerts []Alert // Disconnected storage alerts (top-level error banners on all pages) if len(storagePaths) > 0 { for _, sp := range storagePaths[0] { if sp.Disconnected { label := sp.Label if label == "" { label = sp.Path } alerts = append(alerts, Alert{ ID: "storage-disconnected-" + simpleHash(sp.Path), Level: "error", MessageKey: "alert.storage.disconnected", MessageArgs: []interface{}{label, sp.Path}, Link: "/settings", LinkTextKey: "alert.link.settings", }) } } } // From health check issues (critical) for _, issue := range report.Issues { alerts = append(alerts, Alert{ ID: "health-" + simpleHash(issue), Level: "error", Message: issue, // ON THE WIRE (report health.issues) -- not ours to translate; slice 3 Link: "/monitoring", LinkTextKey: "alert.link.monitoring", }) } // From health check warnings for i, w := range report.Warnings { alert := Alert{ ID: "health-" + simpleHash(w), Level: "warning", Message: w, // ON THE WIRE (report health.warnings) -- not ours to translate; slice 3 Link: "/monitoring", LinkTextKey: "alert.link.monitoring", } // R-553 — WHERE this warning is shown is decided by its KIND, not by the Hungarian words it // contains. The old test was `strings.Contains(w, "meghajtón"/"adattároló"/"meghajtó")`, which // matched exactly one of today's warnings (the not-on-a-separate-drive one, the only lower-case // „meghajtón"); the day that sentence is translated the warning would jump from its quiet place // under the storage bars to the red banner on EVERY page. Pinned by // TestR553_DiskWarningPlacementSurvivesWordingChange. if report.WarningKindAt(i) == monitor.WarnKindStorageNotSeparate { alert.ID = "disk-not-separate" alert.PageOnly = []string{"dashboard", "monitoring"} alert.Inline = true } alerts = append(alerts, alert) } // Hub connection status if !cfg.Hub.Enabled || cfg.Hub.URL == "" { alerts = append(alerts, Alert{ ID: "hub-disabled", Level: "warning", MessageKey: "alert.hub.disabled", Link: "/monitoring", LinkTextKey: "alert.link.monitoring", }) } else if am.hubPushStatusFn != nil { ps := am.hubPushStatusFn() if ps.LastError != "" && (ps.LastSuccess.IsZero() || time.Since(ps.LastSuccess) > 30*time.Minute) { alerts = append(alerts, Alert{ ID: "hub-unreachable", Level: "error", MessageKey: "alert.hub.unreachable", MessageArgs: []interface{}{ps.LastError}, Link: "/monitoring", LinkTextKey: "alert.link.monitoring", }) } } // Backup disabled if !cfg.Backup.Enabled { alerts = append(alerts, Alert{ ID: "backup-disabled", Level: "warning", MessageKey: "alert.backup.disabled", Link: "/settings", LinkTextKey: "alert.link.settings", }) } // Update available if updateAvailable && latestVersion != "" { alerts = append(alerts, Alert{ ID: "update-available", Level: "info", MessageKey: "alert.update.available", MessageArgs: []interface{}{latestVersion}, Link: "/settings", LinkTextKey: "alert.link.update", }) } // Sort: errors first, then warnings, then info sortAlerts(alerts) am.mu.Lock() am.alerts = alerts am.mu.Unlock() } // GetAlerts returns a copy of the current alerts, optionally excluding specific IDs. func (am *AlertManager) GetAlerts(lang string, excludeIDs ...string) []Alert { am.mu.RLock() defer am.mu.RUnlock() if len(am.alerts) == 0 && am.agentChannelAlert == nil && am.endpointDriftAlert == nil && len(am.deadAppAlerts) == 0 { return nil } exclude := make(map[string]bool, len(excludeIDs)) for _, id := range excludeIDs { exclude[id] = true } var result []Alert // Endpoint drift first: it is the actionable CAUSE, and the channel-down banner below is usually // just its symptom. Showing the symptom above the cause is what made the 2026-07-25 outage read // as an infrastructure blip for 17.5 h. if am.endpointDriftAlert != nil && !exclude[am.endpointDriftAlert.ID] { result = append(result, *am.endpointDriftAlert) } // Channel-down is prepended (highest priority — the agent link being dead breaks disk/storage UI). if am.agentChannelAlert != nil && !exclude[am.agentChannelAlert.ID] { result = append(result, *am.agentChannelAlert) } // Dead-app banners (fix-3) — prepended after the channel alert: a deployed app being down is a // high-signal state the operator/customer must see immediately. for _, a := range am.deadAppAlerts { if !exclude[a.ID] { result = append(result, a) } } for _, a := range am.alerts { if exclude[a.ID] { continue } result = append(result, a) } // Cap at 5 visible alerts if len(result) > 5 { overflow := len(result) - 5 result = result[:5] result = append(result, Alert{ ID: "overflow", Level: "info", MessageKey: "alert.overflow", MessageArgs: []interface{}{overflow}, Link: "/monitoring", }) } // Rendered LAST, once, on the way out: the alerts are stored as keys and only the reader knows // the language. Every return path goes through here, so an alert cannot escape with a raw key. for i := range result { result[i] = result[i].rendered(lang) } return result } // GetInlineAlerts returns alerts marked as Inline for a specific page. func (am *AlertManager) GetInlineAlerts(page, lang string) []Alert { am.mu.RLock() defer am.mu.RUnlock() var result []Alert for _, a := range am.alerts { if !a.Inline { continue } if len(a.PageOnly) == 0 { result = append(result, a) continue } for _, p := range a.PageOnly { if p == page { result = append(result, a) break } } } for i := range result { result[i] = result[i].rendered(lang) } return result } // simpleHash returns a short deterministic hash for deduplication. func simpleHash(s string) string { return fmt.Sprintf("%08x", crc32.ChecksumIEEE([]byte(s))) } // sortAlerts sorts alerts by severity: error > warning > info. func sortAlerts(alerts []Alert) { levelOrder := map[string]int{"error": 0, "warning": 1, "info": 2} for i := 1; i < len(alerts); i++ { for j := i; j > 0 && levelOrder[alerts[j].Level] < levelOrder[alerts[j-1].Level]; j-- { alerts[j], alerts[j-1] = alerts[j-1], alerts[j] } } } func countLevel(alerts []Alert, level string) int { n := 0 for _, a := range alerts { if a.Level == level { n++ } } return n }