7c05b59708
gates / gates (push) Successful in 24s
179 Hungarian sentences were built deep inside a package with fmt.Errorf and printed by whoever caught them: too late to translate where they are shown, too early where they are made. Every one now carries its key across that gap. ZERO Hungarian error literals remain. util.MsgError does three things at once, each earned: - Error() is the Hungarian, byte for byte, so every un-converted printer is unchanged; - errors.Is answers for the kind AND for a wrapped cause (KindErrorf dropped the cause); - an error ARGUMENT renders recursively, so "formázás sikertelen: %w" translates whole. A foreign error — restic, docker, ssh, the stdlib — prints verbatim. It is not ours. 76 display sites go through errText, and TestNoErrErrorInPageOutput convicts any that do not. memoryVerdict returns an error rather than a sentence, so the deploy's 409 and the household's language come from one value; UpdateRefusal gained a Cause to carry it. Plurals, one rule, stated once: a key with .one/.other takes its COUNT first. Not a per-call-site flag — the producer somebody forgot would read "3 app is not running". The guard caught a real key collision (alert.deadapp.one) the day the rule landed. TWO DEFECTS FOUND IN MY OWN TOOLING, recorded rather than quietly fixed. The bulk converter silently dropped multi-line concatenations, damaging 7 producers — and the parity gate could not see it, because every surviving fragment WAS a real base literal while the CALL had lost text; two behaviour tests caught it. And the counting script was case-sensitive, so it said "0 left" while five remained. MinAgent: 0.131.0 (unchanged). No hub release needed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
427 lines
15 KiB
Go
427 lines
15 KiB
Go
package web
|
|
|
|
import (
|
|
"fmt"
|
|
"hash/crc32"
|
|
"log"
|
|
"sync"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/backup"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/i18n"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/monitor"
|
|
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
|
|
)
|
|
|
|
// Alert represents a persistent dashboard alert banner.
|
|
//
|
|
// LOCALISATION (v0.252.0, R-557). An alert is BUILT by a background health cycle and READ minutes
|
|
// later by whichever page the household opens, so its text cannot be rendered where it is made —
|
|
// that is the same shape as a flash in a redirect URL, one layer in. An alert therefore carries a
|
|
// bundle KEY plus its parameters, and GetAlerts renders it in the language the reader asked for.
|
|
//
|
|
// Two deliberate exceptions, both because the text is NOT ours to translate:
|
|
//
|
|
// - the health report's Issues and Warnings, which arrive as finished sentences and go ON THE WIRE
|
|
// to the hub (report `health.warnings`, pinned by internal/monitor's wire golden). They keep
|
|
// `Message` and stay Hungarian in every language until slice 3 gives the hub a language;
|
|
// - the agent-channel and endpoint-drift lines, whose text is composed by the channel-health
|
|
// checker. Filed as R-573.
|
|
//
|
|
// An alert with no MessageKey renders `Message` verbatim, which is what makes both exceptions work
|
|
// without a second mechanism.
|
|
type Alert struct {
|
|
ID string // unique identifier for filtering
|
|
Level string // "error", "warning", "info"
|
|
Message string // display text; used verbatim when MessageKey is empty
|
|
MessageKey string // bundle key for Message (preferred); empty = Message is already text
|
|
MessageArgs []interface{} // the key's printf parameters, in order
|
|
Link string // optional link to relevant page
|
|
LinkText string // link display text; used verbatim when LinkTextKey is empty
|
|
LinkTextKey string // bundle key for LinkText
|
|
PageOnly []string // if non-empty, only show on these pages (e.g., ["dashboard", "monitoring"])
|
|
Inline bool // if true, rendered by page template inline, not in layout banner
|
|
}
|
|
|
|
// rendered returns a copy of the alert with its keys resolved in lang. A key the bundle does not
|
|
// carry falls back to whatever Message/LinkText already held, so an alert can never render blank.
|
|
func (a Alert) rendered(lang string) Alert {
|
|
b, err := i18n.Shared()
|
|
if err != nil {
|
|
return a
|
|
}
|
|
if a.MessageKey != "" {
|
|
if len(a.MessageArgs) > 0 {
|
|
a.Message = b.Msgf(lang, a.MessageKey, a.MessageArgs...)
|
|
} else {
|
|
a.Message = b.Msg(lang, a.MessageKey)
|
|
}
|
|
}
|
|
if a.LinkTextKey != "" {
|
|
a.LinkText = b.Msg(lang, a.LinkTextKey)
|
|
}
|
|
return a
|
|
}
|
|
|
|
// AlertManager generates and stores dashboard alerts from health check results.
|
|
// Alerts are state-based (not event-based) — they reflect current system state
|
|
// and are regenerated after each health check cycle.
|
|
type AlertManager struct {
|
|
mu sync.RWMutex
|
|
alerts []Alert
|
|
logger *log.Logger
|
|
hubPushStatusFn func() HubPushStatusData
|
|
// agentChannelAlert is set/cleared by the channel-health checker (out-of-band from Refresh, which
|
|
// is health-report-driven). nil = channel up. It is prepended in GetAlerts so a dead controller→
|
|
// agent link (disk/storage UI broken) is always visible regardless of the health-report cycle.
|
|
agentChannelAlert *Alert
|
|
// deadAppAlerts (fix-3, CAMPAIGN-3) is set each cycle by the health loop from the deployed-app
|
|
// running-state view (which is NOT in the health report — it comes from the stack manager). Same
|
|
// out-of-band, state-based, self-clearing model as agentChannelAlert: passing an empty slice when
|
|
// every deployed app is running clears the banner with no manual dismissal.
|
|
deadAppAlerts []Alert
|
|
// endpointDriftAlert (R-77) is set/cleared at startup by the local_api drift check. Separate from
|
|
// agentChannelAlert on purpose: drift is usually the CAUSE and "agent unreachable" the SYMPTOM,
|
|
// and during the 2026-07-25 outage only the symptom was visible.
|
|
endpointDriftAlert *Alert
|
|
}
|
|
|
|
// NewAlertManager creates a new AlertManager.
|
|
func NewAlertManager(logger *log.Logger) *AlertManager {
|
|
return &AlertManager{
|
|
logger: logger,
|
|
}
|
|
}
|
|
|
|
// SetHubPushStatus sets the hub push status callback for generating hub alerts.
|
|
func (am *AlertManager) SetHubPushStatus(fn func() HubPushStatusData) {
|
|
am.mu.Lock()
|
|
am.hubPushStatusFn = fn
|
|
am.mu.Unlock()
|
|
}
|
|
|
|
// SetAgentChannelAlert sets (down=true) or clears (down=false) the controller→agent channel-down
|
|
// dashboard banner. Called by the channel-health checker each probe (idempotent). msg is the short
|
|
// Hungarian display line.
|
|
func (am *AlertManager) SetAgentChannelAlert(down bool, msg string) {
|
|
am.mu.Lock()
|
|
defer am.mu.Unlock()
|
|
if !down {
|
|
am.agentChannelAlert = nil
|
|
return
|
|
}
|
|
am.agentChannelAlert = &Alert{
|
|
ID: "agent-channel-down",
|
|
Level: "error",
|
|
Message: msg, // composed by the channel-health checker -- R-573
|
|
Link: "/settings",
|
|
LinkTextKey: "alert.link.settings",
|
|
}
|
|
}
|
|
|
|
// SetEndpointDriftAlert sets (drift=true) or clears the local_api endpoint-drift banner (R-77).
|
|
//
|
|
// It is deliberately a SEPARATE alert from SetAgentChannelAlert: during the 2026-07-25 outage the
|
|
// generic "agent unreachable" banner was the ONLY signal, and it looked like a dead agent. The two
|
|
// can also be true at once — a drifted endpoint usually CAUSES the channel to be down — so folding
|
|
// them together would hide the actionable one behind the symptom.
|
|
func (am *AlertManager) SetEndpointDriftAlert(drift bool, msg string) {
|
|
am.mu.Lock()
|
|
defer am.mu.Unlock()
|
|
if !drift {
|
|
am.endpointDriftAlert = nil
|
|
return
|
|
}
|
|
am.endpointDriftAlert = &Alert{
|
|
ID: "local-api-endpoint-drift",
|
|
Level: "error",
|
|
Message: msg, // composed by the endpoint-drift checker -- R-573
|
|
Link: "/settings",
|
|
LinkTextKey: "alert.link.settings",
|
|
}
|
|
}
|
|
|
|
// DeadApp is a deployed app the health loop found not-running (fix-3). State is the container-state
|
|
// string for the display (e.g. "stopped"/"exited").
|
|
type DeadApp struct {
|
|
Name string
|
|
DisplayName string
|
|
State string
|
|
}
|
|
|
|
// deadAppGroupThreshold: above this many dead apps, collapse to ONE grouped alert (a reboot storm
|
|
// with many NAS apps down should not paper the dashboard with a wall of banners — fix-3).
|
|
const deadAppGroupThreshold = 3
|
|
|
|
// buildDeadAppAlerts turns the dead-app list into dashboard alerts (WARN). ≤ threshold → one per app;
|
|
// more → a single grouped alert. Pure → unit-tested. Empty list → nil (clears the banner).
|
|
func buildDeadAppAlerts(dead []DeadApp) []Alert {
|
|
if len(dead) == 0 {
|
|
return nil
|
|
}
|
|
if len(dead) > deadAppGroupThreshold {
|
|
return []Alert{{
|
|
ID: "deadapp-group",
|
|
Level: "warning",
|
|
MessageKey: "alert.deadapp.group",
|
|
MessageArgs: []interface{}{len(dead)},
|
|
Link: "/monitoring",
|
|
LinkTextKey: "alert.link.monitoring",
|
|
}}
|
|
}
|
|
alerts := make([]Alert, 0, len(dead))
|
|
for _, d := range dead {
|
|
name := d.DisplayName
|
|
if name == "" {
|
|
name = d.Name
|
|
}
|
|
key, args := "alert.deadapp.single", []interface{}{name}
|
|
if d.State != "" {
|
|
key, args = "alert.deadapp.single_state", []interface{}{name, d.State}
|
|
}
|
|
alerts = append(alerts, Alert{
|
|
ID: "deadapp-" + simpleHash(d.Name),
|
|
Level: "warning",
|
|
MessageKey: key,
|
|
MessageArgs: args,
|
|
Link: "/monitoring",
|
|
LinkTextKey: "alert.link.monitoring",
|
|
})
|
|
}
|
|
return alerts
|
|
}
|
|
|
|
// SetDeadAppAlerts stores the current deployed-not-running banner set (fix-3). State-based: called
|
|
// each health cycle with the live dead-app list (empty clears it). Included in GetAlerts.
|
|
func (am *AlertManager) SetDeadAppAlerts(dead []DeadApp) {
|
|
alerts := buildDeadAppAlerts(dead)
|
|
am.mu.Lock()
|
|
am.deadAppAlerts = alerts
|
|
am.mu.Unlock()
|
|
}
|
|
|
|
// Refresh regenerates alerts from the latest health check report and config state.
|
|
// Called after each health check cycle (every 5 minutes) and on storage state changes.
|
|
func (am *AlertManager) Refresh(report *monitor.HealthReport, cfg *config.Config, backupMgr *backup.Manager, updateAvailable bool, latestVersion string, storagePaths ...[]settings.StoragePath) {
|
|
var alerts []Alert
|
|
|
|
// Disconnected storage alerts (top-level error banners on all pages)
|
|
if len(storagePaths) > 0 {
|
|
for _, sp := range storagePaths[0] {
|
|
if sp.Disconnected {
|
|
label := sp.Label
|
|
if label == "" {
|
|
label = sp.Path
|
|
}
|
|
alerts = append(alerts, Alert{
|
|
ID: "storage-disconnected-" + simpleHash(sp.Path),
|
|
Level: "error",
|
|
MessageKey: "alert.storage.disconnected",
|
|
MessageArgs: []interface{}{label, sp.Path},
|
|
Link: "/settings",
|
|
LinkTextKey: "alert.link.settings",
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
// From health check issues (critical)
|
|
for _, issue := range report.Issues {
|
|
alerts = append(alerts, Alert{
|
|
ID: "health-" + simpleHash(issue),
|
|
Level: "error",
|
|
Message: issue, // ON THE WIRE (report health.issues) -- not ours to translate; slice 3
|
|
Link: "/monitoring",
|
|
LinkTextKey: "alert.link.monitoring",
|
|
})
|
|
}
|
|
|
|
// From health check warnings
|
|
for i, w := range report.Warnings {
|
|
alert := Alert{
|
|
ID: "health-" + simpleHash(w),
|
|
Level: "warning",
|
|
Message: w, // ON THE WIRE (report health.warnings) -- not ours to translate; slice 3
|
|
Link: "/monitoring",
|
|
LinkTextKey: "alert.link.monitoring",
|
|
}
|
|
// R-553 — WHERE this warning is shown is decided by its KIND, not by the Hungarian words it
|
|
// contains. The old test was `strings.Contains(w, "meghajtón"/"adattároló"/"meghajtó")`, which
|
|
// matched exactly one of today's warnings (the not-on-a-separate-drive one, the only lower-case
|
|
// „meghajtón"); the day that sentence is translated the warning would jump from its quiet place
|
|
// under the storage bars to the red banner on EVERY page. Pinned by
|
|
// TestR553_DiskWarningPlacementSurvivesWordingChange.
|
|
if report.WarningKindAt(i) == monitor.WarnKindStorageNotSeparate {
|
|
alert.ID = "disk-not-separate"
|
|
alert.PageOnly = []string{"dashboard", "monitoring"}
|
|
alert.Inline = true
|
|
}
|
|
alerts = append(alerts, alert)
|
|
}
|
|
|
|
// Hub connection status
|
|
if !cfg.Hub.Enabled || cfg.Hub.URL == "" {
|
|
alerts = append(alerts, Alert{
|
|
ID: "hub-disabled",
|
|
Level: "warning",
|
|
MessageKey: "alert.hub.disabled",
|
|
Link: "/monitoring",
|
|
LinkTextKey: "alert.link.monitoring",
|
|
})
|
|
} else if am.hubPushStatusFn != nil {
|
|
ps := am.hubPushStatusFn()
|
|
if ps.LastError != "" && (ps.LastSuccess.IsZero() || time.Since(ps.LastSuccess) > 30*time.Minute) {
|
|
alerts = append(alerts, Alert{
|
|
ID: "hub-unreachable",
|
|
Level: "error",
|
|
MessageKey: "alert.hub.unreachable",
|
|
MessageArgs: []interface{}{ps.LastError},
|
|
Link: "/monitoring",
|
|
LinkTextKey: "alert.link.monitoring",
|
|
})
|
|
}
|
|
}
|
|
|
|
// Backup disabled
|
|
if !cfg.Backup.Enabled {
|
|
alerts = append(alerts, Alert{
|
|
ID: "backup-disabled",
|
|
Level: "warning",
|
|
MessageKey: "alert.backup.disabled",
|
|
Link: "/settings",
|
|
LinkTextKey: "alert.link.settings",
|
|
})
|
|
}
|
|
|
|
// Update available
|
|
if updateAvailable && latestVersion != "" {
|
|
alerts = append(alerts, Alert{
|
|
ID: "update-available",
|
|
Level: "info",
|
|
MessageKey: "alert.update.available",
|
|
MessageArgs: []interface{}{latestVersion},
|
|
Link: "/settings",
|
|
LinkTextKey: "alert.link.update",
|
|
})
|
|
}
|
|
|
|
// Sort: errors first, then warnings, then info
|
|
sortAlerts(alerts)
|
|
|
|
am.mu.Lock()
|
|
am.alerts = alerts
|
|
am.mu.Unlock()
|
|
}
|
|
|
|
// GetAlerts returns a copy of the current alerts, optionally excluding specific IDs.
|
|
func (am *AlertManager) GetAlerts(lang string, excludeIDs ...string) []Alert {
|
|
am.mu.RLock()
|
|
defer am.mu.RUnlock()
|
|
|
|
if len(am.alerts) == 0 && am.agentChannelAlert == nil && am.endpointDriftAlert == nil && len(am.deadAppAlerts) == 0 {
|
|
return nil
|
|
}
|
|
|
|
exclude := make(map[string]bool, len(excludeIDs))
|
|
for _, id := range excludeIDs {
|
|
exclude[id] = true
|
|
}
|
|
|
|
var result []Alert
|
|
// Endpoint drift first: it is the actionable CAUSE, and the channel-down banner below is usually
|
|
// just its symptom. Showing the symptom above the cause is what made the 2026-07-25 outage read
|
|
// as an infrastructure blip for 17.5 h.
|
|
if am.endpointDriftAlert != nil && !exclude[am.endpointDriftAlert.ID] {
|
|
result = append(result, *am.endpointDriftAlert)
|
|
}
|
|
// Channel-down is prepended (highest priority — the agent link being dead breaks disk/storage UI).
|
|
if am.agentChannelAlert != nil && !exclude[am.agentChannelAlert.ID] {
|
|
result = append(result, *am.agentChannelAlert)
|
|
}
|
|
// Dead-app banners (fix-3) — prepended after the channel alert: a deployed app being down is a
|
|
// high-signal state the operator/customer must see immediately.
|
|
for _, a := range am.deadAppAlerts {
|
|
if !exclude[a.ID] {
|
|
result = append(result, a)
|
|
}
|
|
}
|
|
for _, a := range am.alerts {
|
|
if exclude[a.ID] {
|
|
continue
|
|
}
|
|
result = append(result, a)
|
|
}
|
|
|
|
// Cap at 5 visible alerts
|
|
if len(result) > 5 {
|
|
overflow := len(result) - 5
|
|
result = result[:5]
|
|
result = append(result, Alert{
|
|
ID: "overflow",
|
|
Level: "info",
|
|
MessageKey: "alert.overflow", MessageArgs: []interface{}{overflow},
|
|
Link: "/monitoring",
|
|
})
|
|
}
|
|
|
|
// Rendered LAST, once, on the way out: the alerts are stored as keys and only the reader knows
|
|
// the language. Every return path goes through here, so an alert cannot escape with a raw key.
|
|
for i := range result {
|
|
result[i] = result[i].rendered(lang)
|
|
}
|
|
return result
|
|
}
|
|
|
|
// GetInlineAlerts returns alerts marked as Inline for a specific page.
|
|
func (am *AlertManager) GetInlineAlerts(page, lang string) []Alert {
|
|
am.mu.RLock()
|
|
defer am.mu.RUnlock()
|
|
|
|
var result []Alert
|
|
for _, a := range am.alerts {
|
|
if !a.Inline {
|
|
continue
|
|
}
|
|
if len(a.PageOnly) == 0 {
|
|
result = append(result, a)
|
|
continue
|
|
}
|
|
for _, p := range a.PageOnly {
|
|
if p == page {
|
|
result = append(result, a)
|
|
break
|
|
}
|
|
}
|
|
}
|
|
for i := range result {
|
|
result[i] = result[i].rendered(lang)
|
|
}
|
|
return result
|
|
}
|
|
|
|
// simpleHash returns a short deterministic hash for deduplication.
|
|
func simpleHash(s string) string {
|
|
return fmt.Sprintf("%08x", crc32.ChecksumIEEE([]byte(s)))
|
|
}
|
|
|
|
// sortAlerts sorts alerts by severity: error > warning > info.
|
|
func sortAlerts(alerts []Alert) {
|
|
levelOrder := map[string]int{"error": 0, "warning": 1, "info": 2}
|
|
for i := 1; i < len(alerts); i++ {
|
|
for j := i; j > 0 && levelOrder[alerts[j].Level] < levelOrder[alerts[j-1].Level]; j-- {
|
|
alerts[j], alerts[j-1] = alerts[j-1], alerts[j]
|
|
}
|
|
}
|
|
}
|
|
|
|
func countLevel(alerts []Alert, level string) int {
|
|
n := 0
|
|
for _, a := range alerts {
|
|
if a.Level == level {
|
|
n++
|
|
}
|
|
}
|
|
return n
|
|
}
|