05d81810d4
Hub half of the management-plane break-glass (prereq for felhom-sshd/H1; agent
half = felhom-agent v0.71.0). Closes SPIKE-felhom-sshd §8/#9.
- store.host_recovery + methods: per-host root@pam console password, at-rest,
operator-retrievable (the PVE-web-console fallback when sshd + auto-heal both fail).
- API: PUT /hosts/{id}/recovery-credential (self-scoped, day-0 vaults) + GET
/admin/hosts/{id}/recovery-credential (global key only). Secret never logged
(red-proofed).
- monitor/host_mgmtplane: parses the agent mgmt_plane stanza, raises
mgmt_plane_healed WARNING on a new privsep_healed_at (recurring clobber surfaces
before lockout; complements host_staleness).
- host-install: step_break_glass generates a strong root@pam password (openssl
rand, never logged/filed — stdin to chpasswd + curl), vaults via host key;
idempotent unless --rotate-recovery. Installs the G1 host artifacts (tmpfiles +
agent-independent watchdog timer), RuntimeDirectory-guarded; uninstall removes them.
Hub v0.34.0. Non-hollow tests + red-proofs; full suite green.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PSK5g6qYLknKj8u3QAFEr6
123 lines
4.4 KiB
Go
123 lines
4.4 KiB
Go
package monitor
|
|
|
|
import (
|
|
"encoding/json"
|
|
"log"
|
|
"sync"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
|
|
)
|
|
|
|
// HostMgmtPlaneChecker raises an operator WARNING when a host's agent-independent break-glass watchdog
|
|
// AUTO-HEALED a missing /run/sshd privsep dir (TASK G1). The heal itself is silent and login-free (the
|
|
// point of the watchdog); this surfaces a RECURRING clobber so the operator can find the cause BEFORE
|
|
// it becomes a full management lockout — complementing HostStalenessChecker (which only catches a box
|
|
// gone silent). Sibling of HostLeafChecker; runs on the same 60s sweep.
|
|
//
|
|
// Design (mirrors HostLeafChecker's trust-on-first-report): the state is the host's last-seen
|
|
// privsep_healed_at marker timestamp. The watchdog rewrites the marker on EACH heal, so a new, different
|
|
// timestamp = a new heal event → one warning. The first observation of a non-empty timestamp seeds the
|
|
// baseline WITHOUT alerting (it may be a heal from before the hub was watching — avoid a false alarm on
|
|
// startup; a genuinely recurring cause re-heals and re-alerts on the next occurrence). An empty
|
|
// timestamp (healthy host / old agent) never alerts and never overwrites a baseline.
|
|
type HostMgmtPlaneChecker struct {
|
|
store *store.Store
|
|
logger *log.Logger
|
|
onEvent EventNotifyFunc
|
|
|
|
mu sync.Mutex
|
|
states map[string]string // hostID → last-seen privsep_healed_at
|
|
customerOf map[string]string
|
|
}
|
|
|
|
// NewHostMgmtPlaneChecker seeds per-host baselines from the latest reports. No events on init.
|
|
func NewHostMgmtPlaneChecker(s *store.Store, onEvent EventNotifyFunc, logger *log.Logger) *HostMgmtPlaneChecker {
|
|
mc := &HostMgmtPlaneChecker{
|
|
store: s,
|
|
logger: logger,
|
|
onEvent: onEvent,
|
|
states: make(map[string]string),
|
|
customerOf: make(map[string]string),
|
|
}
|
|
rows, err := s.GetHostMgmtPlaneStates()
|
|
if err != nil {
|
|
logger.Printf("[WARN] Host mgmt-plane checker: failed to seed: %v", err)
|
|
return mc
|
|
}
|
|
seeded := 0
|
|
for _, row := range rows {
|
|
if s.IsCustomerBlocked(row.CustomerID) || row.PrivsepHealedAt == "" {
|
|
continue
|
|
}
|
|
mc.customerOf[row.HostID] = row.CustomerID
|
|
mc.states[row.HostID] = row.PrivsepHealedAt
|
|
seeded++
|
|
}
|
|
logger.Printf("[INFO] Host mgmt-plane checker initialized: %d host heal-state(s) seeded", seeded)
|
|
return mc
|
|
}
|
|
|
|
// Check evaluates all hosts and emits mgmt_plane_healed on a NEW heal timestamp.
|
|
func (mc *HostMgmtPlaneChecker) Check() {
|
|
rows, err := mc.store.GetHostMgmtPlaneStates()
|
|
if err != nil {
|
|
mc.logger.Printf("[WARN] Host mgmt-plane check failed: %v", err)
|
|
return
|
|
}
|
|
mc.mu.Lock()
|
|
defer mc.mu.Unlock()
|
|
|
|
seen := make(map[string]bool, len(rows))
|
|
for _, row := range rows {
|
|
if mc.store.IsCustomerBlocked(row.CustomerID) {
|
|
delete(mc.states, row.HostID)
|
|
continue
|
|
}
|
|
seen[row.HostID] = true
|
|
if row.PrivsepHealedAt == "" {
|
|
continue // no heal marker → healthy / old agent → no alert, no baseline change
|
|
}
|
|
mc.customerOf[row.HostID] = row.CustomerID
|
|
old := mc.states[row.HostID]
|
|
if old == "" {
|
|
mc.states[row.HostID] = row.PrivsepHealedAt // first observation → seed, no event
|
|
continue
|
|
}
|
|
if old == row.PrivsepHealedAt {
|
|
continue // same heal already alerted
|
|
}
|
|
mc.states[row.HostID] = row.PrivsepHealedAt
|
|
mc.emit(row.HostID, row.CustomerID, row.PrivsepHealedAt)
|
|
}
|
|
|
|
for id := range mc.states {
|
|
if !seen[id] {
|
|
delete(mc.states, id)
|
|
}
|
|
}
|
|
}
|
|
|
|
// GetState returns the last-seen heal timestamp for a host ("" if none).
|
|
func (mc *HostMgmtPlaneChecker) GetState(hostID string) string {
|
|
mc.mu.Lock()
|
|
defer mc.mu.Unlock()
|
|
return mc.states[hostID]
|
|
}
|
|
|
|
func (mc *HostMgmtPlaneChecker) emit(hostID, customerID, healedAt string) {
|
|
msg := "Host " + hostID + ": the management-plane privsep dir (/run/sshd) was missing and was AUTO-HEALED by the watchdog at " + healedAt +
|
|
" — a recurring cause can lead to an SSH lockout; investigate (e.g. a unit declaring RuntimeDirectory=sshd)."
|
|
details, _ := json.Marshal(map[string]string{
|
|
"host_id": hostID,
|
|
"privsep_healed_at": healedAt,
|
|
})
|
|
mc.logger.Printf("[WARN] Host mgmt-plane: %s privsep dir auto-healed at %s (mgmt_plane_healed)", hostID, healedAt)
|
|
if _, err := mc.store.SaveEvent(customerID, "mgmt_plane_healed", "warning", msg, string(details), "hub"); err != nil {
|
|
mc.logger.Printf("[WARN] save mgmt_plane_healed for %s: %v", hostID, err)
|
|
return
|
|
}
|
|
if mc.onEvent != nil {
|
|
mc.onEvent(customerID, "mgmt_plane_healed", "warning", msg, string(details), "hub")
|
|
}
|
|
}
|