From 07959e61b577ac2288fc3561c0b721df385d51a6 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Tue, 15 Sep 2026 10:05:41 +0200 Subject: [PATCH] hub v0.114.0: self-bind auto-send while a customer waits for a box (R-509); node_* bypass the quiet hour (ruling 2026-09-15); PBS re-issue adopts an endpoint token (R-511); controller supervisor events (R-523); event registers Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- hub/CHANGELOG.md | 34 +++++ hub/cmd/hub/main.go | 3 + .../api/controller_supervisor_event_test.go | 21 +++ hub/internal/api/handler.go | 11 ++ hub/internal/monitor/controller_supervisor.go | 136 ++++++++++++++++++ .../monitor/controller_supervisor_test.go | 95 ++++++++++++ hub/internal/notify/dispatcher.go | 43 +++++- .../notify/node_liveness_cooldown_test.go | 53 +++++++ hub/internal/store/store.go | 30 ++++ hub/internal/web/configs.go | 18 +++ hub/internal/web/hosts.go | 4 + hub/internal/web/pbsdr.go | 73 +++++++++- hub/internal/web/pbsdr_test.go | 57 +++++++- hub/internal/web/selfbind_mint.go | 33 +++++ hub/internal/web/selfbind_waiting_test.go | 90 ++++++++++++ .../web/templates/customer_unified.html | 1 + 16 files changed, 693 insertions(+), 9 deletions(-) create mode 100644 hub/internal/api/controller_supervisor_event_test.go create mode 100644 hub/internal/monitor/controller_supervisor.go create mode 100644 hub/internal/monitor/controller_supervisor_test.go create mode 100644 hub/internal/notify/node_liveness_cooldown_test.go create mode 100644 hub/internal/web/selfbind_waiting_test.go diff --git a/hub/CHANGELOG.md b/hub/CHANGELOG.md index 826fed61..568cf3e4 100644 --- a/hub/CHANGELOG.md +++ b/hub/CHANGELOG.md @@ -1,3 +1,37 @@ +## v0.114.0 — the connect e-mail goes by itself, „box is down" skips the quiet hour, a stuck PBS token is adopted (2026-09-15, R-509 / R-511 / R-523 / R-518 / R-514) + +- **R-509 (P1, operator decision A 2026-09-15) — the self-bind link goes out whenever a customer is + waiting for a box.** Before: only at customer creation and RESET, so BIGNIGHT's tester-1 (e-mail + added later, previous host deleted) never got the mail its console promised. New triggers, each + re-checking that the customer has NO bound host: an e-mail set or changed on the config save, and a + host delete (`autoMintSelfBindIfWaiting`). Appliance registration is deliberately not a trigger — + it knows no customer (`api/appliance.go`). Every send is recorded as a hub-internal + `selfbind_link_sent` event with its occasion, and the Setup tab reads „Kapcsolódó link elküldve: + ()"; the button stays as the manual resend. Red-proof: without the config-save + call, `TestSelfBind_EmailSetOnWaitingCustomerSendsLink` fails at "sent no link". +- **Operator ruling 2026-09-15 (decision A) — `node_stale`, `node_down`, `node_recovered` bypass the + 1-hour operator cooldown**, with a 5-minute dedupe (`operatorCooldownFor`). This REVERSES the + documented design in `08-alarm-ladder.md` §5, recorded there as the operator's ruling. BIGNIGHT F9: + the dead box's `node_stale` and later `node_recovered` mails were both suppressed by F8's cooldown. + `host_*` siblings are not in the ruling and keep the hour. Red-proof: with the hour for every type, + the 39-minute `node_stale` is suppressed. +- **R-511 — „Re-issue PBS credentials" adopts an endpoint token when the hub has no descriptor.** + Before: `400 No provisioned PBS DR tier` — the one action the WG hook's error told the operator to + use. Now, when the customer's DR-tier flag is on, it re-keys the endpoint's token (`tenantsync + reissue`, the op F-14 already uses) and writes a fresh descriptor + consume-once secret; audit event + `pbsdr_adopted`. DR flag off → still 400, endpoint untouched (pinned). Red-proof: the old 400 in + place of the adopt branch fails the tester-1-shaped test. **Not built:** a „PBS token elengedése" + button on host delete — the endpoint's only removal op (`deprovision`) destroys the namespace and + every backup group, so a token-only release is a new ep0 operation; filed as a row. +- **R-523 — `ControllerSupervisorChecker`.** Reads the agent v0.131.0 `controller_supervisor` report + stanza and mints `controller_restarted_by_agent` (info) when a guest's `last_restart_at` moves and + `controller_crashloop` (error) when `crashloop_since` moves — timestamps, not counters, so an agent + restart cannot lose or invent an event; seeds silently, except a crash-loop at first sight. +- **Event registers.** `controller_restarted_by_agent`, `controller_crashloop`, `backup_tier_skipped` + (controller v0.243.0, R-518) and `app_oom` (controller v0.243.0, R-514) are in `allowedEventTypes` + AND `operatorOnlyEvents`; `selfbind_link_sent` and `pbsdr_adopted` are allowlisted hub-internal + audit rows. Pinned by `TestControllerSupervisorEventsAreAllowlistedAndOperatorOnly`. + ## v0.113.0 — the operator is told to hand over the passphrase, and the mail says who hands it (2026-09-14, R-497) **What was wrong, measured in the 2026-09-14 first-hour drill.** The five-word „Tulajdonosi jelmondat" is diff --git a/hub/cmd/hub/main.go b/hub/cmd/hub/main.go index bb0cf805..8ae2dac3 100644 --- a/hub/cmd/hub/main.go +++ b/hub/cmd/hub/main.go @@ -597,6 +597,8 @@ func main() { // offsite_delivery_stuck (warning, 24h/customer) and self-heals it via the Re-issue path // (offsite_credential_restaged, one restage/customer/24h, R-39(a)-guarded). Cooldowns are // durable (events table), so a hub restart neither floods nor silently re-heals. + // R-523: the agent's in-guest controller supervisor → controller_restarted_by_agent / controller_crashloop. + controllerSupervisorChecker := monitor.NewControllerSupervisorChecker(dataStore, dispatcher.ProcessEvent, logger) offsiteDeliveryChecker := monitor.NewOffsiteDeliveryChecker(dataStore, offsiteHealReissuer, dispatcher.ProcessEvent, logger) go func() { ticker := time.NewTicker(60 * time.Second) @@ -617,6 +619,7 @@ func main() { offsiteChecker.Check() restoreTestChecker.Check() // R-85: restore-test failure + per-tier staleness offsiteDeliveryChecker.Check() + controllerSupervisorChecker.Check() if offsiteBoxChecker != nil { offsiteBoxChecker.Check() // R-5: restic pool-box aggregate (fetch-throttled internally) } diff --git a/hub/internal/api/controller_supervisor_event_test.go b/hub/internal/api/controller_supervisor_event_test.go new file mode 100644 index 00000000..da25c51c --- /dev/null +++ b/hub/internal/api/controller_supervisor_event_test.go @@ -0,0 +1,21 @@ +package api + +import ( + "testing" + + "gitea.dooplex.hu/admin/felhom-hub/internal/notify" +) + +// R-523 — both halves of the register for the two supervisor event types (the R-158 pattern): +// allowlisted (the project's single register of legitimate types) AND operator-only (a missing +// customerMessages entry is not a block — the v0.78.0 defect). +func TestControllerSupervisorEventsAreAllowlistedAndOperatorOnly(t *testing.T) { + for _, et := range []string{"controller_restarted_by_agent", "controller_crashloop"} { + if !allowedEventTypes[et] { + t.Fatalf("%s must be in allowedEventTypes (R-77: an unregistered type is an inert seam)", et) + } + if !notify.IsOperatorOnly(et) { + t.Fatalf("%s must be operator-only — a household would get raw English about host ids", et) + } + } +} diff --git a/hub/internal/api/handler.go b/hub/internal/api/handler.go index f99a5908..3d7bd41e 100644 --- a/hub/internal/api/handler.go +++ b/hub/internal/api/handler.go @@ -2106,6 +2106,17 @@ var allowedEventTypes = map[string]bool{ "storage_fill_critical": true, "expected_backup_missed": true, "expected_dbdump_missed": true, + // R-523 (hub v0.114.0, agent v0.131.0) — hub-generated from the agent's controller_supervisor + // report stanza (monitor/controller_supervisor.go). Both operator-only (notify.operatorOnlyEvents). + "controller_restarted_by_agent": true, + "controller_crashloop": true, + // R-509 / R-511 / R-518 / R-514 (hub v0.114.0). selfbind_link_sent and pbsdr_adopted are + // hub-internal audit rows (stored, not dispatched). backup_tier_skipped and app_oom are pushed by + // controller v0.243.0; both operator-only (notify.operatorOnlyEvents). + "selfbind_link_sent": true, + "pbsdr_adopted": true, + "backup_tier_skipped": true, + "app_oom": true, // Special "test": true, } diff --git a/hub/internal/monitor/controller_supervisor.go b/hub/internal/monitor/controller_supervisor.go new file mode 100644 index 00000000..ec83754d --- /dev/null +++ b/hub/internal/monitor/controller_supervisor.go @@ -0,0 +1,136 @@ +package monitor + +import ( + "encoding/json" + "fmt" + "log" + "sync" + + "gitea.dooplex.hu/admin/felhom-hub/internal/store" +) + +// R-523 (felhom-agent v0.131.0). The agent's in-guest controller supervisor restarts a controller +// container that is not running and, after 3 restarts in 15 minutes, gives up for 30 minutes. The +// agent has no event channel of its own — its heartbeat is the channel — so the per-guest record +// rides the host report as `controller_supervisor`, and this checker turns movement in it into events: +// +// - `controller_restarted_by_agent` (info) when a guest's last_restart_at MOVES to a new value; +// - `controller_crashloop` (error, operator-only) when a guest's crashloop_since MOVES. +// +// TIMESTAMPS, NOT COUNTERS. The agent's record is in-memory, so an agent restart zeroes +// restarts_total; keying on a counter would read that as nothing (fine) but a later 1 would not +// exceed the remembered 3 (a lost event). A timestamp that changes is a new act whatever the counter +// says. +// +// First observation (hub restart, new host): SEED SILENTLY, except a guest that is IN a crash-loop at +// first sight emits once — a hub restarted during an outage must not stay silent about it (the +// HostCapabilityChecker F2 rule). Both types are pinned in allowedEventTypes and operatorOnlyEvents +// by controller_supervisor_event_test.go in the api package. +type ControllerSupervisorChecker struct { + store *store.Store + logger *log.Logger + onEvent EventNotifyFunc + + mu sync.Mutex + seen map[string]supSeen // hostID/vmid → last timestamps seen +} + +type supSeen struct { + lastRestartAt string + crashloopSince string +} + +// ControllerSupervisorGuest mirrors the agent's hub.ControllerSupervisorGuest wire shape. +type ControllerSupervisorGuest struct { + VMID int `json:"vmid"` + RestartsTotal int `json:"restarts_total"` + LastRestartAt string `json:"last_restart_at"` + LastReason string `json:"last_reason"` + Crashloop bool `json:"crashloop"` + CrashloopSince string `json:"crashloop_since"` + Parked bool `json:"parked"` +} + +// ParseControllerSupervisor extracts the stanza from a host report body. A missing or malformed +// stanza (a pre-v0.131.0 agent) yields nil — never an event. +func ParseControllerSupervisor(reportJSON string) []ControllerSupervisorGuest { + var body struct { + CS *struct { + Guests []ControllerSupervisorGuest `json:"guests"` + } `json:"controller_supervisor"` + } + if err := json.Unmarshal([]byte(reportJSON), &body); err != nil || body.CS == nil { + return nil + } + return body.CS.Guests +} + +func NewControllerSupervisorChecker(s *store.Store, onEvent EventNotifyFunc, logger *log.Logger) *ControllerSupervisorChecker { + return &ControllerSupervisorChecker{store: s, logger: logger, onEvent: onEvent, seen: map[string]supSeen{}} +} + +// Check reads every host's latest report and emits on movement. Same 60 s sweep as its siblings. +func (c *ControllerSupervisorChecker) Check() { + rows, err := c.store.GetLatestHostReports() + if err != nil { + c.logger.Printf("[WARN] Controller supervisor check failed: %v", err) + return + } + for _, row := range rows { + if c.store.IsCustomerBlocked(row.CustomerID) { + continue + } + c.observe(row.HostID, row.CustomerID, ParseControllerSupervisor(row.ReportJSON)) + } +} + +func (c *ControllerSupervisorChecker) observe(hostID, customerID string, guests []ControllerSupervisorGuest) { + c.mu.Lock() + defer c.mu.Unlock() + for _, g := range guests { + key := fmt.Sprintf("%s/%d", hostID, g.VMID) + prev, known := c.seen[key] + c.seen[key] = supSeen{lastRestartAt: g.LastRestartAt, crashloopSince: g.CrashloopSince} + if !known { + if g.Crashloop && g.CrashloopSince != "" { + c.emit(customerID, hostID, g, "controller_crashloop") + } + continue + } + if g.LastRestartAt != "" && g.LastRestartAt != prev.lastRestartAt { + c.emit(customerID, hostID, g, "controller_restarted_by_agent") + } + if g.CrashloopSince != "" && g.CrashloopSince != prev.crashloopSince { + c.emit(customerID, hostID, g, "controller_crashloop") + } + } +} + +func (c *ControllerSupervisorChecker) emit(customerID, hostID string, g ControllerSupervisorGuest, eventType string) { + var severity, message string + switch eventType { + case "controller_restarted_by_agent": + severity = "info" + message = fmt.Sprintf("Host %s guest %d: the agent restarted the controller (%s) — restart #%d since the agent started", + hostID, g.VMID, g.LastReason, g.RestartsTotal) + case "controller_crashloop": + severity = "error" + message = fmt.Sprintf("Host %s guest %d: the controller will not stay up — the agent stopped restarting it after repeated attempts (since %s) and will try again in 30 minutes. Last reason: %s", + hostID, g.VMID, g.CrashloopSince, g.LastReason) + default: + return + } + details, _ := json.Marshal(map[string]any{ + "host_id": hostID, "vmid": g.VMID, "restarts_total": g.RestartsTotal, + "last_restart_at": g.LastRestartAt, "last_reason": g.LastReason, + "crashloop_since": g.CrashloopSince, "parked": g.Parked, + }) + c.logger.Printf("[INFO] Controller supervisor: %s %s/%d (%s)", eventType, hostID, g.VMID, g.LastReason) + if _, err := c.store.SaveEvent(customerID, eventType, severity, message, string(details), "hub"); err != nil { + c.logger.Printf("[WARN] save %s for %s: %v", eventType, hostID, err) + return + } + if c.onEvent != nil { + c.onEvent(customerID, eventType, severity, message, string(details), "hub") + } +} diff --git a/hub/internal/monitor/controller_supervisor_test.go b/hub/internal/monitor/controller_supervisor_test.go new file mode 100644 index 00000000..ab96e4a4 --- /dev/null +++ b/hub/internal/monitor/controller_supervisor_test.go @@ -0,0 +1,95 @@ +package monitor + +import ( + "io" + "log" + "sync" + "testing" + + "gitea.dooplex.hu/admin/felhom-hub/internal/store" +) + +// R-523 — the checker that turns the agent's controller_supervisor stanza into events. +// The JSON below is the exact wire shape the agent's TestControllerSupervisorStanza_WireShape pins. + +func supReport(guests string) []byte { + return []byte(`{"host_id":"h1","controller_supervisor":{"guests":` + guests + `}}`) +} + +type evRec struct { + mu sync.Mutex + evs []string +} + +func (r *evRec) fn(_, et, sev, _, _, _ string) { + r.mu.Lock() + defer r.mu.Unlock() + r.evs = append(r.evs, et+"/"+sev) +} +func (r *evRec) take() []string { + r.mu.Lock() + defer r.mu.Unlock() + out := r.evs + r.evs = nil + return out +} + +// The consequence: an agent restart of a controller reaches the dispatcher as +// controller_restarted_by_agent, once; a crash-loop reaches it as controller_crashloop (error). +// +// RED-PROOF: make observe() skip the `LastRestartAt != prev.lastRestartAt` branch → the second +// Check emits nothing → "restart did not produce controller_restarted_by_agent". +func TestControllerSupervisorChecker_EmitsOnMovement(t *testing.T) { + st := newCapStore(t) + rec := &evRec{} + c := NewControllerSupervisorChecker(st, rec.fn, log.New(io.Discard, "", 0)) + + // Seed: an old restart already recorded at first sight → silent. + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":1,"last_restart_at":"2026-09-15T08:00:00Z","last_reason":"exited","crashloop":false,"parked":false}]`), store.HostReportDenorm{}) + c.Check() + if evs := rec.take(); len(evs) != 0 { + t.Fatalf("first observation must seed silently, got %v", evs) + } + // Same record again → silent. + c.Check() + if evs := rec.take(); len(evs) != 0 { + t.Fatalf("unchanged record emitted %v", evs) + } + // A new restart. + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":2,"last_restart_at":"2026-09-15T09:00:00Z","last_reason":"exited","crashloop":false,"parked":false}]`), store.HostReportDenorm{}) + c.Check() + if evs := rec.take(); len(evs) != 1 || evs[0] != "controller_restarted_by_agent/info" { + t.Fatalf("restart did not produce controller_restarted_by_agent (got %v)", evs) + } + // Agent restarted (counter back to 1) with a NEW timestamp → still an event. + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":1,"last_restart_at":"2026-09-15T10:00:00Z","last_reason":"absent","crashloop":false,"parked":false}]`), store.HostReportDenorm{}) + c.Check() + if evs := rec.take(); len(evs) != 1 || evs[0] != "controller_restarted_by_agent/info" { + t.Fatalf("a restart after an agent restart (counter reset) was lost: %v", evs) + } + // Crash-loop. + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":4,"last_restart_at":"2026-09-15T10:00:00Z","last_reason":"exited","crashloop":true,"crashloop_since":"2026-09-15T10:05:00Z","parked":false}]`), store.HostReportDenorm{}) + c.Check() + if evs := rec.take(); len(evs) != 1 || evs[0] != "controller_crashloop/error" { + t.Fatalf("crash-loop did not produce controller_crashloop/error (got %v)", evs) + } + // A pre-v0.131.0 report (no stanza) → nothing. + st.SaveHostReport("h1", "c1", []byte(`{"host_id":"h1"}`), store.HostReportDenorm{}) + c.Check() + if evs := rec.take(); len(evs) != 0 { + t.Fatalf("a report without the stanza emitted %v", evs) + } +} + +// A hub restarted DURING a crash-loop must say so once, not seed it away. +func TestControllerSupervisorChecker_CrashloopAtFirstSightEmits(t *testing.T) { + st := newCapStore(t) + rec := &evRec{} + c := NewControllerSupervisorChecker(st, rec.fn, log.New(io.Discard, "", 0)) + st.SaveHostReport("h1", "c1", supReport(`[{"vmid":9201,"restarts_total":3,"last_restart_at":"2026-09-15T10:00:00Z","crashloop":true,"crashloop_since":"2026-09-15T10:05:00Z"}]`), store.HostReportDenorm{}) + c.Check() + c.Check() + if evs := rec.take(); len(evs) != 1 || evs[0] != "controller_crashloop/error" { + t.Fatalf("crash-loop at first sight: want exactly one controller_crashloop, got %v", evs) + } +} diff --git a/hub/internal/notify/dispatcher.go b/hub/internal/notify/dispatcher.go index 2b191525..7e93e330 100644 --- a/hub/internal/notify/dispatcher.go +++ b/hub/internal/notify/dispatcher.go @@ -396,6 +396,33 @@ func cooldownStackSuffix(eventType, detailsJSON string) string { return ":" + d.StackName } +// nodeLivenessEvents skip the 1-hour operator cooldown (OPERATOR RULING 2026-09-15, decision A; +// 08-alarm-ladder.md §5). BIGNIGHT F9: the box was dead for 33 minutes and the `node_stale` mail was +// suppressed because F8's `node_stale` had used the hour 39 minutes earlier; the `node_recovered` +// mail was suppressed the same way. "The box is down" must not wait out a quiet hour. A 5-minute +// dedupe stays, so a flapping link cannot mail every sweep. The key is unchanged (customer:type), and +// a customer has one box, so the dedupe is per host. host_* (agent-plane) siblings are NOT in the +// ruling and keep the hour. +var nodeLivenessEvents = map[string]bool{ + "node_stale": true, + "node_down": true, + "node_recovered": true, +} + +const ( + operatorCooldown = 1 * time.Hour + nodeLivenessDedupeWindow = 5 * time.Minute +) + +// operatorCooldownFor returns the operator-mail cooldown for an event type. Pinned by +// TestOperatorCooldown_NodeLivenessBypassesQuietHour. +func operatorCooldownFor(eventType string) time.Duration { + if nodeLivenessEvents[eventType] { + return nodeLivenessDedupeWindow + } + return operatorCooldown +} + func (d *Dispatcher) processOperator(customerID, eventType, severity, message, detailsJSON, source string) { if !d.operatorOn || d.operatorEmail == "" { return @@ -406,8 +433,9 @@ func (d *Dispatcher) processOperator(customerID, eventType, severity, message, d cooldownKey := customerID + ":" + eventType + cooldownTierSuffix(detailsJSON) + cooldownRunSuffix(detailsJSON) + cooldownStackSuffix(eventType, detailsJSON) + window := operatorCooldownFor(eventType) d.mu.Lock() - if last, ok := d.opCooldowns[cooldownKey]; ok && time.Since(last) < 1*time.Hour { + if last, ok := d.opCooldowns[cooldownKey]; ok && time.Since(last) < window { d.mu.Unlock() // R-182: RECORD THE SUPPRESSION. This used to be a bare `return` — the event was dropped // before any LogNotification, so a cooldown drop and an event that never happened were @@ -423,7 +451,7 @@ func (d *Dispatcher) processOperator(customerID, eventType, severity, message, d // applies to EVERY operator event, not only the one that exposed it. It makes the drop // visible; it deliberately does NOT change the cooldown's duration or semantics. if err := d.store.LogNotification(customerID, eventType, severity, message, - "suppressed", "operator cooldown 1h, key="+cooldownKey, "operator"); err != nil { + "suppressed", "operator cooldown "+window.String()+", key="+cooldownKey, "operator"); err != nil { d.logger.Printf("[WARN] Failed to record suppressed operator notification for %s/%s: %v", customerID, eventType, err) } @@ -554,6 +582,17 @@ var operatorOnlyEvents = map[string]bool{ // mints the type — an operator-tier type absent from this register reaches customers as raw // English (the v0.78.0 defect recorded above). "escrow_blob_served": true, + // R-523 (v0.114.0). The agent restarted a dead in-guest controller, or gave up after repeated + // restarts. Operator-grade (host ids, vmids, raw docker states) and not actionable by a household + // — the customer's side of this is the dashboard coming back. Registered in the same commit that + // mints the types. + "controller_restarted_by_agent": true, + "controller_crashloop": true, + // R-518 / R-514 (v0.114.0, controller v0.243.0). A whole-guest tier skipped for absent storage is a + // provisioning fact the household cannot act on; an OOM-killed app process carries raw container + // names — the household's side is the dashboard tag. Registered in the same commit. + "backup_tier_skipped": true, + "app_oom": true, } // IsOperatorOnly reports whether an event type is barred from customer dispatch. Exported so the diff --git a/hub/internal/notify/node_liveness_cooldown_test.go b/hub/internal/notify/node_liveness_cooldown_test.go new file mode 100644 index 00000000..08abaf9e --- /dev/null +++ b/hub/internal/notify/node_liveness_cooldown_test.go @@ -0,0 +1,53 @@ +package notify + +import ( + "io" + "log" + "sync" + "testing" + "time" +) + +// Operator ruling 2026-09-15 (decision A) — "box is down" mail is not swallowed by the quiet hour. +// BIGNIGHT F9: a node_stale 39 minutes after F8's node_stale was suppressed by the 1-hour cooldown. +// +// RED-PROOF (run 2026-09-15, recorded in REPORT.md): with operatorCooldownFor returning the hour for +// every type (the pre-ruling behaviour), the 39-minute node_stale was suppressed and this failed at +// "node_stale 39 min after the previous one was suppressed". +func TestOperatorCooldown_NodeLivenessBypassesQuietHour(t *testing.T) { + st := newDispStore(t) + d := NewDispatcher(st, "test-key", "from@felhom.eu", "op@felhom.eu", true, log.New(io.Discard, "", 0)) + var mu sync.Mutex + sent := 0 + d.sendEmailFn = func(string, string, string, map[string]string) error { mu.Lock(); defer mu.Unlock(); sent++; return nil } + count := func() int { mu.Lock(); defer mu.Unlock(); return sent } + + // The previous node_stale mail went 39 minutes ago. + d.mu.Lock() + d.opCooldowns["c1:node_stale"] = time.Now().Add(-39 * time.Minute) + d.opCooldowns["c1:app_start_failed"] = time.Now().Add(-39 * time.Minute) + d.mu.Unlock() + + d.processOperator("c1", "node_stale", "warning", "box stale", "{}", "hub") + if count() != 1 { + t.Fatalf("node_stale 39 min after the previous one was suppressed (sent=%d) — the BIGNIGHT F9 silence", count()) + } + // A flap 1 minute later is deduped. + d.processOperator("c1", "node_stale", "warning", "box stale again", "{}", "hub") + if count() != 1 { + t.Fatalf("node_stale 1 min after a sent one was mailed again (sent=%d) — the 5-minute dedupe is gone", count()) + } + // The design is unchanged for every other type: still the hour. + d.processOperator("c1", "app_start_failed", "warning", "app down", "{}", "hub") + if count() != 1 { + t.Fatalf("a non-liveness type lost its 1-hour cooldown (sent=%d)", count()) + } + for _, et := range []string{"node_down", "node_recovered"} { + if operatorCooldownFor(et) != nodeLivenessDedupeWindow { + t.Fatalf("%s is not in the ruling's bypass", et) + } + } + if operatorCooldownFor("host_stale") != operatorCooldown { + t.Fatal("host_stale is not in the 2026-09-15 ruling and must keep the hour") + } +} diff --git a/hub/internal/store/store.go b/hub/internal/store/store.go index ef49ec6a..944d1109 100644 --- a/hub/internal/store/store.go +++ b/hub/internal/store/store.go @@ -3410,6 +3410,36 @@ type CapabilityStatus struct { Reason string `json:"reason,omitempty"` } +// LatestHostReportRow is one host's newest raw report body (R-523: the controller-supervisor checker +// parses its own stanza, so the store does not grow a per-stanza reader for every consumer). +type LatestHostReportRow struct { + HostID string + CustomerID string + ReportJSON string +} + +// GetLatestHostReports returns every host's newest report body. +func (s *Store) GetLatestHostReports() ([]LatestHostReportRow, error) { + rows, err := s.db.Query(` + SELECT hr.host_id, hr.customer_id, hr.report_json + FROM host_reports hr + JOIN (SELECT host_id, MAX(id) AS mx FROM host_reports GROUP BY host_id) latest + ON hr.id = latest.mx`) + if err != nil { + return nil, err + } + defer rows.Close() + var out []LatestHostReportRow + for rows.Next() { + var r LatestHostReportRow + if err := rows.Scan(&r.HostID, &r.CustomerID, &r.ReportJSON); err != nil { + return nil, err + } + out = append(out, r) + } + return out, rows.Err() +} + // HostCapabilityRow is the per-host capability snapshot the HostCapabilityChecker reads — extracted // from the latest host-report's report_json (no dedicated column; the array rides the report body). type HostCapabilityRow struct { diff --git a/hub/internal/web/configs.go b/hub/internal/web/configs.go index 7b1d4294..bf1b107e 100644 --- a/hub/internal/web/configs.go +++ b/hub/internal/web/configs.go @@ -357,6 +357,11 @@ func (s *Server) handleCustomerUnified(w http.ResponseWriter, r *http.Request, c // nil when no code has been issued yet (pre-arc / never-pulled customer). Claim *store.ClaimState + // SelfBindSentAt / SelfBindSentOccasion (R-509, v0.114.0): when and why the connect link last + // went out (auto on creation / RESET / e-mail set / host delete, or the button). Empty = never. + SelfBindSentAt string + SelfBindSentOccasion string + // StaleSinceReset (v0.67.0, R-37): a RESET completed AFTER the newest report, so every // health number on this page describes a lifecycle that no longer exists. Without this the // page keeps showing pre-RESET warnings as if they were current. @@ -518,6 +523,14 @@ func (s *Server) handleCustomerUnified(w http.ResponseWriter, r *http.Request, c if cs, err := s.store.GetClaim(customerID); err == nil { data.Claim = cs } + if ev, err := s.store.GetLatestEventByType(customerID, selfBindSentEvent); err == nil && ev != nil { + data.SelfBindSentAt = ev.CreatedAt.UTC().Format("2006-01-02 15:04 UTC") + var d struct { + Occasion string `json:"occasion"` + } + _ = json.Unmarshal([]byte(ev.DetailsJSON), &d) + data.SelfBindSentOccasion = d.Occasion + } w.Header().Set("Content-Type", "text/html; charset=utf-8") if err := s.templates.ExecuteTemplate(w, "customer_unified.html", data); err != nil { @@ -752,6 +765,7 @@ func (s *Server) handleConfigUpdate(w http.ResponseWriter, r *http.Request, cust return } + prevEmail := cfg.Email cfg.CustomerName = strings.TrimSpace(r.FormValue("customer_name")) cfg.Domain = strings.TrimSpace(r.FormValue("domain")) cfg.Email = strings.TrimSpace(r.FormValue("email")) @@ -793,6 +807,10 @@ func (s *Server) handleConfigUpdate(w http.ResponseWriter, r *http.Request, cust s.logger.Printf("[INFO] Customer config updated: %s", customerID) s.bumpIntent(customerID) // Direction-2: wake a long-polling box in seconds + // R-509: an address set or changed on a customer with no box → the connect link goes out now. + if cfg.Email != "" && !strings.EqualFold(cfg.Email, prevEmail) { + s.autoMintSelfBindIfWaiting(customerID, "e-mail set on a customer with no box") + } http.Redirect(w, r, "/customers/"+customerID+"?flash=updated#tab=edit", http.StatusSeeOther) } diff --git a/hub/internal/web/hosts.go b/hub/internal/web/hosts.go index 596302c2..b2503863 100644 --- a/hub/internal/web/hosts.go +++ b/hub/internal/web/hosts.go @@ -903,6 +903,10 @@ func (s *Server) handleHostDelete(w http.ResponseWriter, r *http.Request, hostID return } s.logger.Printf("[INFO] host deleted: %s (escrow deleted: %v)", hostID, deleteEscrow) + // R-509: the customer record stays and now waits for a box → send the connect link. + if host.CustomerID != "" { + s.autoMintSelfBindIfWaiting(host.CustomerID, "host delete") + } http.Redirect(w, r, "/hosts", http.StatusSeeOther) } diff --git a/hub/internal/web/pbsdr.go b/hub/internal/web/pbsdr.go index 9e30f39e..a2a6cf2c 100644 --- a/hub/internal/web/pbsdr.go +++ b/hub/internal/web/pbsdr.go @@ -407,14 +407,29 @@ func (s *Server) handlePBSDRReissue(w http.ResponseWriter, r *http.Request, cust return } cur := readPBSDR(host.DesiredJSON) - if cur == nil || cur.Namespace == "" { - http.Error(w, "No provisioned PBS DR tier for this customer", http.StatusBadRequest) - return - } - // Same detached-ctx discipline as applyPBSDR: reissue→store→bump is the atom. ctx, cancel := context.WithTimeout(context.WithoutCancel(r.Context()), 2*time.Minute) defer cancel() + if cur == nil || cur.Namespace == "" { + // R-511 (v0.114.0) — ADOPT. The endpoint can hold this customer's token while the hub has no + // descriptor (a host delete keeps tenancy; the rebuilt box's WG hook then refuses and its error + // sends the operator HERE). This used to 400 „No provisioned PBS DR tier" — the one button the + // message advised refused. Now the explicit operator act re-keys the endpoint's token and builds + // the descriptor from what the endpoint returns. Gated on the customer's DR-tier flag: without + // the operator's intent on record, the endpoint is not touched. + if !cfg.DRTier { + http.Error(w, "No provisioned PBS DR tier for this customer, and the DR tier is not enabled on the customer — enable it first", http.StatusBadRequest) + return + } + if err := s.pbsdrAdopt(ctx, customerID, host); err != nil { + s.logger.Printf("[ERROR] pbsdr adopt for %s: %v", customerID, err) + http.Error(w, "PBS credential adopt failed: "+err.Error(), http.StatusBadGateway) + return + } + s.poke.PokeHost(host.HostID) + http.Redirect(w, r, "/customers/"+customerID+"?flash=pbsdr_reissued#tab=edit", http.StatusSeeOther) + return + } res, err := s.tenantsync.Reissue(ctx, customerID) if err != nil { s.logger.Printf("[ERROR] pbsdr reissue for %s: %v", customerID, err) @@ -446,6 +461,54 @@ func (s *Server) handlePBSDRReissue(w http.ResponseWriter, r *http.Request, cust http.Redirect(w, r, "/customers/"+customerID+"?flash=pbsdr_reissued#tab=edit", http.StatusSeeOther) } +// pbsdrAdopt (R-511) re-keys the customer's EXISTING endpoint token and writes a fresh descriptor — +// the explicit-operator twin of the F-14 auto re-issue inside pbsdrProvisionAtom, for the shape that +// path refuses (a token on the endpoint, no descriptor, no acknowledged deletion). Preconditions are +// the provisioning atom's: the host's WG peer and the endpoint record must exist. +func (s *Server) pbsdrAdopt(ctx context.Context, customerID string, host *store.Host) error { + if _, err := s.store.GetWGPeerForHost(host.HostID); err != nil { + return fmt.Errorf("host %s has no WG peer yet — the tunnel must exist before the PBS DR tier: %w", host.HostID, err) + } + ep, err := s.store.GetWGEndpoint() + if err != nil { + return fmt.Errorf("wg endpoint record unavailable: %w", err) + } + res, err := s.tenantsync.Reissue(ctx, customerID) + if err != nil { + return fmt.Errorf("endpoint re-issue: %w", err) + } + secretGen, err := s.store.SaveHostPBSSecret(host.HostID, res.TokenSecret) + if err != nil { + return fmt.Errorf("re-issued on the endpoint but storing the secret failed — press re-issue again: %w", err) + } + desc := &pbsDRDescriptor{ + Enabled: true, + StorageID: defaultPBSStorageID, + PBSTunnelIP: ep.PBSTunnelIP, + Datastore: res.Datastore, + Namespace: res.Namespace, + TokenID: res.TokenID, + Fingerprint: res.Fingerprint, + SecretGeneration: secretGen, + } + merged, err := mergePBSDR(host.DesiredJSON, desc) + if err != nil { + return err + } + gen, err := s.store.SetHostDesired(host.HostID, []byte(merged)) + if err != nil { + return fmt.Errorf("descriptor write: %w", err) + } + details, _ := json.Marshal(map[string]string{"host_id": host.HostID, "token_id": res.TokenID, "namespace": res.Namespace}) + if _, eerr := s.store.SaveEvent(customerID, "pbsdr_adopted", "info", + "PBS DR token adopted by operator re-issue (endpoint held a token, hub had no descriptor)", string(details), "hub"); eerr != nil { + s.logger.Printf("[WARN] pbsdr adopt audit event for %s not stored: %v", customerID, eerr) + } + s.logger.Printf("[INFO] pbsdr ADOPTED for %s (host %s, ns %s, token_id %s, gen %d; fresh consume-once secret stored, withheld from logs)", + customerID, host.HostID, res.Namespace, res.TokenID, gen) + return nil +} + // pbsDRView is the config form's render model for the DR-tier section, including the v0.51.0 // cascade stages (host → WG peer → descriptor → escrow) — one flag, ordered rollout, honest // intermediate states (scenario D). diff --git a/hub/internal/web/pbsdr_test.go b/hub/internal/web/pbsdr_test.go index 6dfbe475..d8a87f18 100644 --- a/hub/internal/web/pbsdr_test.go +++ b/hub/internal/web/pbsdr_test.go @@ -530,6 +530,8 @@ func TestPBSDR_Reissue(t *testing.T) { } } +// R-511 (v0.114.0): without a descriptor AND without the DR-tier flag, the endpoint is still never +// touched — the operator's intent must be on record before re-issue adopts anything. func TestPBSDR_ReissueRequiresProvisionedState(t *testing.T) { fake := &fakeTenancy{secret: "S"} s, _, _ := newPBSDRServer(t, fake) @@ -537,10 +539,61 @@ func TestPBSDR_ReissueRequiresProvisionedState(t *testing.T) { rr := httptest.NewRecorder() s.handlePBSDRReissue(rr, req, "peti") if rr.Code != 400 { - t.Fatalf("reissue without a provisioned tier = %d, want 400", rr.Code) + t.Fatalf("reissue without a provisioned tier or DR flag = %d, want 400", rr.Code) } if fake.reissueCalls != 0 { - t.Errorf("endpoint touched without a descriptor (%d calls)", fake.reissueCalls) + t.Errorf("endpoint touched without a descriptor or DR intent (%d calls)", fake.reissueCalls) + } +} + +// R-511 — tester-1's stuck shape: DR tier ON, the endpoint holds a token (Provision → ErrTokenExists), +// no descriptor, no acknowledged deletion. The WG hook refuses and sends the operator to re-issue; +// re-issue must now END WITH A DESCRIPTOR THE NEXT BOX CAN USE. +// +// RED-PROOF (run 2026-09-15, recorded in REPORT.md): with the adopt branch replaced by the old +// `400 No provisioned PBS DR tier`, this failed at "stuck-state re-issue = 400". +func TestPBSDR_ReissueAdoptsEndpointTokenWhenDescriptorMissing(t *testing.T) { + fake := &fakeTenancy{secret: "ADOPT-SECRET", provisionErr: tenantsync.ErrTokenExists} + s, st, logBuf := newPBSDRServer(t, fake) + cfg, _ := st.GetCustomerConfig("peti") + cfg.DRTier = true + if err := st.SaveCustomerConfig(cfg); err != nil { + t.Fatal(err) + } + // The stuck state: the form save's provision hits token_exists and refuses. + if rr := postUpdate(t, s, url.Values{"dr_tier": {"on"}}); rr.Code == 303 { + t.Fatalf("control: the stuck shape should refuse the save, got 303") + } + if d, _, _ := hostState(t, st); d != nil { + t.Fatalf("control: no descriptor expected before adopt, got %+v", d) + } + + req := httptest.NewRequest("POST", "/configs/peti/pbsdr-reissue", nil) + rr := httptest.NewRecorder() + s.handlePBSDRReissue(rr, req, "peti") + if rr.Code != 303 { + t.Fatalf("stuck-state re-issue = %d (%s), want 303 — the button the error message advises must work", rr.Code, rr.Body.String()) + } + d, _, _ := hostState(t, st) + if d == nil || !d.Enabled || d.Namespace != "peti" || d.TokenID != "felhom@pbs!peti" || d.PBSTunnelIP != "10.77.0.1" || d.StorageID != defaultPBSStorageID || d.SecretGeneration == 0 { + t.Fatalf("adopt did not leave a usable descriptor: %+v", d) + } + if got, err := st.ConsumeHostPBSSecret("peti-01"); err != nil || got != "ADOPT-SECRET" { + t.Fatalf("consume-once secret not stored: (%q, %v)", got, err) + } + if strings.Contains(logBuf.String(), "ADOPT-SECRET") { + t.Error("secret leaked into the hub log") + } + if ev, _ := st.GetLatestEventByType("peti", "pbsdr_adopted"); ev == nil { + t.Error("no pbsdr_adopted audit event") + } + // A save after adopt is now the already-provisioned no-op. + before := fake.reissueCalls + if rr := postUpdate(t, s, url.Values{"dr_tier": {"on"}}); rr.Code != 303 { + t.Fatalf("save after adopt = %d, want 303", rr.Code) + } + if fake.reissueCalls != before { + t.Error("a save after adopt re-keyed again") } } diff --git a/hub/internal/web/selfbind_mint.go b/hub/internal/web/selfbind_mint.go index eab388e0..05cac4d6 100644 --- a/hub/internal/web/selfbind_mint.go +++ b/hub/internal/web/selfbind_mint.go @@ -3,6 +3,7 @@ package web import ( "crypto/sha256" "encoding/hex" + "encoding/json" "net/http" "time" @@ -118,6 +119,7 @@ func (s *Server) autoMintSelfBindLink(customerID, email, occasion string) { switch outcome { case selfBindSent: s.logger.Printf("[INFO] self-bind link auto-minted for %s on %s (the console banner's promised email now exists)", customerID, occasion) + s.recordSelfBindSent(customerID, occasion) case selfBindSkippedNoMailer: s.logger.Printf("[INFO] self-bind link NOT auto-minted for %s on %s: no mailer configured on this hub", customerID, occasion) case selfBindSkippedNoEmail: @@ -129,6 +131,36 @@ func (s *Server) autoMintSelfBindLink(customerID, email, occasion string) { } } +// selfBindSentEvent (R-509, v0.114.0) records WHEN and WHY a self-bind link went out, so the customer +// page can say „Kapcsolódó link elküldve: ()". Hub-internal: stored, never dispatched. +const selfBindSentEvent = "selfbind_link_sent" + +func (s *Server) recordSelfBindSent(customerID, occasion string) { + details, _ := json.Marshal(map[string]string{"occasion": occasion}) + if _, err := s.store.SaveEvent(customerID, selfBindSentEvent, "info", "Self-bind link e-mailed ("+occasion+")", string(details), "hub"); err != nil { + s.logger.Printf("[WARN] self-bind: sent-record for %s not stored: %v", customerID, err) + } +} + +// autoMintSelfBindIfWaiting (R-509, operator decision A 2026-09-15) sends the link whenever an event +// leaves a customer WAITING FOR A BOX: an e-mail set or changed on a customer with no bound host, and a +// host delete that keeps the customer. Appliance registration is deliberately NOT a trigger — it knows +// no customer (api/appliance.go). No host bound is re-checked here, so a customer who still has a box +// is never mailed a pairing link. +func (s *Server) autoMintSelfBindIfWaiting(customerID, occasion string) { + cfg, err := s.store.GetCustomerConfig(customerID) + if err != nil || cfg == nil { + return + } + if h, herr := s.store.GetHostByCustomer(customerID); herr != nil || h != nil { + if herr != nil { + s.logger.Printf("[WARN] self-bind auto-send for %s on %s skipped: host lookup failed: %v", customerID, occasion, herr) + } + return + } + s.autoMintSelfBindLink(customerID, cfg.Email, occasion) +} + // handleSelfBindLinkSend — POST /customers/{id}/selfbind-link. Mints a single-active capability token // for the customer and emails the public bind link. Honesty rules: // - F1: no registered email → nothing is minted, LOUD flash (a link no one can receive is useless). @@ -158,6 +190,7 @@ func (s *Server) handleSelfBindLinkSend(w http.ResponseWriter, r *http.Request, s.logger.Printf("[ERROR] self-bind link for %s: %v", customerID, err) http.Error(w, "Internal error", http.StatusInternalServerError) default: // selfBindSent (the no-mailer case was refused above) + s.recordSelfBindSent(customerID, "operator button") http.Redirect(w, r, "/customers/"+customerID+"?flash=selfbind-sent#tab=setup", http.StatusSeeOther) } } diff --git a/hub/internal/web/selfbind_waiting_test.go b/hub/internal/web/selfbind_waiting_test.go new file mode 100644 index 00000000..07946119 --- /dev/null +++ b/hub/internal/web/selfbind_waiting_test.go @@ -0,0 +1,90 @@ +package web + +import ( + "net/http/httptest" + "net/url" + "strings" + "testing" + + "gitea.dooplex.hu/admin/felhom-hub/internal/store" +) + +// R-509 (operator decision A, 2026-09-15) — the connect e-mail goes out by itself whenever a customer +// is left WAITING FOR A BOX. BIGNIGHT: tester-1 had its e-mail added after creation and its previous +// host deleted; the box's console promised a mail; the mailbox held 0 messages ten minutes later. + +func updateEmail(t *testing.T, s *Server, id, email string) { + t.Helper() + form := url.Values{"customer_name": {id}, "domain": {id + ".hu"}, "email": {email}} + req := httptest.NewRequest("POST", "/configs/"+id, strings.NewReader(form.Encode())) + req.Header.Set("Content-Type", "application/x-www-form-urlencoded") + rr := httptest.NewRecorder() + s.handleConfigUpdate(rr, req, id) + if rr.Code != 303 { + t.Fatalf("config update = %d (%s)", rr.Code, rr.Body.String()) + } +} + +// RED-PROOF (run 2026-09-15, recorded in REPORT.md): with the autoMintSelfBindIfWaiting call removed +// from handleConfigUpdate, no link went out and this failed at "e-mail set on a customer with no box +// sent no link". +func TestSelfBind_EmailSetOnWaitingCustomerSendsLink(t *testing.T) { + s, st := newTestServer(t) + m := &stubMailer{} + s.SetSelfBindMailer(m) + seedForMint(t, st, "tester", "") + + updateEmail(t, s, "tester", "tester1@felhom.example") + if m.link == "" { + t.Fatal("e-mail set on a customer with no box sent no link") + } + ev, _ := st.GetLatestEventByType("tester", selfBindSentEvent) + if ev == nil || !strings.Contains(ev.DetailsJSON, "no box") { + t.Fatalf("the send was not recorded with its occasion: %+v", ev) + } + + // The same address saved again → no second mail. + m.link = "" + updateEmail(t, s, "tester", "tester1@felhom.example") + if m.link != "" { + t.Fatal("an unchanged e-mail re-sent the link on every save") + } +} + +func TestSelfBind_CustomerWithBoxGetsNoLink(t *testing.T) { + s, st := newTestServer(t) + m := &stubMailer{} + s.SetSelfBindMailer(m) + seedForMint(t, st, "hasbox", "") + if err := st.UpsertHost(&store.Host{HostID: "hasbox-01", CustomerID: "hasbox", APIKey: "h"}); err != nil { + t.Fatal(err) + } + updateEmail(t, s, "hasbox", "owner@hasbox.example") + if m.link != "" { + t.Fatal("a customer that has a box was mailed a pairing link") + } +} + +func TestSelfBind_HostDeleteSendsLink(t *testing.T) { + s, st := newTestServer(t) + m := &stubMailer{} + s.SetSelfBindMailer(m) + seedForMint(t, st, "rebuilt", "owner@rebuilt.example") + if err := st.UpsertHost(&store.Host{HostID: "rebuilt-01", CustomerID: "rebuilt", APIKey: "h"}); err != nil { + t.Fatal(err) + } + form := url.Values{"confirm_host_id": {"rebuilt-01"}} + req := httptest.NewRequest("POST", "/hosts/rebuilt-01/delete", strings.NewReader(form.Encode())) + req.Header.Set("Content-Type", "application/x-www-form-urlencoded") + rr := httptest.NewRecorder() + s.handleHostDelete(rr, req, "rebuilt-01") + if rr.Code != 303 { + t.Fatalf("host delete = %d (%s)", rr.Code, rr.Body.String()) + } + if m.link == "" { + t.Fatal("host delete kept the customer waiting for a box and sent no link") + } + if ev, _ := st.GetLatestEventByType("rebuilt", selfBindSentEvent); ev == nil || !strings.Contains(ev.DetailsJSON, "host delete") { + t.Fatalf("host-delete send not recorded: %+v", ev) + } +} diff --git a/hub/internal/web/templates/customer_unified.html b/hub/internal/web/templates/customer_unified.html index 34d12c8b..7be1f7dc 100644 --- a/hub/internal/web/templates/customer_unified.html +++ b/hub/internal/web/templates/customer_unified.html @@ -519,6 +519,7 @@ {{.CSRFField}} + {{if .SelfBindSentAt}}Kapcsolódó link elküldve: {{.SelfBindSentAt}} ({{.SelfBindSentOccasion}}){{else}}Kapcsolódó link még nem ment ki.{{end}} It is sent automatically when the customer is created or RESET, when an e-mail is set on a customer with no box, and when this customer's host is deleted.