hub v0.126.0: fresh connect link from the old one (R-719); day-one mails (R-723); bind-page wording (R-725); volunteer guide current (R-722); ep0 cleanup and release evidence
gates / gates (push) Successful in 29s

Red-proofs RP40-RP42.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-09-30 10:35:17 +02:00
parent 27641ce9e7
commit 80aeac71f6
21 changed files with 562 additions and 40 deletions
+58
View File
@@ -0,0 +1,58 @@
package monitor
import (
"database/sql"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
)
func hostFor(t *testing.T, st *store.Store, path string, createdAgo time.Duration) {
t.Helper()
if err := st.UpsertHost(&store.Host{HostID: "c1-abc123", CustomerID: "c1", APIKey: "hk"}); err != nil {
t.Fatal(err)
}
db, err := sql.Open("sqlite", path)
if err != nil {
t.Fatal(err)
}
defer db.Close()
if _, err := db.Exec(`UPDATE hosts SET created_at = ? WHERE host_id = 'c1-abc123'`,
time.Now().UTC().Add(-createdAgo).Format("2006-01-02 15:04:05")); err != nil {
t.Fatal(err)
}
}
// R-723 (v0.126.0) — the 2026-09-29 shape: the customer's previous box went silent 12 days ago (the hub
// seeds the customer as down), a NEW box enrolls and sends its first report. The CONSEQUENCE asserted: no
// `node_recovered` (it is the operator mail „recovered" for an outage the new box never had).
// COMPANION RED-PROOF: make isNewBox return false → the event list is [node_recovered].
func TestR723_ANewBoxIsNotARecovery(t *testing.T) {
st, path := seedStalenessCustomer(t, "ok", 12*24*time.Hour)
sc, events := newChecker(t, st) // seeds c1 as down (the old box's silence)
if sc.GetState("c1") != "down" {
t.Fatalf("setup: want down, got %q", sc.GetState("c1"))
}
hostFor(t, st, path, 2*time.Minute) // the new box enrolled 2 min ago
saveReportAged(t, st, path, "ok", 0) // …and reported
sc.Check()
if len(*events) != 0 {
t.Fatalf("a new box's first report emitted %v", *events)
}
if sc.GetState("c1") != "ok" {
t.Fatalf("state %q, want ok", sc.GetState("c1"))
}
}
// Control: the SAME box, enrolled long ago, coming back after an outage IS a recovery and still mails.
func TestR723_AnOldBoxComingBackIsStillARecovery(t *testing.T) {
st, path := seedStalenessCustomer(t, "ok", 12*24*time.Hour)
sc, events := newChecker(t, st)
hostFor(t, st, path, 40*24*time.Hour)
saveReportAged(t, st, path, "ok", 0)
sc.Check()
if len(*events) != 1 || (*events)[0] != "node_recovered" {
t.Fatalf("a real recovery must still emit node_recovered, got %v", *events)
}
}
+25
View File
@@ -171,6 +171,19 @@ func (sc *StalenessChecker) Check() {
continue
}
// R-723 (v0.126.0): a customer's NEW box is not a recovery. Staleness is tracked per CUSTOMER, so a
// customer whose previous box went silent reads stale/down, and the new box's first report used to
// send `node_recovered` for an outage the new box never had (measured 2026-09-29: tester-1, silent 12
// days, new box enrolled 19:20:07Z, operator mail „recovered" 2 s after its first report). The
// discriminator is the host record: enrolled AFTER the outage began (or, when the hub restarted during
// the outage and has no start time, within the stale threshold) → a first observation, no event.
if newState == "ok" && (oldState == "stale" || oldState == "down") && sc.isNewBox(c.CustomerID) {
sc.logger.Printf("[INFO] Staleness: %s → ok on the first reports of a NEW box — not a recovery, no event", c.CustomerID)
sc.states[c.CustomerID] = newState
delete(sc.downtimeStart, c.CustomerID)
continue
}
// State transition — emit event
sc.states[c.CustomerID] = newState
if newState == "stale" && oldState == "ok" {
@@ -250,3 +263,15 @@ func formatDuration(d time.Duration) string {
}
return fmt.Sprintf("%dh%dm", h, m)
}
// isNewBox reports whether the customer's current host was enrolled after its outage began (R-723).
func (sc *StalenessChecker) isNewBox(customerID string) bool {
h, err := sc.store.GetHostByCustomer(customerID)
if err != nil || h == nil || h.CreatedAt.IsZero() {
return false // no evidence of a new box → keep the old behaviour (a real recovery mails)
}
if since, ok := sc.downtimeStart[customerID]; ok {
return h.CreatedAt.After(since)
}
return time.Since(h.CreatedAt) < sc.threshold
}