controller (unreleased): R-921 check-first pre-check (no stop while another tier's job is in flight); R-922 email_cleared on a household's deliberate clear
gates / gates (push) Successful in 1m0s

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-09 12:57:34 +02:00
parent 13eeb484ab
commit ec9b997151
16 changed files with 798 additions and 18 deletions
@@ -127,6 +127,33 @@ func (c *Client) BackupStatusFor(ctx context.Context, target string) (StatusResp
return out, nil
}
// BackupJobsInFlight (R-921) lists the tiers whose backup job the agent holds in flight right now — phase
// running or snapshotted, the agent's own definition (localapi backupInFlight). A job in flight on another
// tier is the agent's first reason to refuse POST /backup (HTTP 409), so the controller asks this BEFORE it
// stops any app. It reads only GET /backup/tiers and GET /backup/status?target= (both agent >= v0.97.0); a
// pre-R-82 agent (no /backup/tiers) answers nil, nil. Any other error is returned — the caller fails toward
// backing up. It does NOT see the agent's host-wide heavy-operation gate: no endpoint serves it.
func (c *Client) BackupJobsInFlight(ctx context.Context) ([]string, error) {
tiers, err := c.BackupTiers(ctx)
if errors.Is(err, ErrTiersUnsupported) {
return nil, nil
}
if err != nil {
return nil, err
}
var out []string
for _, t := range tiers.Tiers {
st, err := c.BackupStatusFor(ctx, t.Target)
if err != nil {
return nil, err
}
if st.Phase == PhaseRunning || st.Phase == PhaseSnapshotted {
out = append(out, t.Target)
}
}
return out, nil
}
// SetBackupTargetResponse mirrors POST /backup/target (agent >= v0.113.0).
type SetBackupTargetResponse struct {
Target string `json:"target"`
+3
View File
@@ -255,6 +255,9 @@ const (
PhaseRunning = "running"
PhaseDone = "done"
PhaseFailed = "failed"
// PhaseSnapshotted (8B.2): the storage snapshot is taken, the upload still runs and still holds the
// guest — the agent counts it as IN FLIGHT (felhom-agent localapi backupInFlight).
PhaseSnapshotted = "snapshotted"
)
// BackupDue reports whether a policy-scheduled backup is due for this guest (the quiesce trigger).
@@ -0,0 +1,75 @@
package agentapi
import (
"context"
"net/http"
"net/http/httptest"
"reflect"
"strings"
"testing"
)
// R-921: the pre-check reads the agent's real wire shapes — GET /backup/tiers, then GET /backup/status per
// target — and reports exactly the tiers whose job is running or snapshotted (the agent's in-flight set).
func TestBackupJobsInFlight_ReportsRunningAndSnapshottedTiers(t *testing.T) {
phases := map[string]string{"local": "done", "felhom-pbs": "snapshotted", "extra": "running"}
mux := http.NewServeMux()
mux.HandleFunc("GET /backup/tiers", func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{"ok":true,"data":{"vmid":9201,"tiers":[` +
`{"target":"local","cadence_seconds":86400,"primary":true,"storage":"present"},` +
`{"target":"felhom-pbs","cadence_seconds":604800,"primary":false,"storage":"present"},` +
`{"target":"extra","cadence_seconds":604800,"primary":false,"storage":"present"}]}}`))
})
mux.HandleFunc("GET /backup/status", func(w http.ResponseWriter, r *http.Request) {
tg := r.URL.Query().Get("target")
ph, ok := phases[tg]
if !ok {
http.Error(w, "unexpected target "+tg, http.StatusBadRequest)
return
}
_, _ = w.Write([]byte(`{"ok":true,"data":{"vmid":9201,"phase":"` + ph + `","target":"` + tg + `"}}`))
})
s := httptest.NewTLSServer(mux)
defer s.Close()
c := clientFor(t, s, strings.TrimPrefix(s.URL, "https://"))
got, err := c.BackupJobsInFlight(context.Background())
if err != nil {
t.Fatalf("BackupJobsInFlight: %v", err)
}
if want := []string{"felhom-pbs", "extra"}; !reflect.DeepEqual(got, want) {
t.Fatalf("in-flight tiers = %v, want %v (done must not count; running and snapshotted must)", got, want)
}
}
// A pre-R-82 agent (no /backup/tiers → 404) has no per-tier jobs: nil, nil — never an error that would be
// read as "cannot tell", and never a busy verdict.
func TestBackupJobsInFlight_PreR82AgentIsNotBusy(t *testing.T) {
mux := http.NewServeMux()
s := httptest.NewTLSServer(mux) // every route 404s
defer s.Close()
c := clientFor(t, s, strings.TrimPrefix(s.URL, "https://"))
got, err := c.BackupJobsInFlight(context.Background())
if err != nil || got != nil {
t.Fatalf("pre-R-82 agent: got %v, %v — want nil, nil", got, err)
}
}
// A status read that fails is returned, so the loop can fail toward backing up and say so.
func TestBackupJobsInFlight_StatusErrorIsReturned(t *testing.T) {
mux := http.NewServeMux()
mux.HandleFunc("GET /backup/tiers", func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{"ok":true,"data":{"vmid":9201,"tiers":[{"target":"local","primary":true}]}}`))
})
mux.HandleFunc("GET /backup/status", func(w http.ResponseWriter, r *http.Request) {
http.Error(w, "boom", http.StatusInternalServerError)
})
s := httptest.NewTLSServer(mux)
defer s.Close()
c := clientFor(t, s, strings.TrimPrefix(s.URL, "https://"))
if _, err := c.BackupJobsInFlight(context.Background()); err == nil {
t.Fatal("a failed status read was swallowed — the caller could not tell it from 'nothing in flight'")
}
}
+8 -2
View File
@@ -1029,11 +1029,16 @@ type preferencesRequest struct {
Email string `json:"email"`
EnabledEvents []string `json:"enabled_events"`
CooldownHours int `json:"cooldown_hours,omitempty"`
// EmailCleared (R-922): the household deliberately cleared its address — the hub deletes the one it
// holds. OMITTED unless true: an empty address without it keeps the hub's no-clobber guard (an
// unconfigured box must never wipe a seeded address). Decoded by the hub's savePreferences.
EmailCleared bool `json:"email_cleared,omitempty"`
}
// SyncPreferences pushes the current notification preferences to the hub.
// Synchronous — returns error for the handler to display to the user.
func (n *Notifier) SyncPreferences(email string, enabledEvents []string, cooldownHours int) error {
// emailCleared (R-922) is settings.NotificationPrefs.EmailCleared — true only after a household's clear.
func (n *Notifier) SyncPreferences(email string, enabledEvents []string, cooldownHours int, emailCleared bool) error {
if !n.enabled {
return util.MsgError("err.notify.hub_nem_konfiguralt")
}
@@ -1043,6 +1048,7 @@ func (n *Notifier) SyncPreferences(email string, enabledEvents []string, cooldow
Email: email,
EnabledEvents: enabledEvents,
CooldownHours: cooldownHours,
EmailCleared: emailCleared,
}
jsonData, err := json.Marshal(payload)
@@ -1076,7 +1082,7 @@ func (n *Notifier) SyncPreferences(email string, enabledEvents []string, cooldow
if n.debug {
n.logger.Printf("[DEBUG] SyncPreferences: response HTTP %d", resp.StatusCode)
}
n.logger.Printf("[INFO] Notification preferences synced to hub: email=%s, events=%v, cooldown=%dh", email, enabledEvents, cooldownHours)
n.logger.Printf("[INFO] Notification preferences synced to hub: email=%s, events=%v, cooldown=%dh, email_cleared=%t", email, enabledEvents, cooldownHours, emailCleared)
return nil
}
+58
View File
@@ -0,0 +1,58 @@
package quiesce
import "context"
// R-921 — CHECK FIRST, STOP SECOND.
//
// MEASURED 2026-10-08 night on demo-hp (felhom.eu documentation/audits/dooplex-survival-2026-10-09/partE/
// R-518.txt): the controller stopped every app for the off-site tier, the agent refused the backup
// („a heavy operation is already in flight", busy=backup:local — the night OS step that follows the local
// copy), and the apps were down about a minute for no copy.
//
// What the agent lets the controller see BEFORE a stop (felhom-agent v0.154.0, read 2026-10-09): the agent
// refuses POST /backup for two reasons. (1) A job of ANOTHER tier of this guest is in flight
// (localapi otherTierInFlight) — that IS readable, per tier, from GET /backup/status?target=…. (2) Its
// host-wide heavy-operation gate is held (backup.InFlight: the OS step after the night's local copy, a
// restore-test, fstrim) — that is served by NO endpoint (InFlight.Busy() exists and nothing exposes it).
// So the pre-check below covers reason (1) only. Reason (2) — the measured instance — is met by the
// BUSY path in quiesceAndPollTiers, which resumes the apps at once (pinned by
// TestR921_BusyRefusalResumesAtOnce); closing it before the stop needs the agent to serve its gate.
//
// The race (free at the check, busy at the start) needs nothing new: it is the BUSY path again.
// InFlightProber is the OPTIONAL pre-check surface (R-921). An adapter that does not implement it keeps
// the pre-R-921 behaviour exactly: stop, ask, and on refusal resume at once.
type InFlightProber interface {
// BackupJobsInFlight returns the tiers whose backup job the agent holds in flight right now
// (phase running or snapshotted). An agent without per-tier jobs (pre-R-82) answers nil, nil.
BackupJobsInFlight(ctx context.Context) ([]string, error)
}
// anotherTierInFlight reports whether the agent would refuse the window's first tier because a job of a
// DIFFERENT tier is in flight — so the window must not stop any app. The window's OWN tier in flight is
// not a refusal (the agent answers its start with the running job, 202), and that path is left as it was.
//
// Fail toward backing up: an unanswerable pre-check, an untargeted window (pre-R-82 agent) or an adapter
// without the surface all return false — the window runs as before.
func (l *Loop) anotherTierInFlight(ctx context.Context, first string) bool {
if first == "" {
return false
}
p, ok := l.backend.(InFlightProber)
if !ok {
return false
}
busy, err := p.BackupJobsInFlight(ctx)
if err != nil {
l.logger.Printf("[WARN] [quiesce] pre-check: could not ask the agent which backup jobs are in flight (%v) — stopping the apps and asking as before (R-921)", err)
return false
}
for _, t := range busy {
if t != first {
l.logger.Printf("[INFO] [quiesce] tier %s is due, but the agent still holds a backup job on tier %s — no app is stopped; the tier stays due and is asked again at the next poll (R-921)",
tierLabel(first), tierLabel(t))
return true
}
}
return false
}
+5
View File
@@ -556,6 +556,11 @@ func (l *Loop) quiesceAndPollTiers(ctx context.Context, tiers []dueTier) error {
if len(tiers) == 0 {
return nil
}
// R-921: check first, stop second — a tier the agent will refuse (another tier's job in flight) stops
// no app. Pinned by TestR921_NoStopWhileAnotherTiersJobIsInFlight (precheck.go says what it cannot see).
if l.anotherTierInFlight(ctx, tiers[0].target) {
return nil
}
running := l.stacks.RunningAppStacks()
marker := Marker{Active: true, StartedAt: l.now(), StoppedStacks: running}
if err := l.writeMarker(marker); err != nil {
@@ -0,0 +1,286 @@
package quiesce
import (
"context"
"errors"
"io"
"log"
"path/filepath"
"strings"
"sync"
"testing"
"time"
)
// R-921 (measured 2026-10-08 night on demo-hp): the controller stopped every app for the off-site tier,
// the agent refused the backup (`a heavy operation is already in flight`), and the apps were down about a
// minute for nothing. The rule this file pins: CHECK FIRST, STOP SECOND — and when the check could not
// see the refusal coming, RESUME AT ONCE.
//
// The agent's local API serves no read of its host-wide heavy-operation gate (felhom-agent
// internal/backup.InFlight.Busy() is served by no endpoint). What it DOES serve is each tier's job phase
// (GET /backup/status?target=…), and a job in flight on ANOTHER tier is the agent's first refusal reason
// (localapi handleBackup → otherTierInFlight → HTTP 409). So the pre-check covers that reason; the other
// (the gate held by the night OS step, a restore-test, fstrim) can only be met by the immediate resume.
// r921Backend is a tiered fake agent that records every call into a shared, ordered event log with the
// fake stacks, so a test can say "nothing happened between the refusal and the first restart".
type r921Backend struct {
mu sync.Mutex
events *[]string
tiers []BackupTier
due map[string]bool
inFlight []string // targets whose job the agent holds in flight (running|snapshotted)
probeErr error
busyOn map[string]bool // StartBackupFor answers ErrTierBusy for these targets
starts []string
}
func (b *r921Backend) log(e string) {
b.mu.Lock()
defer b.mu.Unlock()
*b.events = append(*b.events, e)
}
func (b *r921Backend) Tiers(context.Context) ([]BackupTier, error) { return b.tiers, nil }
func (b *r921Backend) DueFor(_ context.Context, target string) (bool, *int64, string, error) {
return b.due[target], nil, "", nil
}
func (b *r921Backend) StartBackupFor(_ context.Context, target string) (string, error) {
b.mu.Lock()
b.starts = append(b.starts, target)
b.mu.Unlock()
if b.busyOn[target] {
b.log("start:" + target + "=BUSY")
return "", busyErr()
}
b.log("start:" + target)
return "job-" + target, nil
}
func (b *r921Backend) BackupStatusFor(_ context.Context, target string) (string, error) {
b.log("status:" + target)
return phaseDone, nil
}
func (b *r921Backend) BackupJobsInFlight(context.Context) ([]string, error) {
b.log("probe")
if b.probeErr != nil {
return nil, b.probeErr
}
return append([]string(nil), b.inFlight...), nil
}
func (b *r921Backend) Due(context.Context) (bool, *int64, error) { return false, nil, nil }
func (b *r921Backend) StartBackup(context.Context) (string, error) { return "", errors.New("unused") }
func (b *r921Backend) BackupStatus(context.Context) (string, error) {
return phaseDone, nil
}
// r921Stacks logs stops and starts into the same event log.
type r921Stacks struct {
mu sync.Mutex
events *[]string
running []string
}
func (s *r921Stacks) RunningAppStacks() []string { return append([]string(nil), s.running...) }
func (s *r921Stacks) StopStack(n string) error {
s.mu.Lock()
defer s.mu.Unlock()
*s.events = append(*s.events, "stop:"+n)
return nil
}
func (s *r921Stacks) StartStack(n string) error {
s.mu.Lock()
defer s.mu.Unlock()
*s.events = append(*s.events, "startstack:"+n)
return nil
}
func r921Setup(t *testing.T) (*r921Backend, *r921Stacks, *Loop, *[]string, *strings.Builder) {
t.Helper()
var events []string
be := &r921Backend{
events: &events,
tiers: []BackupTier{{Target: "local", Primary: true}, {Target: "felhom-pbs"}},
due: map[string]bool{},
busyOn: map[string]bool{},
}
st := &r921Stacks{events: &events, running: []string{"immich", "paperless-ngx", "vaultwarden"}}
var logs strings.Builder
l := New(Options{
Backend: be,
Stacks: st,
MarkerPath: filepath.Join(t.TempDir(), "quiesce-state.json"),
StatusPoll: time.Millisecond,
MaxQuiesce: 30 * time.Second,
Logger: log.New(&logs, "", 0),
})
return be, st, l, &events, &logs
}
func countPrefix(events []string, prefix string) int {
n := 0
for _, e := range events {
if strings.HasPrefix(e, prefix) {
n++
}
}
return n
}
// RED TEST. Another tier's job is in flight on the agent (a first off-site upload that outlived the
// quiesce bound, or a controller restarted mid-upload) and the local tier is due. The agent WILL refuse
// the local backup (otherTierInFlight → 409). Today's code stops every app, asks, is refused, restarts
// them: a stop for nothing. Asserted on the CONSEQUENCE — no app was stopped and no backup was requested
// — not on the mechanism. It must also leave the tier due (no breaker, no contention verdict): the next
// poll asks again, which costs the household nothing.
func TestR921_NoStopWhileAnotherTiersJobIsInFlight(t *testing.T) {
be, _, l, events, logs := r921Setup(t)
be.due["local"] = true
be.inFlight = []string{"felhom-pbs"}
be.busyOn["local"] = true // what the real agent answers while another tier's job is in flight
if err := l.runOnce(context.Background()); err != nil {
t.Fatalf("runOnce: %v", err)
}
if n := countPrefix(*events, "stop:"); n != 0 {
t.Fatalf("R-921: %d app(s) stopped for a tier the agent refuses (another tier's job is in flight); events=%v", n, *events)
}
if len(be.starts) != 0 {
t.Fatalf("R-921: a backup was requested although another tier's job is in flight: %v", be.starts)
}
if got := l.breaker.failuresFor("local"); got != 0 {
t.Errorf("the pre-check armed the failure breaker (%d) — nothing failed", got)
}
if _, blocked := l.contention.blocked("local", l.now()); blocked {
t.Errorf("the pre-check set a contention verdict — the tier must simply stay due and be asked again next poll")
}
if !strings.Contains(logs.String(), "felhom-pbs") || !strings.Contains(logs.String(), "R-921") {
t.Errorf("the skip is not explained in the log (want the in-flight tier and R-921): %q", logs.String())
}
// The job ends → the next poll backs the local tier up as normal (the skip did not swallow the tier).
be.inFlight = nil
be.busyOn["local"] = false
if err := l.runOnce(context.Background()); err != nil {
t.Fatalf("runOnce 2: %v", err)
}
if len(be.starts) != 1 || be.starts[0] != "local" {
t.Fatalf("after the job ended the local tier was not backed up: starts=%v", be.starts)
}
}
// The measured instance and the race, both on the BUSY path: the pre-check cannot see the refusal
// coming (the night OS step holds the gate — not readable from the agent — or the agent became busy
// between the check and the start). Then the resume must follow the refusal AT ONCE: the very next
// event after the refused start is the first app's restart — no other agent call, no further tier, no
// wait in between. Measured on the ordered event log, not on a clock.
func TestR921_BusyRefusalResumesAtOnce(t *testing.T) {
for _, tc := range []struct {
name string
due []string
busy []string
probe bool // the pre-check answered "free" (the race) vs. a backend without the pre-check
}{
{"measured: only the off-site tier due, refused (no pre-check)", []string{"felhom-pbs"}, []string{"felhom-pbs"}, false},
{"race: pre-check said free, the start was refused", []string{"felhom-pbs"}, []string{"felhom-pbs"}, true},
} {
t.Run(tc.name, func(t *testing.T) {
be, st, l, events, _ := r921Setup(t)
for _, d := range tc.due {
be.due[d] = true
}
for _, b := range tc.busy {
be.busyOn[b] = true
}
var loop *Loop = l
if !tc.probe {
// A backend WITHOUT the pre-check surface (an adapter that does not implement it).
nb := &noProbeBackend{be}
loop = New(Options{Backend: nb, Stacks: st, MarkerPath: filepath.Join(t.TempDir(), "m.json"),
StatusPoll: time.Millisecond, MaxQuiesce: 30 * time.Second, Logger: log.New(io.Discard, "", 0)})
}
if err := loop.runOnce(context.Background()); err != nil {
t.Fatalf("runOnce: %v", err)
}
ev := *events
idx := -1
for i, e := range ev {
if strings.HasSuffix(e, "=BUSY") {
idx = i
}
}
if idx < 0 {
t.Fatalf("no refused start in the events: %v", ev)
}
if idx+1 >= len(ev) || !strings.HasPrefix(ev[idx+1], "startstack:") {
t.Fatalf("the resume did not follow the refusal at once; events after the refusal: %v", ev[idx+1:])
}
if got, want := countPrefix(ev[idx+1:], "startstack:"), len(st.running); got != want {
t.Fatalf("restarted %d app(s) after the refusal, want all %d; events=%v", got, want, ev)
}
for _, e := range ev[idx+1:] {
if !strings.HasPrefix(e, "startstack:") {
t.Fatalf("an agent call (%q) came after the refusal — the BUSY path must only resume; events=%v", e, ev)
}
}
if _, blocked := loop.contention.blocked("felhom-pbs", loop.now()); !blocked {
t.Errorf("the refused tier was not deferred — the next poll would stop the apps again")
}
})
}
}
// A pre-check that cannot be answered must not suppress a backup: fail toward backing up, exactly as
// before R-921 (stop, ask, and on refusal resume at once).
func TestR921_PrecheckErrorFailsTowardBackingUp(t *testing.T) {
be, _, l, events, logs := r921Setup(t)
be.due["local"] = true
be.probeErr = errors.New("agentapi: GET /backup/tiers: connection refused")
if err := l.runOnce(context.Background()); err != nil {
t.Fatalf("runOnce: %v", err)
}
if len(be.starts) != 1 || be.starts[0] != "local" {
t.Fatalf("an unanswerable pre-check suppressed the backup: starts=%v events=%v", be.starts, *events)
}
if !strings.Contains(logs.String(), "[WARN]") {
t.Errorf("the failed pre-check was not logged at WARN: %q", logs.String())
}
}
// The window's OWN tier already in flight is NOT skipped: the agent answers that start with the running
// job (202, idempotent), the loop attaches to it and records its outcome — unchanged behaviour, chosen
// deliberately (only a tier the agent REFUSES is skipped before the stop).
func TestR921_SameTierInFlightIsNotSkipped(t *testing.T) {
be, _, l, _, _ := r921Setup(t)
be.due["felhom-pbs"] = true
be.inFlight = []string{"felhom-pbs"}
if err := l.runOnce(context.Background()); err != nil {
t.Fatalf("runOnce: %v", err)
}
if len(be.starts) != 1 || be.starts[0] != "felhom-pbs" {
t.Fatalf("the window's own in-flight tier was skipped; starts=%v", be.starts)
}
}
// noProbeBackend hides BackupJobsInFlight (an adapter without the pre-check surface).
type noProbeBackend struct{ b *r921Backend }
func (n *noProbeBackend) Tiers(ctx context.Context) ([]BackupTier, error) { return n.b.Tiers(ctx) }
func (n *noProbeBackend) DueFor(ctx context.Context, t string) (bool, *int64, string, error) {
return n.b.DueFor(ctx, t)
}
func (n *noProbeBackend) StartBackupFor(ctx context.Context, t string) (string, error) {
return n.b.StartBackupFor(ctx, t)
}
func (n *noProbeBackend) BackupStatusFor(ctx context.Context, t string) (string, error) {
return n.b.BackupStatusFor(ctx, t)
}
func (n *noProbeBackend) Due(ctx context.Context) (bool, *int64, error) { return n.b.Due(ctx) }
func (n *noProbeBackend) StartBackup(ctx context.Context) (string, error) {
return n.b.StartBackup(ctx)
}
func (n *noProbeBackend) BackupStatus(ctx context.Context) (string, error) {
return n.b.BackupStatus(ctx)
}
@@ -0,0 +1,57 @@
package settings
import (
"io"
"log"
"path/filepath"
"testing"
)
// R-922: the clear marker's rule as a table — set only by a deliberate clear, kept while the address stays
// empty, dropped by a new address — and it survives a reload from disk (a clear the hub missed is pushed
// again by the next start: SyncOnStartup).
func TestR922_ClearedByRule(t *testing.T) {
for _, tc := range []struct {
name string
stored *NotificationPrefs
newEmail string
want bool
}{
{"never configured, empty save", &NotificationPrefs{}, "", false},
{"nil stored prefs", nil, "", false},
{"had an address, empty save = clear", &NotificationPrefs{Email: "a@example.hu"}, "", true},
{"already cleared, empty save keeps it", &NotificationPrefs{EmailCleared: true}, "", true},
{"cleared, new address drops it", &NotificationPrefs{EmailCleared: true}, "b@example.hu", false},
{"address changed", &NotificationPrefs{Email: "a@example.hu"}, "b@example.hu", false},
} {
if got := tc.stored.ClearedBy(tc.newEmail); got != tc.want {
t.Errorf("%s: ClearedBy(%q) = %v, want %v", tc.name, tc.newEmail, got, tc.want)
}
}
}
func TestR922_MarkerPersistsAndDrivesStartupSync(t *testing.T) {
path := filepath.Join(t.TempDir(), "settings.json")
lg := log.New(io.Discard, "", 0)
s, err := Load(path, lg)
if err != nil {
t.Fatal(err)
}
if s.GetNotificationPrefs().SyncOnStartup() {
t.Fatal("a never-configured box would push on startup")
}
if err := s.SetNotificationPrefs(&NotificationPrefs{EmailCleared: true, CooldownHours: 6}); err != nil {
t.Fatal(err)
}
s2, err := Load(path, lg)
if err != nil {
t.Fatal(err)
}
p := s2.GetNotificationPrefs()
if !p.EmailCleared {
t.Fatal("the clear marker did not survive a reload")
}
if !p.SyncOnStartup() {
t.Fatal("a cleared box does not push its clear on startup — a clear the hub missed would never reach it")
}
}
+22
View File
@@ -622,6 +622,28 @@ type NotificationPrefs struct {
Email string `json:"email,omitempty"`
EnabledEvents []string `json:"enabled_events,omitempty"`
CooldownHours int `json:"cooldown_hours,omitempty"` // default: 6
// EmailCleared (R-922, operator ruling 2026-10-09 option A) marks a DELIBERATE clear: the household
// saved an empty address where one was stored. Every hub push carries it as `email_cleared: true`
// while the address stays empty, so the hub deletes the address it holds — and a push the hub missed
// is repaired by the next one. A box that never had an address never sets it (the hub's no-clobber
// guard must keep protecting a seeded address); a new address drops it. Set only by
// web.settingsNotificationsHandler via ClearedBy. Pinned by internal/web TestR922_*.
EmailCleared bool `json:"email_cleared,omitempty"`
}
// ClearedBy reports whether saving `newEmail` over these stored prefs is (or keeps) a deliberate clear:
// the new address is empty AND the stored one was not — or the stored prefs already record a clear.
func (p *NotificationPrefs) ClearedBy(newEmail string) bool {
if newEmail != "" || p == nil {
return false
}
return p.Email != "" || p.EmailCleared
}
// SyncOnStartup reports whether the startup push should run: an address to restore after a hub DB
// rebuild, or a household clear the hub may not have received yet (R-922).
func (p *NotificationPrefs) SyncOnStartup() bool {
return p != nil && (p.Email != "" || p.EmailCleared)
}
// DefaultEnabledEvents are the events enabled by default for new customers.
+1 -1
View File
@@ -616,7 +616,7 @@ func (s *Server) debugPreferencesSync(w http.ResponseWriter, r *http.Request) {
return
}
prefs := s.settings.GetNotificationPrefs()
if err := s.notifier.SyncPreferences(prefs.Email, prefs.EnabledEvents, prefs.CooldownHours); err != nil {
if err := s.notifier.SyncPreferences(prefs.Email, prefs.EnabledEvents, prefs.CooldownHours, prefs.EmailCleared); err != nil {
writeDebugJSON(w, http.StatusOK, false, s.errText(r, err), nil)
return
}
+7 -2
View File
@@ -2957,10 +2957,15 @@ func (s *Server) settingsNotificationsHandler(w http.ResponseWriter, r *http.Req
return
}
// R-922 (operator ruling 2026-10-09, option A): an empty address where one was stored is the household's
// deliberate clear — persisted, so this push AND every later one carry `email_cleared` while the address
// stays empty, and the hub deletes the address it holds. A box that never had one does not set it.
emailCleared := s.settings.GetNotificationPrefs().ClearedBy(email)
prefs := &settings.NotificationPrefs{
Email: email,
EnabledEvents: enabledEvents,
CooldownHours: cooldownHours,
EmailCleared: emailCleared,
}
if err := s.settings.SetNotificationPrefs(prefs); err != nil {
@@ -2971,13 +2976,13 @@ func (s *Server) settingsNotificationsHandler(w http.ResponseWriter, r *http.Req
return
}
s.logger.Printf("[INFO] [web] Notification preferences updated: email=%s, events=%v", email, enabledEvents)
s.logger.Printf("[INFO] [web] Notification preferences updated: email=%s, events=%v, email_cleared=%t", email, enabledEvents, emailCleared)
s.reportTriggerNow() // v0.139.0: hub reflects the saved prefs in seconds, not next cycle
// Sync preferences to hub
data := s.notificationsPageData()
if s.notifier != nil && s.notifier.IsEnabled() {
if err := s.notifier.SyncPreferences(email, enabledEvents, cooldownHours); err != nil {
if err := s.notifier.SyncPreferences(email, enabledEvents, cooldownHours, emailCleared); err != nil {
s.logger.Printf("[WARN] [web] Failed to sync preferences to hub: %v", err)
data["NotificationSuccess"] = s.msg(r, "settings.notify_saved_sync_failed", err)
} else {
@@ -0,0 +1,187 @@
package web
// R-922 (operator ruling 2026-10-09 11:19, option A): when the household clears its notification address on
// the dashboard, the hub deletes the stored address. The hub cannot tell that apart from an unconfigured box
// pushing an empty address (its no-clobber guard, audit F12) — so the controller SAYS so: the push carries
// `"email_cleared": true`, and ONLY after a deliberate clear. Asserted on the WIRE (a real Notifier against an
// httptest hub), not on a mock's call count.
import (
"encoding/json"
"io"
"log"
"net/http"
"net/http/httptest"
"net/url"
"sync"
"testing"
"gitea.dooplex.hu/admin/felhom-controller/internal/notify"
"gitea.dooplex.hu/admin/felhom-controller/internal/settings"
)
type prefsHub struct {
mu sync.Mutex
bodies []map[string]json.RawMessage
}
func (h *prefsHub) last(t *testing.T) map[string]json.RawMessage {
t.Helper()
h.mu.Lock()
defer h.mu.Unlock()
if len(h.bodies) == 0 {
t.Fatal("no preferences push reached the hub")
}
return h.bodies[len(h.bodies)-1]
}
func (h *prefsHub) count() int {
h.mu.Lock()
defer h.mu.Unlock()
return len(h.bodies)
}
// r922Server is notifyGuardServer with a REAL notifier aimed at a capturing hub.
func r922Server(t *testing.T) (*Server, *settings.Settings, *prefsHub) {
t.Helper()
s, sett := notifyGuardServer(t)
hub := &prefsHub{}
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path != "/api/v1/preferences" {
w.WriteHeader(http.StatusOK)
return
}
raw, _ := io.ReadAll(r.Body)
var m map[string]json.RawMessage
if err := json.Unmarshal(raw, &m); err != nil {
t.Errorf("preferences push is not a JSON object: %v", err)
}
hub.mu.Lock()
hub.bodies = append(hub.bodies, m)
hub.mu.Unlock()
w.WriteHeader(http.StatusOK)
}))
t.Cleanup(srv.Close)
s.notifier = notify.New(srv.URL, "test-key", "c1", sett, log.New(io.Discard, "", 0), false)
if !s.notifier.IsEnabled() {
t.Fatal("test notifier not enabled")
}
return s, sett, hub
}
func clearForm() url.Values {
return url.Values{"notification_email": {""}, "cooldown_hours": {"6"}} // no event_* boxes
}
func flagOf(t *testing.T, body map[string]json.RawMessage) (present bool, value bool) {
t.Helper()
raw, ok := body["email_cleared"]
if !ok {
return false, false
}
var v bool
if err := json.Unmarshal(raw, &v); err != nil {
t.Fatalf("email_cleared is not a boolean: %s", raw)
}
return true, v
}
func emailOf(t *testing.T, body map[string]json.RawMessage) string {
t.Helper()
var e string
_ = json.Unmarshal(body["email"], &e)
return e
}
// RED TEST. A household that had an address and clears it: the push carries email_cleared:true with an empty
// address. Today's push carries no flag, so the hub keeps the old address (seen on Tester 1, 2026-10-09).
// The flag also rides the NEXT push (the debug resync reads the stored prefs) and a second empty save —
// it stays while the address stays empty, so a push the hub missed is repaired by the next one.
func TestR922_DeliberateClearSendsEmailCleared(t *testing.T) {
s, sett, hub := r922Server(t)
if err := sett.SetNotificationPrefs(&settings.NotificationPrefs{
Email: "household@example.hu", EnabledEvents: []string{"backup_failed"}, CooldownHours: 6,
}); err != nil {
t.Fatal(err)
}
postNotifications(t, s, clearForm())
body := hub.last(t)
if present, v := flagOf(t, body); !present || !v {
t.Fatalf("R-922: the push after a deliberate clear carries no email_cleared:true (present=%v value=%v) — the hub keeps the old address; body keys=%v", present, v, keysOf(body))
}
if e := emailOf(t, body); e != "" {
t.Fatalf("email = %q, want empty beside the clear flag", e)
}
// The next push from the STORED prefs (the debug resync) carries it too.
before := hub.count()
s.debugPreferencesSync(httptest.NewRecorder(), httptest.NewRequest(http.MethodPost, "/api/debug/preferences/sync", nil))
if hub.count() != before+1 {
t.Fatalf("the resync did not push")
}
if present, v := flagOf(t, hub.last(t)); !present || !v {
t.Fatalf("the stored clear marker did not ride the next push (present=%v value=%v)", present, v)
}
// A second empty save keeps it (the stored address is already empty; the household's clear still stands).
postNotifications(t, s, clearForm())
if present, v := flagOf(t, hub.last(t)); !present || !v {
t.Fatalf("a second empty save dropped the clear flag (present=%v value=%v)", present, v)
}
}
// A box that never had an address must NOT send the flag: its empty push is exactly the case the hub's
// no-clobber guard protects (a provisioning-seeded address must survive an unconfigured box).
func TestR922_NeverConfiguredBoxSendsNoFlag(t *testing.T) {
s, _, hub := r922Server(t)
postNotifications(t, s, clearForm())
if present, _ := flagOf(t, hub.last(t)); present {
t.Fatalf("a never-configured box sent email_cleared — the hub would delete a seeded address; body keys=%v", keysOf(hub.last(t)))
}
s.debugPreferencesSync(httptest.NewRecorder(), httptest.NewRequest(http.MethodPost, "/api/debug/preferences/sync", nil))
if present, _ := flagOf(t, hub.last(t)); present {
t.Fatalf("the resync of a never-configured box sent email_cleared")
}
}
// Setting a new address after a clear drops the flag: the push carries the new address and no flag, and the
// later pushes stay clean.
func TestR922_NewAddressDropsFlag(t *testing.T) {
s, sett, hub := r922Server(t)
if err := sett.SetNotificationPrefs(&settings.NotificationPrefs{
Email: "old@example.hu", EnabledEvents: []string{"backup_failed"}, CooldownHours: 6,
}); err != nil {
t.Fatal(err)
}
postNotifications(t, s, clearForm())
if present, v := flagOf(t, hub.last(t)); !present || !v {
t.Fatalf("precondition: the clear did not send the flag (present=%v value=%v)", present, v)
}
postNotifications(t, s, url.Values{
"notification_email": {"new@example.hu"},
"event_backup_failed": {"on"},
"cooldown_hours": {"6"},
})
body := hub.last(t)
if present, _ := flagOf(t, body); present {
t.Fatalf("the push with a new address still carries email_cleared; body keys=%v", keysOf(body))
}
if e := emailOf(t, body); e != "new@example.hu" {
t.Fatalf("email = %q, want new@example.hu", e)
}
s.debugPreferencesSync(httptest.NewRecorder(), httptest.NewRequest(http.MethodPost, "/api/debug/preferences/sync", nil))
if present, _ := flagOf(t, hub.last(t)); present {
t.Fatalf("the stored marker survived a new address — the next push still carries email_cleared")
}
}
func keysOf(m map[string]json.RawMessage) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
return out
}