R-204 item 4 (hub half): the hub answers a rebuilt box's request (hub v0.96.0)

New internal/offsiteheal, the sibling of pbsdrheal: it acts ONLY on the state the
box declares, sustained across two distinct reports, re-staging the stored
credential before ever minting a new one. A healthy box is a pure no-op; it never
blind-timer-reissues and never re-runs a provisioning step.

RESTAGE IS POSSIBLE because the stored value survives a consume — established from
the schema and ConsumeOneTimeSecret (which stamps consumed_at and nothing else),
not inherited from the PBS analogy, and pinned by a test that asserts the SAME
value comes back.

reportHasOffsite is TIGHTENED to require enabled:true. Its comment asserted that
presence == applied-on-the-box, and the declaration deliberately breaks that
premise; left alone it would have read a request for help as proof the tier was
applied. Provably a no-op for every report shape that existed before, because an
attached object has always carried enabled:true.

R-192's guard half is CLOSED BY REPLACEMENT: the delivery checker's counting
inference read the OLDEST 500 reports after a consume — all predating a rebuild,
which is why demo-hp sat stranded for 108 reports under a confident regressed-shape
verdict. A declaration outranks both inferred shapes, and the checker stands down
with a record so the two mechanisms cannot double-issue.

No escrow ceremony is ever run or requested: credential automatic, key
customer-present.
This commit is contained in:
2026-08-05 10:48:18 +02:00
parent c917251eeb
commit f62a115891
8 changed files with 1026 additions and 16 deletions
+310
View File
@@ -0,0 +1,310 @@
package offsiteheal
// Non-hollow reconciler tests mapping 1:1 to the task's integration scenarios AF. A REAL store
// (t.TempDir sqlite) supplies the work set, the customer configs and the reports; a FAKE Actions seam
// counts restage/reissue calls without touching a storage provider. ReconcileOnce is driven directly
// for determinism — the debounce is exercised by feeding DISTINCT REPORTS, never by sleeping, because
// the debounce counts fresh evidence rather than elapsed time.
import (
"context"
"encoding/json"
"fmt"
"io"
"log"
"path/filepath"
"sync"
"testing"
"gitea.dooplex.hu/admin/felhom-hub/internal/store"
)
type fakeActions struct {
mu sync.Mutex
restaged []string
reissued []string
restageResult bool // does a stored secret row exist?
restageErr error
reissueErr error
}
func (f *fakeActions) Restage(customerID string) (bool, error) {
f.mu.Lock()
defer f.mu.Unlock()
f.restaged = append(f.restaged, customerID)
return f.restageResult, f.restageErr
}
func (f *fakeActions) Reissue(_ context.Context, customerID string) error {
f.mu.Lock()
defer f.mu.Unlock()
f.reissued = append(f.reissued, customerID)
return f.reissueErr
}
func (f *fakeActions) restages() int { f.mu.Lock(); defer f.mu.Unlock(); return len(f.restaged) }
func (f *fakeActions) reissues() int { f.mu.Lock(); defer f.mu.Unlock(); return len(f.reissued) }
func newHealStore(t *testing.T) *store.Store {
t.Helper()
st, err := store.New(filepath.Join(t.TempDir(), "test.db"), log.New(io.Discard, "", 0))
if err != nil {
t.Fatalf("store.New: %v", err)
}
t.Cleanup(func() { st.Close() })
return st
}
// seedCustomer writes an ACTIVE customer whose offsite descriptor is enabled (or not).
func seedCustomer(t *testing.T, st *store.Store, customerID string, offsiteEnabled bool) {
t.Helper()
cfgJSON := `{"offsite":{"enabled":false}}`
if offsiteEnabled {
cfgJSON = `{"offsite":{"enabled":true,"type":"shared","host":"x.your-storagebox.de","user":"u","port":23,"repo_path":"/home/felhom-repo"}}`
}
if err := st.SaveCustomerConfig(&store.CustomerConfig{
CustomerID: customerID, CustomerName: customerID, Domain: customerID + ".hu",
APIKey: "k-" + customerID, ConfigJSON: cfgJSON,
}); err != nil {
t.Fatalf("SaveCustomerConfig %s: %v", customerID, err)
}
}
// pushReport writes ONE controller report. state=="" → a healthy configured box (enabled offsite
// object); otherwise the declaration shape controller v0.199.0 emits. Each call is a DISTINCT report,
// which is the debounce's currency.
func pushReport(t *testing.T, st *store.Store, customerID, state string, seq int) {
t.Helper()
var off map[string]any
if state == "" {
off = map[string]any{"enabled": true, "escrow_state": "escrowed", "last_status": "ok"}
} else {
off = map[string]any{"enabled": false, "state": state}
}
b, err := json.Marshal(map[string]any{
"controller_version": fmt.Sprintf("0.199.0-%d", seq), // makes each row distinguishable
"offsite": off,
})
if err != nil {
t.Fatal(err)
}
if err := st.SaveReport(customerID, b); err != nil {
t.Fatalf("SaveReport %s: %v", customerID, err)
}
}
// pushReportNoOffsite writes a report with NO offsite object at all — the pre-v0.199.0 shape a
// stranded box used to send, and the shape a box that never had off-site backups still sends.
func pushReportNoOffsite(t *testing.T, st *store.Store, customerID string, seq int) {
t.Helper()
b, _ := json.Marshal(map[string]any{"controller_version": fmt.Sprintf("0.198.0-%d", seq)})
if err := st.SaveReport(customerID, b); err != nil {
t.Fatalf("SaveReport %s: %v", customerID, err)
}
}
func newRec(st *store.Store, act Actions) *Reconciler {
return NewReconciler(st, act, log.New(io.Discard, "", 0))
}
func countEvents(t *testing.T, st *store.Store, customerID, eventType string) int {
t.Helper()
ev, err := st.GetLatestEventByType(customerID, eventType)
if err != nil {
t.Fatalf("GetLatestEventByType: %v", err)
}
if ev == nil {
return 0
}
return 1
}
// SCENARIO A + D — a declaring box is served, and the STORED secret is re-armed rather than a fresh
// credential minted.
//
// RED-PROOF (D): invert the order in heal() so Reissue is called before Restage — the assertion
// "no mint" fails, and the test names the cost: external churn for a credential the hub already has.
func TestScenarioAD_DeclaringBoxIsRestagedNotMinted(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true} // a stored secret exists
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
pushReport(t, st, "c1", StateNeedsCredential, 1)
r.ReconcileOnce(context.Background())
if act.restages() != 0 || act.reissues() != 0 {
t.Fatalf("acted on a SINGLE declaration — the debounce did not hold (restages=%d reissues=%d)", act.restages(), act.reissues())
}
pushReport(t, st, "c1", StateNeedsCredential, 2) // a second DISTINCT report
r.ReconcileOnce(context.Background())
if act.restages() != 1 {
t.Fatalf("a sustained declaration was not served: restages=%d", act.restages())
}
if act.reissues() != 0 {
t.Fatalf("a fresh credential was minted although a stored one could be re-armed — external churn for nothing (reissues=%d)", act.reissues())
}
if countEvents(t, st, "c1", eventRestaged) != 1 {
t.Error("no offsite_selfheal_restaged audit event — an automatic remediation must be visible")
}
if countEvents(t, st, "c1", eventReissued) != 0 {
t.Error("a reissue event was recorded for a restage")
}
}
// SCENARIO B — a box that never had off-site backups says nothing, so nothing happens. (The
// declaration itself is the controller's guard; here we assert the hub side is a no-op on silence.)
func TestScenarioB_SilentBoxIsNeverActedOn(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true}
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
for i := 1; i <= 4; i++ {
pushReportNoOffsite(t, st, "c1", i)
r.ReconcileOnce(context.Background())
}
if act.restages() != 0 || act.reissues() != 0 {
t.Fatalf("a silent box was acted on: restages=%d reissues=%d", act.restages(), act.reissues())
}
}
// SCENARIO C — a healthy box is a PURE no-op, repeatedly. No restage, no mint, no event.
func TestScenarioC_HealthyBoxIsAPureNoOp(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true}
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
for i := 1; i <= 5; i++ {
pushReport(t, st, "c1", "", i) // healthy: enabled offsite object, no declaration
r.ReconcileOnce(context.Background())
}
if act.restages() != 0 || act.reissues() != 0 {
t.Fatalf("a healthy box was acted on: restages=%d reissues=%d", act.restages(), act.reissues())
}
if countEvents(t, st, "c1", eventRestaged)+countEvents(t, st, "c1", eventReissued) != 0 {
t.Error("a healthy box produced a self-heal event")
}
}
// SCENARIO E — nothing to re-arm escalates to a mint, ONCE, and does not repeat while the staged
// secret sits unconsumed (a fresh report must re-confirm before acting again).
func TestScenarioE_NothingToRestageEscalatesOnce(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: false} // NO stored secret
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
pushReport(t, st, "c1", StateNeedsCredential, 1)
r.ReconcileOnce(context.Background())
pushReport(t, st, "c1", StateNeedsCredential, 2)
r.ReconcileOnce(context.Background())
if act.reissues() != 1 {
t.Fatalf("escalation did not happen exactly once: reissues=%d", act.reissues())
}
if countEvents(t, st, "c1", eventReissued) != 1 {
t.Error("no offsite_selfheal_reissued audit event")
}
// Repeated ticks on the SAME report must not mint again — this is the "a mint on every tick"
// failure the sibling's resetAfterHeal exists to prevent.
for i := 0; i < 5; i++ {
r.ReconcileOnce(context.Background())
}
if act.reissues() != 1 {
t.Fatalf("minted repeatedly on an unchanged report: reissues=%d", act.reissues())
}
}
// SCENARIO F — a BLIP is absorbed: one declaring report followed by a healthy one triggers nothing.
//
// RED-PROOF: set debounceReports to 1 (i.e. remove the debounce) — the blip acts, and a credential is
// churned by what was only a restart.
func TestScenarioF_BlipIsAbsorbedByTheDebounce(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true}
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
pushReport(t, st, "c1", StateNeedsCredential, 1) // the blip
r.ReconcileOnce(context.Background())
pushReport(t, st, "c1", "", 2) // healthy again
r.ReconcileOnce(context.Background())
if act.restages() != 0 || act.reissues() != 0 {
t.Fatalf("a one-report blip triggered a credential action: restages=%d reissues=%d", act.restages(), act.reissues())
}
// And the streak must have been FORGOTTEN, not merely paused: a single later declaration must
// still not act.
pushReport(t, st, "c1", StateNeedsCredential, 3)
r.ReconcileOnce(context.Background())
if act.restages() != 0 {
t.Fatal("the debounce streak survived a healthy report — a flapping box would be healed on every other cycle")
}
}
// A DISABLED offsite descriptor is the operator's own choice and must never be re-credentialed,
// even if the box declares (e.g. an old declaration left in the newest report).
func TestDisabledDescriptorIsNeverHealed(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true}
r := newRec(st, act)
seedCustomer(t, st, "c1", false) // offsite disabled in the config
for i := 1; i <= 4; i++ {
pushReport(t, st, "c1", StateNeedsCredential, i)
r.ReconcileOnce(context.Background())
}
if act.restages() != 0 || act.reissues() != 0 {
t.Fatalf("a customer whose offsite is DISABLED was re-credentialed: restages=%d reissues=%d", act.restages(), act.reissues())
}
}
// A blocked customer is never healed.
func TestBlockedCustomerIsNeverHealed(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true}
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
// SaveCustomerConfig always writes status 'active' (its INSERT does not carry the column), so the
// block must go through the real setter — asserted below, because a test that silently failed to
// block would pass for the wrong reason.
if err := st.SetCustomerConfigStatus("c1", "blocked"); err != nil {
t.Fatalf("SetCustomerConfigStatus: %v", err)
}
if !st.IsCustomerBlocked("c1") {
t.Fatal("precondition: the customer is not actually blocked — this test would pass vacuously")
}
for i := 1; i <= 4; i++ {
pushReport(t, st, "c1", StateNeedsCredential, i)
r.ReconcileOnce(context.Background())
}
if act.restages() != 0 || act.reissues() != 0 {
t.Fatalf("a BLOCKED customer was healed: restages=%d reissues=%d", act.restages(), act.reissues())
}
}
// A restage ERROR must not escalate to a mint — the reconciler does nothing and retries. Acting on a
// failed read is the "never act on a partial view" rule.
func TestRestageErrorDoesNotEscalate(t *testing.T) {
st := newHealStore(t)
act := &fakeActions{restageResult: true, restageErr: fmt.Errorf("db is busy")}
r := newRec(st, act)
seedCustomer(t, st, "c1", true)
pushReport(t, st, "c1", StateNeedsCredential, 1)
r.ReconcileOnce(context.Background())
pushReport(t, st, "c1", StateNeedsCredential, 2)
r.ReconcileOnce(context.Background())
if act.reissues() != 0 {
t.Fatalf("a failed re-stage escalated to an external mint: reissues=%d", act.reissues())
}
if countEvents(t, st, "c1", eventRestaged) != 0 {
t.Error("a failed re-stage recorded a success event")
}
}