55bb6c3d32
gates / gates (push) Successful in 27s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
383 lines
18 KiB
Go
383 lines
18 KiB
Go
// Package offsiteapply is the controller-side apply-bridge: it turns the hub-served offsite descriptor into
|
||
// a working off-site target reached through an APPEND-ONLY key. It VERIFIES the Storage Box host key against
|
||
// the hub-captured fingerprint (no blind TOFU), sends the box's PUBLIC key to the hub's key registrar, which
|
||
// writes it into the sub-account pinned to `rclone serve restic --stdio --append-only <repo>` (decision 69,
|
||
// controller v0.289.0), proves the key reaches the pinned server, and configures the offbox target →
|
||
// EscrowState="pending" (the fork-4 enable path). The box NEVER receives the sub-account password any more —
|
||
// that password can rewrite authorized_keys and remove the pin (R-820). Idempotent and fail-safe (any step
|
||
// fails → nothing persisted, retried next cycle).
|
||
package offsiteapply
|
||
|
||
import (
|
||
"context"
|
||
"crypto/sha256"
|
||
"encoding/hex"
|
||
"fmt"
|
||
"log"
|
||
"os"
|
||
"path/filepath"
|
||
"time"
|
||
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/config"
|
||
"gitea.dooplex.hu/admin/felhom-controller/internal/util"
|
||
)
|
||
|
||
// The apply-bridge seams (tests inject fakes — no live SSH / hub calls in unit tests).
|
||
type (
|
||
// KeyRegistrar is the hub's key registrar (decision 69): install a PUBLIC key pinned append-only,
|
||
// then confirm the one the box uses (the hub drops every other line).
|
||
KeyRegistrar interface {
|
||
Register(ctx context.Context, pub string) (fingerprint string, err error)
|
||
Confirm(ctx context.Context, fingerprint string) error
|
||
}
|
||
// HostKeyScanner returns the box's host-key fingerprint (SHA256:…) + the known_hosts line to pin.
|
||
HostKeyScanner interface {
|
||
Scan(ctx context.Context, host string, port int) (fingerprint, knownHostsLine string, err error)
|
||
}
|
||
// KeyGenerator produces a fresh keypair: the private key (PEM) and the authorized_keys pub line.
|
||
KeyGenerator interface {
|
||
Generate() (privPEM, pubAuthorized string, err error)
|
||
}
|
||
// OffboxEnabler configures the offbox target (key + known_hosts + target + soft quota) and goes
|
||
// EscrowState="pending" (the fork-4 enable path). quotaGB=0 = no soft limit (dedicated boxes).
|
||
OffboxEnabler interface {
|
||
ConfigureOffbox(ctx context.Context, host, user string, port int, repoPath, privPEM, knownHosts string, quotaGB int) error
|
||
}
|
||
// PinnedKeyProber proves a key reaches the PINNED append-only server (not merely that it
|
||
// authenticates — an unpinned key authenticates too, and can delete).
|
||
PinnedKeyProber interface {
|
||
Probe(ctx context.Context, host, user string, port int, knownHosts, privPEM string) bool
|
||
}
|
||
// ExistingKey returns the box's installed off-site private key, "" when none.
|
||
ExistingKey func() string
|
||
// SettleProvider reports the managed-update settle state so the bridge can DEFER consuming the
|
||
// one-time password until any imminent managed floor-update has converged (R-71a — the structural
|
||
// fix for the F10 day-0 race). The failure it prevents: a fresh box boots below the operator floor,
|
||
// the bridge consumes the single-use password, then ~35 s later the managed auto-floor update
|
||
// replaces the container mid-install → the new process finds no installed key → consume → 404 →
|
||
// offsite dead until an operator Re-issue. version = the running controller version; floor = the
|
||
// operator-enforced minimum (from the hub report ACK); updateRunning = a swap is in flight;
|
||
// floorKnown = the floor has been learned yet (false = the first report ACK hasn't landed). A hub
|
||
// that hasn't served a floor cannot serve a consume either, so !floorKnown carries NO burn risk.
|
||
SettleProvider interface {
|
||
SettleState() (version, floor string, updateRunning, floorKnown bool)
|
||
}
|
||
)
|
||
|
||
// Settle-gate timing (R-71a). Named constants with rationale so the trade-offs stay visible.
|
||
const (
|
||
// reconcileTimeout bounds the actual Reconcile once the gate releases (moved here from main.go so
|
||
// ReconcileWhenSettled owns the whole "gate THEN reconcile" contract). The gate's own wait must NOT
|
||
// eat this budget — the reconcile context is created only after the gate returns.
|
||
reconcileTimeout = 3 * time.Minute
|
||
// settlePoll is the gate's poll cadence. The bridge is a background reconcile with no user-visible
|
||
// latency, so a coarse poll is free; the at/above-floor happy path returns on the FIRST evaluation
|
||
// with no sleep at all (the B′ invariant), so this cadence only ever paces the deferral cases.
|
||
settlePoll = 10 * time.Second
|
||
// settleFloorSubBound: if the hub never tells us the floor (report ACK), stop waiting and GO+WARN.
|
||
// Sized to the report-ACK latency observed at source: the startup report fires ~5 s after boot
|
||
// (main.go's startup goroutine sleeps 5 s), and SetFloor runs SYNCHRONOUSLY inside that report's
|
||
// ACK handler (main.go OnPushResponse → updater.SetFloor), so the floor is normally known within
|
||
// ~5–10 s. The startup report retries up to 3× with 15 s gaps, so a slow first report can push
|
||
// floor-knowledge to ~45 s; 90 s is generous headroom over that worst case. Proceeding here cannot
|
||
// burn a one-time password: a hub that cannot serve a floor cannot serve a consume.
|
||
settleFloorSubBound = 90 * time.Second
|
||
// settleOverallBound: the absolute cap. If we are still below floor after this (a managed
|
||
// floor-update that never lands), proceed anyway — the R-71c hub self-heal restage is the belt for
|
||
// a consume that a genuinely stuck update might later burn.
|
||
settleOverallBound = 5 * time.Minute
|
||
)
|
||
|
||
// belowFloor reports whether the running version is strictly below the operator floor. An
|
||
// unparseable version or floor (a dev build, or a malformed floor) is treated as NOT below — the gate
|
||
// must never wedge on a version it cannot compare (and a dev build never auto-floor-updates anyway).
|
||
func belowFloor(version, floor string) bool {
|
||
v, err1 := util.ParseVersion(version)
|
||
f, err2 := util.ParseVersion(floor)
|
||
if err1 != nil || err2 != nil {
|
||
return false
|
||
}
|
||
return v.Compare(f) < 0
|
||
}
|
||
|
||
// Bridge reconciles the offsite descriptor into a configured offbox target.
|
||
type Bridge struct {
|
||
Cfg *config.Config
|
||
Registrar KeyRegistrar
|
||
Scanner HostKeyScanner
|
||
KeyGen KeyGenerator
|
||
Enabler OffboxEnabler
|
||
Prober PinnedKeyProber
|
||
Existing ExistingKey // nil → no existing key (always a fresh keypair)
|
||
MarkerPath string // where the applied-descriptor-hash is persisted (e.g. <dataDir>/offbox/applied_marker)
|
||
Logger *log.Logger
|
||
|
||
// Settle gates the consume/install path behind managed-update convergence (R-71a). nil → no gate
|
||
// (old behavior: reconcile immediately). Wired only when a self-updater exists — with no update
|
||
// mechanism there is no floor-update to race, so no gate is needed.
|
||
Settle SettleProvider
|
||
// Now/Sleep are clock seams for the settle-gate ONLY (tests inject a fake clock so the bounds are
|
||
// exercised with zero real sleeps). nil → the real wall clock and a context-aware sleep.
|
||
Now func() time.Time
|
||
Sleep func(ctx context.Context, d time.Duration)
|
||
}
|
||
|
||
func (b *Bridge) nowFn() func() time.Time {
|
||
if b.Now != nil {
|
||
return b.Now
|
||
}
|
||
return time.Now
|
||
}
|
||
|
||
func (b *Bridge) sleepFn() func(context.Context, time.Duration) {
|
||
if b.Sleep != nil {
|
||
return b.Sleep
|
||
}
|
||
return func(ctx context.Context, d time.Duration) {
|
||
t := time.NewTimer(d)
|
||
defer t.Stop()
|
||
select {
|
||
case <-ctx.Done():
|
||
case <-t.C:
|
||
}
|
||
}
|
||
}
|
||
|
||
func (b *Bridge) logf(f string, a ...any) {
|
||
if b.Logger != nil {
|
||
b.Logger.Printf(f, a...)
|
||
}
|
||
}
|
||
|
||
// descriptorHash is the applied-marker key: a hash of the APPLY-RELEVANT descriptor fields. A change
|
||
// (re-provision → new host/user/fingerprint, or a quota adjustment) yields a new hash → the bridge
|
||
// re-applies. QuotaGB is included (SLICE 4) so a hub-side quota raise reaches the target — on an
|
||
// already-provisioned guest that re-apply is a cheap key-auth-first re-pin (no password consumed).
|
||
func descriptorHash(o config.OffsiteConfig) string {
|
||
// "|pinned-v1" (v0.289.0): a box upgraded from the password era re-applies ONCE, which is what moves
|
||
// its key to the pinned append-only line. Without it the old marker matches and nothing migrates.
|
||
s := fmt.Sprintf("%s|%s|%s|%d|%s|%s|%d|pinned-v1", o.Type, o.Host, o.User, o.Port, o.RepoPath, o.HostFingerprint, o.QuotaGB)
|
||
sum := sha256.Sum256([]byte(s))
|
||
return hex.EncodeToString(sum[:])
|
||
}
|
||
|
||
func (b *Bridge) readMarker() string {
|
||
data, err := os.ReadFile(b.MarkerPath)
|
||
if err != nil {
|
||
return ""
|
||
}
|
||
return string(data)
|
||
}
|
||
|
||
func (b *Bridge) writeMarker(h string) error {
|
||
if err := os.MkdirAll(filepath.Dir(b.MarkerPath), 0o700); err != nil {
|
||
return err
|
||
}
|
||
tmp := b.MarkerPath + ".tmp"
|
||
if err := os.WriteFile(tmp, []byte(h), 0o600); err != nil {
|
||
return err
|
||
}
|
||
return os.Rename(tmp, b.MarkerPath)
|
||
}
|
||
|
||
// Reconcile applies the offsite descriptor. Safe to call repeatedly (idempotent) and on any error leaves
|
||
// nothing half-configured (fail-safe). Returns an error for logging; callers run it async and retry.
|
||
func (b *Bridge) Reconcile(ctx context.Context) error {
|
||
o := b.Cfg.Offsite
|
||
if !o.Enabled {
|
||
return nil // disabled → the fork-4 gate blocks runs; nothing to apply
|
||
}
|
||
port := o.Port
|
||
if port == 0 {
|
||
port = 23
|
||
}
|
||
h := descriptorHash(o)
|
||
if b.readMarker() == h {
|
||
return nil // already applied for this descriptor (idempotent) — do NOT re-consume a spent password
|
||
}
|
||
if o.HostFingerprint == "" {
|
||
return fmt.Errorf("offsite-apply: descriptor has no host_fingerprint — refusing (no blind TOFU)")
|
||
}
|
||
if o.Host == "" || o.User == "" || o.RepoPath == "" {
|
||
return fmt.Errorf("offsite-apply: descriptor missing host/user/repo_path")
|
||
}
|
||
|
||
// 1) Scan + VERIFY the host key BEFORE consuming the password (don't waste it on a mismatch).
|
||
scannedFP, knownHostsLine, err := b.Scanner.Scan(ctx, o.Host, port)
|
||
if err != nil {
|
||
return fmt.Errorf("offsite-apply: host-key scan: %w", err)
|
||
}
|
||
if scannedFP != o.HostFingerprint {
|
||
return fmt.Errorf("offsite-apply: host-key MISMATCH for %s (got %s, want %s) — refusing to pin/install (possible MITM)", o.Host, scannedFP, o.HostFingerprint)
|
||
}
|
||
|
||
// 2) The key: the one already installed (a running box, or one upgraded from the password era — the
|
||
// SAME key is re-registered and comes back pinned), else a fresh pair.
|
||
privPEM, pub, fresh := "", "", false
|
||
if b.Existing != nil {
|
||
if pem := b.Existing(); pem != "" {
|
||
if p, perr := PublicKeyOf(pem); perr == nil {
|
||
privPEM, pub = pem, p
|
||
} else {
|
||
b.logf("[WARN] [offsite-apply] the installed key does not parse (%v) — generating a fresh one", perr)
|
||
}
|
||
}
|
||
}
|
||
if privPEM == "" {
|
||
if privPEM, pub, err = b.KeyGen.Generate(); err != nil {
|
||
return fmt.Errorf("offsite-apply: keygen: %w", err)
|
||
}
|
||
fresh = true
|
||
}
|
||
fp, err := FingerprintOf(pub)
|
||
if err != nil {
|
||
return fmt.Errorf("offsite-apply: own key: %w", err)
|
||
}
|
||
|
||
// 3) Already pinned? (a descriptor change on a healthy box, or a restart) → no hub write needed.
|
||
pinned := b.Prober.Probe(ctx, o.Host, o.User, port, knownHostsLine, privPEM)
|
||
if !pinned {
|
||
// 4) Ask the hub's registrar to install it pinned append-only. NO password crosses this call.
|
||
hubFP, rerr := b.Registrar.Register(ctx, pub)
|
||
if rerr != nil {
|
||
return fmt.Errorf("offsite-apply: register key with the hub: %w", rerr)
|
||
}
|
||
if hubFP != fp {
|
||
return fmt.Errorf("offsite-apply: hub installed key %s, ours is %s — refusing", hubFP, fp)
|
||
}
|
||
// 5) Prove the key now reaches the PINNED server (positive observable, not just auth).
|
||
if !b.Prober.Probe(ctx, o.Host, o.User, port, knownHostsLine, privPEM) {
|
||
return fmt.Errorf("offsite-apply: key %s registered but does not reach the pinned append-only server", fp)
|
||
}
|
||
b.logf("[INFO] [offsite-apply] the hub installed key %s append-only on %s@%s (fresh=%v)", fp, o.User, o.Host, fresh)
|
||
}
|
||
|
||
// 6) Configure the offbox target (writes the key) + go EscrowState="pending" when new (fork-4).
|
||
if err := b.Enabler.ConfigureOffbox(ctx, o.Host, o.User, port, o.RepoPath, privPEM, knownHostsLine, o.QuotaGB); err != nil {
|
||
return fmt.Errorf("offsite-apply: configure offbox: %w", err)
|
||
}
|
||
|
||
// 7) Confirm: the hub removes every other key line (rotation's last step). Best effort — a failed
|
||
// confirm leaves an extra PINNED line, which cannot delete; the next apply retries it.
|
||
if err := b.Registrar.Confirm(ctx, fp); err != nil {
|
||
b.logf("[WARN] [offsite-apply] confirm key %s with the hub failed (an extra pinned line may remain; retried on the next apply): %v", fp, err)
|
||
}
|
||
|
||
// 8) Persist the marker LAST — only a fully-applied descriptor is recorded (fail-safe).
|
||
if err := b.writeMarker(h); err != nil {
|
||
b.logf("[WARN] [offsite-apply] applied offsite for %s but failed to persist the marker (re-applies next cycle; harmless — no password is involved): %v", o.Host, err)
|
||
return err
|
||
}
|
||
b.logf("[INFO] [offsite-apply] offsite configured append-only for %s@%s:%s (key %s)", o.User, o.Host, o.RepoPath, fp)
|
||
return nil
|
||
}
|
||
|
||
// AwaitSettle blocks until it is safe to run the apply-bridge, then returns (R-71a). It removes the
|
||
// SYSTEMATIC trigger for the F10 day-0 race by refusing to consume the one-time password while a
|
||
// managed floor-update is in flight or imminent (we are below the floor): that update's restart would
|
||
// supersede the bridge and kill it mid-install, spending the password for nothing. It NEVER blocks
|
||
// when it is genuinely safe — the overwhelmingly common shape (a restart of an at/above-floor box)
|
||
// evaluates GO on the first poll with no sleep at all (the B′ invariant: zero new latency).
|
||
//
|
||
// It is a strict no-op unless a SettleProvider is wired (nil → old behavior, reconcile immediately).
|
||
// The gate has its own bounds (settlePoll/settleFloorSubBound/settleOverallBound) and its own context
|
||
// so the deferral never eats the reconcile budget.
|
||
func (b *Bridge) AwaitSettle(ctx context.Context) {
|
||
if b.Settle == nil {
|
||
return
|
||
}
|
||
now, sleep := b.nowFn(), b.sleepFn()
|
||
start := now()
|
||
var loggedUpdate, loggedDefer, loggedFloorWait bool
|
||
for {
|
||
version, floor, updateRunning, floorKnown := b.Settle.SettleState()
|
||
elapsed := now().Sub(start)
|
||
switch {
|
||
case updateRunning:
|
||
// A swap is in flight; its restart supersedes us. Wait it out.
|
||
if !loggedUpdate {
|
||
b.logf("[INFO] [offsite-apply] settle-gate: a managed update is in progress — deferring offsite apply until it converges")
|
||
loggedUpdate = true
|
||
}
|
||
case floorKnown && belowFloor(version, floor):
|
||
// The auto-floor update is imminent (below floor + floor known). Do NOT consume — the
|
||
// update's restart would burn the password. Wait for the update to land (which restarts us
|
||
// at floor → the GO branch below).
|
||
if !loggedDefer {
|
||
b.logf("[INFO] [offsite-apply] deferring offsite apply: managed update to floor %s pending (we are %s)", floor, version)
|
||
loggedDefer = true
|
||
}
|
||
case floorKnown:
|
||
// At/above floor, no update running — the safe steady state. GO.
|
||
b.logf("[INFO] [offsite-apply] settle-gate: GO — at/above floor %s (we are %s), no managed update running", floor, version)
|
||
return
|
||
case elapsed >= settleFloorSubBound:
|
||
// Floor never became known within the sub-bound. A hub that will not tell us the floor
|
||
// cannot serve a consume either, so the burn risk is nil — don't hold offsite hostage.
|
||
b.logf("[WARN] [offsite-apply] settle-gate: GO — floor still unknown after %s; a hub that cannot serve a floor cannot serve a consume (no burn risk)", settleFloorSubBound)
|
||
return
|
||
default:
|
||
// Floor not known yet, still inside the sub-bound — wait for the report ACK.
|
||
if !loggedFloorWait {
|
||
b.logf("[INFO] [offsite-apply] settle-gate: awaiting floor knowledge (first report ACK) before offsite apply")
|
||
loggedFloorWait = true
|
||
}
|
||
}
|
||
if elapsed >= settleOverallBound {
|
||
b.logf("[WARN] [offsite-apply] settle-gate bound exhausted after %s — proceeding; R-71c self-heal is the belt", settleOverallBound)
|
||
return
|
||
}
|
||
sleep(ctx, settlePoll)
|
||
if ctx.Err() != nil {
|
||
return // shutdown / cancellation — abandon the gate (the next start retries)
|
||
}
|
||
}
|
||
}
|
||
|
||
// ReconcileWhenSettled runs the settle-gate (R-71a) and THEN Reconcile under a FRESH reconcile
|
||
// context. The gate's deferral must not eat the reconcile budget, so the reconcile timeout starts
|
||
// only after the gate releases. gateCtx bounds the gate (e.g. process shutdown); a cancelled gate
|
||
// skips the reconcile (the next start retries).
|
||
func (b *Bridge) ReconcileWhenSettled(gateCtx context.Context) error {
|
||
b.AwaitSettle(gateCtx)
|
||
if gateCtx.Err() != nil {
|
||
return gateCtx.Err()
|
||
}
|
||
ctx, cancel := context.WithTimeout(context.Background(), reconcileTimeout)
|
||
defer cancel()
|
||
return b.Reconcile(ctx)
|
||
}
|
||
|
||
// ── R-218, THE CONSUME HALF ──────────────────────────────────────────────────────────────────────
|
||
//
|
||
// `Reconcile` was correct from the day it shipped and was simply never run again. It fires at
|
||
// start-up and once more when the recovery screen drives it (R-219) — and BOTH precede the moment the
|
||
// hub has anything staged, because the hub stages in RESPONSE to the declaration those runs come
|
||
// before. So the hub held a credential the box would never fetch.
|
||
//
|
||
// Measured on the R-201 re-walk, 2026-08-06: unlock reconcile 11:43:07 · hub staged 11:44:57 saying
|
||
// "the box re-consumes on its next cycle" · a full report cycle ran 11:55:46 · still unconsumed at
|
||
// 12:06. A guest command line moved it in 18 seconds — everything was fine except the trigger.
|
||
|
||
// NeedsCredentialFunc reports whether the box STILL declares it needs a transport credential. It is
|
||
// deliberately the box's own published declaration (`backup.OffboxReportStatus().State`) rather than a
|
||
// second predicate: the hub acts on that statement, so driving the retry from anything else would let
|
||
// the two disagree about whether a retry is wanted.
|
||
type NeedsCredentialFunc func() bool
|
||
|
||
// RetryIfDeclared is ONE tick of the consume half.
|
||
//
|
||
// It reconciles **only while the box declares a need**, which is what makes it stop: the instant a
|
||
// target exists the declaration goes false, this returns immediately, and a healthy box does no work
|
||
// and logs nothing. The settle gate is deliberately preserved — `ReconcileWhenSettled` waits for floor
|
||
// knowledge exactly as the start-up path does, because the day-0 race it guards is unchanged.
|
||
//
|
||
// Returns whether a reconcile was ATTEMPTED, so a caller (and a test) can tell "declined to run" from
|
||
// "ran and failed" without reading the log.
|
||
func (b *Bridge) RetryIfDeclared(ctx context.Context, declared NeedsCredentialFunc) (attempted bool, err error) {
|
||
if b == nil || declared == nil || !declared() {
|
||
return false, nil
|
||
}
|
||
return true, b.ReconcileWhenSettled(ctx)
|
||
}
|