f5c9411e5e
The customer page's Backup card read `Snapshots 0 / Repo Size 0 MB / Integrity
Unknown` for EVERY customer, indefinitely. Measured on demo-hp 2026-08-30 while
that night's controller log said `[offbox] backup OK: 8 app(s) backed up, 67
snapshot(s), 2m14s` and the box held snapshot_count:67, repo_size_bytes:
140829678, stats_known:true.
A card reading "no backups" over a working backup is worse than no card -- the
R-88 direction of failure (degrade to NO BACKUP rather than to UNKNOWN) on the
one screen that answers "is this customer protected?".
The data was never missing. The card rendered the report's `backup` object,
whose snapshot/size/integrity fields have had no producer since slice 8C. The
live numbers are in the `offsite` object, which THIS PACKAGE already reads for
the Offsite page and which monitor.OffsiteChecker already alarms from. Proof the
bytes were arriving: the Offsite page rendered demo-hp's usage as 0.1 GB from
that very object while the Backup card said 0 MB. So this is a render fix over
an existing feed, not a new pipeline.
Not a one-line swap, because snapshot_count:0 means two opposite things --
"holds nothing" and "never measured". R-225 measured that confusion one layer
down. backup_card.go resolves a three-way ruling in Go (a {{if}} chain over
map[string]interface{} float64s cannot keep the absent/zero distinction the card
is entirely about):
no offsite object -> "No off-site data reported", and says explicitly that
this is NOT the same as "no backups"
disabled + state -> names the blocker (needs_credential)
stats_known:false -> em-dash + "never been measured". NEVER 0
stats_known:true -> the real numbers, INCLUDING a real 0
A pre-v0.225.0 controller sends no stats_known -> false -> "unknown". That
direction is pinned: upgrading the hub ahead of the fleet must not report every
un-upgraded customer as having zero backups.
The Integrity row is DELETED, not re-sourced: nothing produces it, the
controller runs no integrity check, and NotifyIntegrityOK/Failed are called from
nowhere.
RED-PROOF: restore the pre-fix card markup -> all four tests fail, reporting 67
and 134.3 MB absent from the rendered page and the Integrity row present. The
tests drive handleCustomerUnified and grep the HTML on purpose: the defect was
the template's choice of source object, so a test one layer below it would have
been green against the shipped bug.
Green gate clean: 18 packages, rc 0.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LB8FmJaGd2cyjvy6dbEjpM
338 lines
12 KiB
YAML
338 lines
12 KiB
YAML
# Felhom Hub — Multi-customer dashboard
|
|
# Dashboard: https://hub.felhom.eu
|
|
# API: POST /api/v1/report (Bearer token auth)
|
|
#
|
|
# Receives health reports from customer controllers and displays
|
|
# a centralized overview dashboard for the operator (Viktor).
|
|
#
|
|
# Namespace: felhom-system (shared with healthchecks and other felhom infra)
|
|
#
|
|
# PREREQUISITES:
|
|
# 1. Build and push the hub image:
|
|
# cd ~/build/felhom-hub && ./build.sh v0.2.0 --push
|
|
#
|
|
# 2. Generate a bcrypt password hash for dashboard login:
|
|
# htpasswd -nbBC 10 "" "your-password" | cut -d: -f2
|
|
# Update the ConfigMap password_hash field below.
|
|
#
|
|
# 3. Create the operator/global bearer key Secret (out-of-band, NEVER committed):
|
|
# openssl rand -hex 32 # mint
|
|
# kubectl -n felhom-system create secret generic report-api \
|
|
# --from-literal=REPORT_API_KEY=<minted-key>
|
|
# (Customer boxes use per-customer/per-host keys generated by the hub — the global
|
|
# key is the operator's own, e.g. felhom-ops -hub-key.)
|
|
#
|
|
# 4. Apply this manifest:
|
|
# kubectl apply -f manifests/hub.yaml
|
|
#
|
|
# 5. Configure DNS:
|
|
# Add hub.felhom.eu → k3s cluster IP in Cloudflare
|
|
#
|
|
# DEBUGGING:
|
|
# kubectl logs -n felhom-system deploy/hub -f
|
|
# kubectl exec -it -n felhom-system deploy/hub -- ls /data/
|
|
# kubectl describe ingress -n felhom-system hub
|
|
|
|
# =============================================================================
|
|
# PERSISTENT STORAGE
|
|
# =============================================================================
|
|
---
|
|
apiVersion: v1
|
|
kind: PersistentVolumeClaim
|
|
metadata:
|
|
name: hub-data
|
|
namespace: felhom-system
|
|
labels:
|
|
app: hub
|
|
recurring-job-group.longhorn.io/default: disabled
|
|
spec:
|
|
accessModes:
|
|
- ReadWriteOnce
|
|
storageClassName: longhorn
|
|
resources:
|
|
requests:
|
|
storage: 1Gi
|
|
|
|
# =============================================================================
|
|
# CONFIGURATION
|
|
# =============================================================================
|
|
---
|
|
apiVersion: v1
|
|
kind: ConfigMap
|
|
metadata:
|
|
name: hub-config
|
|
namespace: felhom-system
|
|
data:
|
|
hub.yaml: |
|
|
auth:
|
|
# Bcrypt hash for dashboard login (Viktor only)
|
|
# Generate: htpasswd -nbBC 10 "" "your-password" | cut -d: -f2
|
|
password_hash: "$2y$10$N5.O9jBnc.1tIlJT/irx3OlVjJQemlCHRnfqIJg/EyZofnzXSCpeG"
|
|
api:
|
|
# Operator/global bearer key. NOT stored here since v0.53.0 — injected at runtime from
|
|
# Secret/report-api via the REPORT_API_KEY env var (see Deployment below). The Secret is
|
|
# created out-of-band and NOT committed (documentation/runbooks/secrets.md); the previously
|
|
# committed literal is retired by ROTATION (see the publish-runbook notes). Leave empty.
|
|
report_api_key: ""
|
|
retention:
|
|
max_days: 90
|
|
prune_schedule: "04:30"
|
|
alerting:
|
|
stale_threshold: "30m"
|
|
notifications:
|
|
# Resend API key is NOT stored here. It is injected at runtime from Secret/resend-api
|
|
# via the RESEND_API_KEY env var (see Deployment below). The Secret is created out-of-band
|
|
# and is NOT committed — see documentation/runbooks/secrets.md. Leave this empty.
|
|
resend_api_key: ""
|
|
# Operator alert recipient + enable. WITHOUT both, Dispatcher.processOperator returns early and
|
|
# NO operator email is ever sent — the self-health pipeline (probe→report→checker→dispatch) stops
|
|
# one hop short of the inbox (TESTRUN finding: the unproven hop). The address is the operator's own
|
|
# and is not a secret. from_email defaults to monitoring@felhom.eu.
|
|
operator_email: "admin@felhom.eu"
|
|
operator_enabled: true
|
|
registry:
|
|
image: "gitea.dooplex.hu/admin/felhom-controller"
|
|
# username + token injected via REGISTRY_USERNAME / REGISTRY_TOKEN env vars
|
|
# from Secret/gitea-creds (see Deployment below)
|
|
check_interval: "6h"
|
|
template_interval: "1h"
|
|
server:
|
|
listen: ":8080"
|
|
data_dir: "/data"
|
|
|
|
# =============================================================================
|
|
# DEPLOYMENT
|
|
# =============================================================================
|
|
---
|
|
apiVersion: apps/v1
|
|
kind: Deployment
|
|
metadata:
|
|
name: hub
|
|
namespace: felhom-system
|
|
labels:
|
|
app: hub
|
|
spec:
|
|
replicas: 1
|
|
strategy:
|
|
type: Recreate
|
|
selector:
|
|
matchLabels:
|
|
app: hub
|
|
template:
|
|
metadata:
|
|
labels:
|
|
app: hub
|
|
spec:
|
|
containers:
|
|
- name: hub
|
|
image: gitea.dooplex.hu/admin/felhom-hub:0.109.0
|
|
ports:
|
|
- containerPort: 8080
|
|
name: http
|
|
env:
|
|
- name: TZ
|
|
value: "Europe/Budapest"
|
|
# Phase 2 managed updates: global controller-version FLOOR fallback. Any reporting box below
|
|
# this auto-updates to it (unless a per-customer override is set via the operator UI).
|
|
- name: DEFAULT_MIN_CONTROLLER_VERSION
|
|
value: "0.120.0"
|
|
# Resend API key — injected from the out-of-band Secret/resend-api (NOT committed).
|
|
# See documentation/runbooks/secrets.md. Overrides the empty ConfigMap placeholder.
|
|
- name: RESEND_API_KEY
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: resend-api
|
|
key: RESEND_API_KEY
|
|
# Operator/global bearer key — injected from the out-of-band Secret/report-api
|
|
# (NOT committed; documentation/runbooks/secrets.md). Deliberately NOT optional:
|
|
# a missing Secret must fail the pod Ready rather than boot an unauthenticatable
|
|
# hub with an empty bearer key. Create the Secret BEFORE syncing this manifest.
|
|
- name: REPORT_API_KEY
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: report-api
|
|
key: REPORT_API_KEY
|
|
- name: REGISTRY_USERNAME
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: gitea-creds
|
|
key: username
|
|
- name: REGISTRY_TOKEN
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: gitea-creds
|
|
key: password
|
|
# S1 offsite connectivity: the WG peer-sync push channel (doc 06 §5 + runbook
|
|
# offsite-endpoint.md). Addr is dev-phase literal (the throwaway endpoint); the SSH
|
|
# private key + (non-secret) pinned host key come from Secret/wg-endpoint-ssh,
|
|
# created out-of-band in runbook step 6 — optional so the pod starts before it
|
|
# exists (the hub logs peer-sync disabled until then).
|
|
- name: WG_ENDPOINT_SSH_ADDR
|
|
value: "167.233.158.164:22"
|
|
- name: WG_ENDPOINT_SSH_USER
|
|
value: "felhom-peersync"
|
|
- name: WG_ENDPOINT_SSH_KEY_FILE
|
|
value: "/etc/hub-secrets/wg-endpoint-ssh/key"
|
|
- name: WG_ENDPOINT_SSH_HOSTKEY
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: wg-endpoint-ssh
|
|
key: hostkey
|
|
optional: true
|
|
# PBS DR tier (SLICE 1): the tenantsync channel — same endpoint + pinned host key as
|
|
# peersync (env above), its OWN private key from Secret/tenantsync (out-of-band,
|
|
# runbook offsite-endpoint.md §10). Optional: absent → the hub logs tenantsync disabled.
|
|
- name: TENANTSYNC_SSH_KEY_FILE
|
|
value: "/etc/hub-secrets/tenantsync/key"
|
|
# Agent-plane immediate-sync (Direction-2a, v0.59.0): the poke sender — same endpoint +
|
|
# pinned host key + peersync user as above, its OWN forced-command key from Secret/agent-poke
|
|
# (out-of-band; the ep0 authorized_keys line carries the PUBLIC half, command="felhom-poke").
|
|
# Optional: absent → the hub logs the poke disabled; agent-plane saves still reconcile in
|
|
# ≤15 min. See documentation/runbooks/offsite-endpoint.md (poke section).
|
|
- name: POKE_SSH_KEY_FILE
|
|
value: "/etc/hub-secrets/agent-poke/key"
|
|
# Offsite provisioning (SLICE 1+2): Hetzner Storage Box API token + the NUMERIC id of the
|
|
# pool box, from the out-of-band Secret/storagebox (NOT committed). The token MUST be scoped
|
|
# to the dedicated storage project — NEVER the shared-project token (it can touch ep0).
|
|
# HETZNER_POOL_BOX_ID is the numeric box id (console #id), not the box name. Optional so the
|
|
# pod starts before the secret exists (hub then logs offsite provisioning disabled).
|
|
- name: HETZNER_POOL_BOX_ID
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: storagebox
|
|
key: HETZNER_POOL_BOX_ID
|
|
optional: true
|
|
- name: HETZNER_TOKEN
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: storagebox
|
|
key: HETZNER_TOKEN
|
|
optional: true
|
|
# Non-secret; the hub defaults to fsn1 anyway — explicit for clarity.
|
|
- name: HETZNER_LOCATION
|
|
value: "fsn1"
|
|
resources:
|
|
requests:
|
|
memory: "64Mi"
|
|
cpu: "50m"
|
|
limits:
|
|
memory: "256Mi"
|
|
cpu: "500m"
|
|
volumeMounts:
|
|
- name: data
|
|
mountPath: /data
|
|
- name: config
|
|
mountPath: /etc/felhom-hub
|
|
- name: wg-endpoint-ssh
|
|
mountPath: /etc/hub-secrets/wg-endpoint-ssh
|
|
readOnly: true
|
|
- name: tenantsync
|
|
mountPath: /etc/hub-secrets/tenantsync
|
|
readOnly: true
|
|
- name: agent-poke
|
|
mountPath: /etc/hub-secrets/agent-poke
|
|
readOnly: true
|
|
livenessProbe:
|
|
httpGet:
|
|
path: /healthz
|
|
port: 8080
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 30
|
|
timeoutSeconds: 5
|
|
readinessProbe:
|
|
httpGet:
|
|
path: /healthz
|
|
port: 8080
|
|
initialDelaySeconds: 3
|
|
periodSeconds: 10
|
|
timeoutSeconds: 3
|
|
volumes:
|
|
- name: data
|
|
persistentVolumeClaim:
|
|
claimName: hub-data
|
|
- name: config
|
|
configMap:
|
|
name: hub-config
|
|
- name: wg-endpoint-ssh
|
|
secret:
|
|
secretName: wg-endpoint-ssh
|
|
optional: true
|
|
items:
|
|
- key: key
|
|
path: key
|
|
mode: 0400
|
|
- name: tenantsync
|
|
secret:
|
|
secretName: tenantsync
|
|
optional: true
|
|
items:
|
|
- key: key
|
|
path: key
|
|
mode: 0400
|
|
- name: agent-poke
|
|
secret:
|
|
secretName: agent-poke
|
|
optional: true
|
|
items:
|
|
- key: key
|
|
path: key
|
|
mode: 0400
|
|
|
|
# =============================================================================
|
|
# SERVICE
|
|
# =============================================================================
|
|
---
|
|
apiVersion: v1
|
|
kind: Service
|
|
metadata:
|
|
name: hub
|
|
namespace: felhom-system
|
|
labels:
|
|
app: hub
|
|
spec:
|
|
selector:
|
|
app: hub
|
|
ports:
|
|
- port: 8080
|
|
targetPort: 8080
|
|
name: http
|
|
|
|
# =============================================================================
|
|
# INGRESS — hub.felhom.eu
|
|
# =============================================================================
|
|
---
|
|
apiVersion: networking.k8s.io/v1
|
|
kind: Ingress
|
|
metadata:
|
|
name: hub
|
|
namespace: felhom-system
|
|
annotations:
|
|
cert-manager.io/cluster-issuer: letsencrypt-prod
|
|
nginx.ingress.kubernetes.io/proxy-body-size: "2m"
|
|
# Geo-restrict to Hungary (operator-only dashboard)
|
|
# NOTE: /api/v1/report must also be reachable — all customers are in HU
|
|
nginx.ingress.kubernetes.io/configuration-snippet: |
|
|
set $geo_allowed 0;
|
|
if ($remote_addr ~ "^192\.168\.") { set $geo_allowed 1; }
|
|
if ($remote_addr ~ "^10\.") { set $geo_allowed 1; }
|
|
if ($geoip2_country_code = "HU") { set $geo_allowed 1; }
|
|
if ($geo_allowed = 0) {
|
|
return 403 "Access restricted to Hungary";
|
|
}
|
|
spec:
|
|
ingressClassName: nginx-internal
|
|
tls:
|
|
- hosts:
|
|
- hub.felhom.eu
|
|
secretName: hub-felhom-eu-tls
|
|
rules:
|
|
- host: hub.felhom.eu
|
|
http:
|
|
paths:
|
|
- path: /
|
|
pathType: Prefix
|
|
backend:
|
|
service:
|
|
name: hub
|
|
port:
|
|
number: 8080 |