From cea8502f0bbdcee2cc4d006dbb3db760fcddb9e3 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Mon, 5 Oct 2026 15:41:08 +0200 Subject: [PATCH] scripts/hub-db-backup: DooPlex push (02:30) + weekly restore test (Sun 04:30) of the hub DB to ep0 (R-173, R-231); 15 tests, red-proofs P1-P9; Part A/B evidence Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- .../partA/step1-expansion-failure.txt | 10 + .../partA/step1-im-restart.txt | 100 +++++++ .../partA/step1-labels.txt | 22 ++ .../partA/step1-offline-expansion.txt | 16 ++ .../partB/ep0-after.txt | 44 +++ .../partB/ep0-before.txt | 53 ++++ .../partB/ep0-changes.txt | 46 ++++ .../partC/red-proof.txt | 76 +++++ scripts/hub-db-backup/felhom-hub-db-backup | 65 +++++ .../felhom-hub-db-backup.service | 19 ++ .../hub-db-backup/felhom-hub-db-backup.timer | 11 + .../hub-db-backup/felhom-hub-db-restore-test | 64 +++++ .../felhom-hub-db-restore-test.service | 19 ++ .../felhom-hub-db-restore-test.timer | 11 + scripts/hub-db-backup/install.sh | 27 ++ scripts/hub-db-backup/test_hub_db_backup.py | 259 ++++++++++++++++++ 16 files changed, 842 insertions(+) create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partA/step1-expansion-failure.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partA/step1-im-restart.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partA/step1-labels.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partA/step1-offline-expansion.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-after.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-before.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-changes.txt create mode 100644 documentation/audits/hub-db-offsite-2026-10-05/partC/red-proof.txt create mode 100755 scripts/hub-db-backup/felhom-hub-db-backup create mode 100644 scripts/hub-db-backup/felhom-hub-db-backup.service create mode 100644 scripts/hub-db-backup/felhom-hub-db-backup.timer create mode 100755 scripts/hub-db-backup/felhom-hub-db-restore-test create mode 100644 scripts/hub-db-backup/felhom-hub-db-restore-test.service create mode 100644 scripts/hub-db-backup/felhom-hub-db-restore-test.timer create mode 100755 scripts/hub-db-backup/install.sh create mode 100644 scripts/hub-db-backup/test_hub_db_backup.py diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-expansion-failure.txt b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-expansion-failure.txt new file mode 100644 index 00000000..f4a16282 --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-expansion-failure.txt @@ -0,0 +1,10 @@ +## Longhorn online expansion FAILS — 2026-10-05T12:37:44Z +LAST SEEN TYPE REASON OBJECT MESSAGE +7m34s Normal ExternalExpanding persistentvolumeclaim/hub-data waiting for an external controller to expand this PVC +6m4s Warning VolumeResizeFailed persistentvolumeclaim/hub-data resize volume "pvc-486c9809-4672-4b56-b70e-0bf01d0c3628" by resizer "driver.longhorn.io" failed: rpc error: code = DeadlineExceeded desc = volume pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 expansion from existing capacity 1073741824 to requested capacity 2147483648 failed +4s Normal Resizing persistentvolumeclaim/hub-data External resizer is resizing volume pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 +4s Warning VolumeResizeFailed persistentvolumeclaim/hub-data resize volume "pvc-486c9809-4672-4b56-b70e-0bf01d0c3628" by resizer "driver.longhorn.io" failed: rpc error: code = DeadlineExceeded desc = volume pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 expansion from existing capacity 2147483648 to requested capacity 2147483648 failed +engine pvc-486c9809-4672-4b56-b70e-0bf01d0c3628-e-0 spec=2147483648 current=1073741824 +[pvc-486c9809-4672-4b56-b70e-0bf01d0c3628-e-0] time="2026-10-05T12:37:45.150634697Z" level=error msg="Failed to expand the frontend" func="controller.(*Controller).Expand.func1.1" file="control.go:332" error="device pvc-486c9809-4672-4b56-b70e-0bf01d0c3628: fail to refresh iSCSI initiator: failed to execute: /usr/bin/nsenter [nsenter --mount=/host/proc/196610/ns/mnt --net=/host/proc/196610/ns/net iscsiadm --version], output , stderr nsenter: cannot open /host/proc/196610/ns/mnt: No such file or directory: exit status 1" volume=pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 + 996780 744880 235516 76% /data +instance-manager-703c624848c15b89cb295ca89db85615 1/1 Running 0 116d 10.42.0.229 dooplex diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-im-restart.txt b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-im-restart.txt new file mode 100644 index 00000000..125fbf4f --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-im-restart.txt @@ -0,0 +1,100 @@ +## BEFORE 2026-10-05T13:10:25Z +pvc-0113bbbf-fd5c-441b-9d58-6943ffa8c133 kisfenyo-filebrowser-data kisfenyo-system attached healthy 524288000 +pvc-0267c9d6-07e4-476c-8866-0731dcd85fff romm-resources arcade-system attached healthy 10737418240 +pvc-0eae2db7-47f1-477b-b9a9-9968c2d49a6d gokapi-config fileshare-system attached healthy 1073741824 +pvc-11b257b2-8edc-4bf6-b846-00979feb195f zipline-data zipline-system attached healthy 10737418240 +pvc-17e3b3cc-9c6a-4d70-85cc-c2ae72e0b4ec plantit-db plantit-system attached healthy 2147483648 +pvc-1ae51535-0b2b-4975-bd9f-e651cc7a1edd nextcloud-postgresql-data nextcloud-system attached healthy 5368709120 +pvc-1c47727a-d5ac-4171-ac36-26c2df3d0d63 nextcloud-nextcloud nextcloud-system attached healthy 10737418240 +pvc-1d4ec717-7ee5-40a0-b093-3a62fd2a81a4 authentik-media auth-system attached healthy 2147483648 +pvc-202e3bf2-9e61-4c59-ae0c-43fa6a42b403 tailscale-state admin-system attached healthy 1073741824 +pvc-22343573-266d-492e-bd0a-37296217b523 postgresql-2 database-system attached healthy 53687091200 +pvc-22c87ddc-afd7-4520-8473-9e4bf2b6425a immich-machine-learning-ssd2 immich-system attached healthy 10737418240 +pvc-2382e7bb-9514-409b-a2cc-e97c9725512a uptimekuma-data-ssd2 uptimekuma-system attached healthy 5368709120 +pvc-297c20e6-4ac0-42c4-9c69-24580a7f12ee act-runner-data gitea-system attached healthy 5368709120 +pvc-2e2e8c4c-4ba3-4dcf-9ed4-78acaf7ef92b sparkyfitness-uploads workout-system attached healthy 5368709120 +pvc-328d0394-1ca7-47be-8300-2dc8a1740235 paperless-redis paperless-system attached healthy 1073741824 +pvc-34a1540e-8b89-4c1d-8465-796bdbfb172f pms-config-plex-plex-media-server-0 mediaserver-system attached healthy 37580963840 +pvc-38e8c624-5717-483b-8da1-613e703e36c8 umami-db-data felhom-system attached healthy 2147483648 +pvc-3ae955c9-9968-4099-8a7b-af2bd87d9014 immich-valkey-ssd2 immich-system attached healthy 1073741824 +pvc-3b23c214-ea15-40f8-aac5-61f783bf5beb plantit-uploads plantit-system attached healthy 5368709120 +pvc-3d3d043a-92e7-4c34-8383-9d08deb96689 crafty-backups crafty-system attached healthy 53687091200 +pvc-42695a34-d921-4007-b5e5-e40aa728e104 code-server-config code-system attached healthy 2147483648 +pvc-459e6d2a-92b8-438e-a339-ff7e40089f3d pihole pihole-system attached healthy 5368709120 +pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 hub-data felhom-system attached healthy 2147483648 +pvc-49844e6d-dbe7-427a-a021-8a36a5996478 onlyoffice-logs office-system attached healthy 2147483648 +pvc-4b9796b8-4c76-4c12-b20e-bce7eb720baf outline-redis outline-system attached healthy 1073741824 +pvc-4c75264d-5ca8-4298-9916-e84fe7130b3b onlyoffice-data office-system attached healthy 5368709120 +pvc-4e3fd336-9f09-4251-bd90-cdc259ef739e wanderer-db wanderer-system attached healthy 5368709120 +pvc-4e4aba1e-b414-4016-8895-ecdf0fb04a6b grafana-data mon-system attached healthy 8589934592 +pvc-4fdd6da2-e795-4fa9-9e36-96f212c7018b audiobookshelf-config audiobookshelf-system attached healthy 5368709120 +pvc-59667477-c72a-49ea-bf83-b51a672cc727 crafty-import crafty-system attached healthy 5368709120 +pvc-5acabaa7-6017-4793-b45e-f1f4fd80ab97 code-server-local code-system attached healthy 5368709120 +pvc-5c436be9-dbec-4b02-ac26-cb27ce341815 onlyoffice-lib office-system attached healthy 5368709120 +pvc-5cdbcb85-dfb8-480b-9dc3-2e5d03628dd1 dev-jarr-postgres jarrs-system attached healthy 5368709120 +pvc-5fbd3bbc-fb14-406d-954c-1a232ce313ce gitea-data gitea-system attached healthy 53687091200 +pvc-63f3b34f-3ac6-41a5-8790-3a27db0b5a2b revfulop-calendar-data orsi-system attached healthy 268435456 +pvc-69577be6-c9f1-40f7-8edf-89ce7cac7c5b prometheus-data-ssd1 mon-system attached healthy 32212254720 +pvc-6db10505-5d7d-4997-9f7a-6df13cf847ba glance-helper-data glance-system attached healthy 209715200 +pvc-6ed74c1a-3f10-4638-9baa-9330cbff980d opengist-data opengist-system attached healthy 5368709120 +pvc-71cbab32-86f9-40a5-8f3b-220ad626c492 vaultwarden-data vaultwarden-system attached healthy 5368709120 +pvc-724a5306-f7a4-4b05-ac28-630597b9b8e2 gokapi-data fileshare-system attached healthy 53687091200 +pvc-72c77aad-fdc4-4551-a79a-7142fae94fa9 serverpackcreator-mongo-data crafty-system attached healthy 42949672960 +pvc-79478678-07e4-4140-aef1-771ff763ab3f romm-config arcade-system attached healthy 1073741824 +pvc-7d6dec2c-0cda-43c7-ad1d-44ffebcb8cb6 seerr-config-pvc servarr-system attached healthy 1073741824 +pvc-7eb1f195-6a35-46c3-bbb1-5bab48209624 prowlarr-config-pvc servarr-system attached healthy 4294967296 +pvc-7fc61dcd-52f0-45b4-97f1-9c82b28ab714 sparkyfitness-postgres workout-system attached healthy 5368709120 +pvc-83701410-61f7-4d16-a8f6-158668110685 calcom-redis booking-system attached healthy 1073741824 +pvc-8401df55-4668-40e3-97d2-d53e2c1090f1 tautulli-config-pvc servarr-system attached healthy 2147483648 +pvc-85be87dc-3351-4459-9f82-55f8cc598a2f actualbudget-data actualbudget-system attached healthy 5368709120 +pvc-86a960ec-ac7f-41a1-b595-00deac982ff5 serverpackcreator-data crafty-system attached healthy 42949672960 +pvc-88b4d7b4-cb5d-48e0-871e-9e150872d0bd calibre-web-automated-config calibre-system attached healthy 10737418240 +pvc-8931072a-e34e-4333-9ab5-9629b6078505 upsnap-data admin-system attached healthy 1073741824 +pvc-8af636d2-2d00-46ec-9947-7b3143563c16 dev-jarr-redis jarrs-system attached healthy 1073741824 +pvc-8ed5eea5-67d4-46fc-9806-6b8fdd61e42e filebrowser-files felhom-system attached healthy 1073741824 +pvc-8fcb5bc5-865a-4fc3-8bac-b808b1233dbd tandoor-staticfiles tandoor-system attached healthy 1073741824 +pvc-9a76846d-b1cf-4928-b950-2db5500531e1 wanderer-meilisearch wanderer-system attached healthy 5368709120 +pvc-9ca6a486-40bb-46ac-ac48-bd49d4399f4b sonarr-config-pvc servarr-system attached healthy 8589934592 +pvc-a9dd6c7c-19eb-4a0a-935d-d2c30545c378 romm-db arcade-system attached healthy 2147483648 +pvc-b0e94be5-7cd8-4a4c-9295-901614e9eeb7 crafty-app-config crafty-system attached healthy 2147483648 +pvc-b10da48e-6fcc-42ef-8414-d5d617028d68 qbittorrent-config-pvc servarr-system attached healthy 1073741824 +pvc-b359d7de-1c7c-4b6a-9f6d-6776c2f876ef minecraft-tlauncher-data crafty-system attached healthy 21474836480 +pvc-b8959b5c-6b88-4e65-81f2-0eabbf4a6034 privatebin-data privatebin-system attached healthy 2147483648 +pvc-bd481dfa-68d6-4f7a-a634-4d3b76dbf57b recipe-importer-data tandoor-system attached healthy 134217728 +pvc-be9d7813-1881-462e-8951-1bd1315f7dc5 filebrowser-db felhom-system attached healthy 104857600 +pvc-beb0ad8a-755a-4152-a0af-f7367b32b19c paperless-config paperless-system attached healthy 10737418240 +pvc-c4dfbc43-2c35-4ef0-9cfe-39b3dc885dae radarrkids-config-pvc servarr-system attached healthy 3221225472 +pvc-c6b9ab63-24b8-46f2-b52a-06da86393de8 filebrowser-config web-system attached healthy 104857600 +pvc-c8c0d0f1-c9ea-4ed3-9e22-1e254a026c49 radarr-config-pvc servarr-system attached healthy 8589934592 +pvc-c911715e-56ee-4d97-836e-89efc448ad9d immich-postgres-ssd2 immich-system attached healthy 10737418240 +pvc-cbfd55b0-4f44-487d-90fd-9c80b916ced7 crafty-servers crafty-system attached healthy 53687091200 +pvc-d2ea5b66-d251-4467-9829-14654fab7a2f bookstack-mariadb-ssd2 bookstack-system attached healthy 5368709120 +pvc-d9f09398-17cf-44b0-b3a2-9475bc773fec adventurelog-postgres adventurelog-system attached healthy 5368709120 +pvc-e2151b14-2b24-4128-993a-c392c8d9706a code-server-workspace code-system attached healthy 21474836480 +pvc-e3b26fdc-fd52-4fdf-9fa2-1d1ecfd029e4 termix-data termix-system attached healthy 5368709120 +pvc-e57c220f-bc92-4815-89c9-0dee35eab3c3 orsi-filebrowser-config orsi-system attached healthy 104857600 +pvc-e5b8f824-7720-454c-b546-a098608df24a alertmanager-data mon-system attached healthy 1073741824 +pvc-ea977627-fb0a-4dee-8678-9ceceea448da bookstack-config-ssd2 bookstack-system attached healthy 5368709120 +pvc-ef87a570-5d8d-45e5-9bbf-c700cbee1508 onlyoffice-db office-system attached healthy 5368709120 +pvc-fc3d2f7f-6569-473c-acb4-002c613e5f18 filebrowser-data web-system attached healthy 5368709120 +## settings +auto-delete-pod-when-volume-detached-unexpectedly=true +auto-salvage=true +## pods not Running/Completed +## instance managers +instance-manager-703c624848c15b89cb295ca89db85615 1/1 Running 0 116d 10.42.0.229 dooplex +instance-manager-703c624848c15b89cb295ca89db85615 aio running +## 2026-10-05T13:20:10Z DELETE instance-manager pod +pod "instance-manager-703c624848c15b89cb295ca89db85615" deleted +13:20:21Z instance-manager: 1/1 Running +instance-manager-703c624848c15b89cb295ca89db85615 1/1 Running 0 1s 10.42.0.235 dooplex +13:22:00Z not attached+healthy: 0 +## AFTER 2026-10-05T13:35:41Z +volumes not attached+healthy: 0 +pods not Running/Completed: +Running but not ready: +zipline: was CrashLoop on :latest=4.8.0 (prisma->drizzle refusal); pinned 4.7.0 in homelab-manifests 90f60e4+4c8ec7a; running, 4 migrations applied +## hub-data +capacity=2Gi conditions= +engine current=2147483648 + 2028392 745616 1266392 37% /data diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-labels.txt b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-labels.txt new file mode 100644 index 00000000..59c830bc --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-labels.txt @@ -0,0 +1,22 @@ +## BEFORE sync 2026-10-05T12:29:57Z +pvc labels={"app":"hub","recurring-job-group.longhorn.io/default":"disabled"} request=1Gi capacity=1Gi +volume labels={"backup-target":"default","longhornvolume":"pvc-486c9809-4672-4b56-b70e-0bf01d0c3628","recurring-job-group.longhorn.io/default":"enabled","setting.longhorn.io/remove-snapshots-during-filesystem-trim":"ignored","setting.longhorn.io/replica-auto-balance":"ignored","setting.longhorn.io/snapshot-data-integrity":"ignored"} size=1073741824 robustness=healthy +image=gitea.dooplex.hu/admin/felhom-hub:0.135.0 +## AFTER sync 2026-10-05T12:30:55Z +pvc labels={"app":"hub","recurring-job-group.longhorn.io/default":"enabled"} request=2Gi capacity=1Gi conditions=[{"lastProbeTime":null,"lastTransitionTime":"2026-10-05T12:30:10Z","status":"True","type":"Resizing"}] +volume labels={"backup-target":"default","longhornvolume":"pvc-486c9809-4672-4b56-b70e-0bf01d0c3628","recurring-job-group.longhorn.io/default":"enabled","setting.longhorn.io/remove-snapshots-during-filesystem-trim":"ignored","setting.longhorn.io/replica-auto-balance":"ignored","setting.longhorn.io/snapshot-data-integrity":"ignored"} size=2147483648 robustness=healthy +image=gitea.dooplex.hu/admin/felhom-hub:0.136.0 +## hub log (snapshot + version lines) +2026/10/05 14:30:14 [INFO] felhom-hub 0.136.0 starting +2026/10/05 14:30:14 [INFO] Default controller-version floor: 0.120.0 +2026/10/05 14:30:15 [INFO] Gitea artifact browser enabled (Day-0 version dropdowns) via http://gitea.gitea-system.svc.cluster.local:3000 +2026/10/05 14:30:15 [INFO] Registry version checker started (every 6h) +2026/10/05 14:30:16 [DEBUG] Registry version check: latest = 0.296.0 +2026/10/05 14:30:37 [INFO] db-snapshot: next run at 2026-10-06 02:00 CEST (in 11h29m23s) +## /data +Filesystem 1K-blocks Used Available Use% Mounted on +/dev/longhorn/pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 + 996780 507808 472588 52% /data +total 124836 +-rw-r--r-- 1 root root 127840256 Oct 5 14:30 hub-20261005T123037Z.db.tmp +-rw-r--r-- 1 root root 1024 Oct 5 14:30 hub-20261005T123037Z.db.tmp-journal diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-offline-expansion.txt b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-offline-expansion.txt new file mode 100644 index 00000000..780732d0 --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partA/step1-offline-expansion.txt @@ -0,0 +1,16 @@ +## 2026-10-05T12:53:18Z scale hub to 0 +deployment.apps/hub scaled +12:56:30Z volume state=attached +13:01:42Z attached expansionRequired=true +engine spec=2147483648 current=1073741824 +30m Warning VolumeResizeFailed persistentvolumeclaim/hub-data resize volume "pvc-486c9809-4672-4b56-b70e-0bf01d0c3628" by resizer "driver.longhorn.io" failed: rpc error: code = DeadlineExceeded desc = volume pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 expansion from existing capacity 1073741824 to requested capacity 2147483648 failed +2s Normal Resizing persistentvolumeclaim/hub-data External resizer is resizing volume pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 +2s Warning VolumeResizeFailed persistentvolumeclaim/hub-data resize volume "pvc-486c9809-4672-4b56-b70e-0bf01d0c3628" by resizer "driver.longhorn.io" failed: rpc error: code = DeadlineExceeded desc = volume pvc-486c9809-4672-4b56-b70e-0bf01d0c3628 expansion from existing capacity 2147483648 to requested capacity 2147483648 failed +## 2026-10-05T13:01:50Z volume never detached — attachment tickets: +{"volume-expansion-controller-pvc-486c9809-4672-4b56-b70e-0bf01d0c3628":{"generation":0,"id":"volume-expansion-controller-pvc-486c9809-4672-4b56-b70e-0bf01d0c3628","nodeID":"dooplex","parameters":{"disableFrontend":"false"},"type":"volume-expansion-controller"}} +## 2026-10-05T13:01:51Z scale hub to 1 +deployment.apps/hub scaled +Waiting for deployment "hub" rollout to finish: 0 of 1 updated replicas are available... +deployment "hub" successfully rolled out +NAME READY STATUS RESTARTS AGE +hub-586df4748f-ppn4x 1/1 Running 0 54s diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-after.txt b/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-after.txt new file mode 100644 index 00000000..04f0e0c7 --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-after.txt @@ -0,0 +1,44 @@ +## 2026-10-05T13:37:43Z ACLs +## ep0 AFTER 2026-10-05T13:37:43Z +## prune jobs +[ + { + "comment": "R-82 retention keep-last=2, server-side (box tokens are write-only)", + "id": "prune-demo-felhom", + "keep-last": 2, + "max-depth": 0, + "ns": "demo-felhom", + "schedule": "03:30", + "store": "felhom-offsite" + }, + { + "comment": "R-82 retention keep-last=2, server-side (box tokens are write-only)", + "id": "prune-demo-hp", + "keep-last": 2, + "max-depth": 0, + "ns": "demo-hp", + "schedule": "03:30", + "store": "felhom-offsite" + }, + { + "comment": "R-173 hub DB copies; operator ns only", + "id": "prune-operator-hubdb", + "keep-daily": 14, + "keep-weekly": 8, + "max-depth": 0, + "ns": "operator", + "schedule": "03:45", + "store": "felhom-offsite" + } +] +## gc +felhom-offsite sun 04:30 +## users +['felhom@pbs', 'dooplex-hub@pbs', 'root@pam'] +## tokens (ids only) +['dooplex-hub@pbs!restore', 'dooplex-hub@pbs!push'] +## ACLs naming dooplex-hub +/datastore/felhom-offsite/operator dooplex-hub@pbs!restore DatastoreReader True +/datastore/felhom-offsite/operator dooplex-hub@pbs DatastoreReader True +/datastore/felhom-offsite/operator dooplex-hub@pbs DatastoreBackup True +/datastore/felhom-offsite/operator dooplex-hub@pbs!push DatastoreBackup True diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-before.txt b/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-before.txt new file mode 100644 index 00000000..682f9a7b --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-before.txt @@ -0,0 +1,53 @@ +## ep0 BEFORE 2026-10-05T13:36:16Z host=felhom-hetzner +proxmox-backup-server 4.2.8-1 running version: 4.2.5 +## prune jobs +[ + { + "comment": "R-82 retention keep-last=2, server-side (box tokens are write-only)", + "id": "prune-demo-hp", + "keep-last": 2, + "max-depth": 0, + "ns": "demo-hp", + "schedule": "03:30", + "store": "felhom-offsite" + }, + { + "comment": "R-82 retention keep-last=2, server-side (box tokens are write-only)", + "id": "prune-demo-felhom", + "keep-last": 2, + "max-depth": 0, + "ns": "demo-felhom", + "schedule": "03:30", + "store": "felhom-offsite" + } +] +## garbage-collection +[ + { + "cache-stats": { + "hits": 6726, + "misses": 8569 + }, + "disk-bytes": 11794377956, + "disk-chunks": 8570, + "duration": 44, + "index-data-bytes": 51257596032, + "index-file-count": 8, + "last-run-endtime": 1791088244, + "last-run-state": "OK", + "next-run": 1791693000, + "pending-bytes": 0, + "pending-chunks": 0, + "removed-bad": 0, + "removed-bytes": 8871579815, + "removed-chunks": 7465, + "schedule": "sun 04:30", + "still-bad": 0, + "store": "felhom-offsite", + "upid": "UPID:felhom-hetzner:00086AE7:07B20B30:000002A6:6AC1D648:garbage_collection:felhom\\x2doffsite:root@pam:" + } +] +## users (id only) +['root@pam', 'felhom@pbs'] +## operator ns exists? +False diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-changes.txt b/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-changes.txt new file mode 100644 index 00000000..43ee6d90 --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partB/ep0-changes.txt @@ -0,0 +1,46 @@ +## 2026-10-05T13:37:04Z changes +user created + +thread 'main' (2343121) panicked at /usr/share/cargo/registry/proxmox-router-3.2.8/src/cli/text_table.rs:812:13: +not implemented +note: run with `RUST_BACKTRACE=1` environment variable to display a backtrace +Error: parameter verification failed - 'schedule': unable to parse calendar event at 'daily' - Context("weekday") +Usage: proxmox-backup-manager prune-job create --schedule --store [OPTIONS] + + + Job ID. + + --schedule + Run prune job at specified schedule. + + --store + Datastore name. + +Optional parameters: + + --comment + Comment. + --disable (default=false) + Disable this job. + --keep-daily (1 - N) + Number of daily backups to keep. + --keep-hourly (1 - N) + Number of hourly backups to keep. + --keep-last (1 - N) + Number of backups to keep. + --keep-monthly (1 - N) + Number of monthly backups to keep. + --keep-weekly (1 - N) + Number of weekly backups to keep. + --keep-yearly (1 - N) + Number of yearly backups to keep. + --max-depth (0 - 7) + How many levels of namespaces should be operated on (0 == no + recursion, empty == automatic full recursion, namespace depths + reduce maximum allowed value) + --ns + Namespace. +## 2026-10-05T13:37:12Z check + retry +operator ns exists: True +Prune job created: prune-operator-hubdb +prune job created diff --git a/documentation/audits/hub-db-offsite-2026-10-05/partC/red-proof.txt b/documentation/audits/hub-db-offsite-2026-10-05/partC/red-proof.txt new file mode 100644 index 00000000..24af2c7e --- /dev/null +++ b/documentation/audits/hub-db-offsite-2026-10-05/partC/red-proof.txt @@ -0,0 +1,76 @@ +### P1 push: integrity_check ignored +test_corrupt_copy_is_never_pushed (__main__.Push.test_corrupt_copy_is_never_pushed) ... FAIL +FAIL: test_corrupt_copy_is_never_pushed (__main__.Push.test_corrupt_copy_is_never_pushed) +AssertionError: 0 == 0 +Ran 1 test in 0.205s +FAILED (failures=1) + +### P2 push: snapshot age ignored +test_stale_snapshot_is_not_pushed_again (__main__.Push.test_stale_snapshot_is_not_pushed_again) ... FAIL +FAIL: test_stale_snapshot_is_not_pushed_again (__main__.Push.test_stale_snapshot_is_not_pushed_again) +AssertionError: 0 == 0 +Ran 1 test in 0.201s +FAILED (failures=1) + +### P3 push: success signal after a failed push +test_failed_push_writes_no_signal (__main__.Push.test_failed_push_writes_no_signal) ... FAIL +FAIL: test_failed_push_writes_no_signal (__main__.Push.test_failed_push_writes_no_signal) +AssertionError: 0 == 0 +Ran 1 test in 0.206s +FAILED (failures=1) + +### P4 push: size check removed +test_truncated_copy_is_not_pushed (__main__.Push.test_truncated_copy_is_not_pushed) ... ok +Ran 1 test in 0.156s +OK + +### P5 push: no encryption flag +test_happy_path_pushes_encrypted_to_operator_and_writes_signal (__main__.Push.test_happy_path_pushes_encrypted_to_operator_and_writes_signal) ... ERROR +ERROR: test_happy_path_pushes_encrypted_to_operator_and_writes_signal (__main__.Push.test_happy_path_pushes_encrypted_to_operator_and_writes_signal) +Ran 1 test in 0.211s +FAILED (errors=1) + +### P6 restore: readable console passwords allowed +test_readable_console_password_fails (__main__.RestoreTest.test_readable_console_password_fails) ... FAIL +FAIL: test_readable_console_password_fails (__main__.RestoreTest.test_readable_console_password_fails) +AssertionError: 0 == 0 +Ran 1 test in 0.357s +FAILED (failures=1) + +### P7 restore: uses the push token +test_happy_path_writes_signal_with_the_read_only_token (__main__.RestoreTest.test_happy_path_writes_signal_with_the_read_only_token) ... FAIL +FAIL: test_happy_path_writes_signal_with_the_read_only_token (__main__.RestoreTest.test_happy_path_writes_signal_with_the_read_only_token) +AssertionError: False is not true : {'argv': ['snapshot', 'list', 'host/dooplex-hub', '--ns', 'operator', '--output-format', 'json', '--repository', 'dooplex-hub@pbs!restore@127.0.0.1:18007:felhom-offsite'], 'pw': '/tmp/tmp0xabeyiz/conf/token-push', 'fp': 'aa:bb'} +Ran 1 test in 0.371s +FAILED (failures=1) + +### P8 restore: copy age ignored +test_old_copy_fails (__main__.RestoreTest.test_old_copy_fails) ... FAIL +FAIL: test_old_copy_fails (__main__.RestoreTest.test_old_copy_fails) +AssertionError: 0 == 0 +Ran 1 test in 0.352s +FAILED (failures=1) + +### P9 push: staged plaintext left behind +test_happy_path_pushes_encrypted_to_operator_and_writes_signal (__main__.Push.test_happy_path_pushes_encrypted_to_operator_and_writes_signal) ... FAIL +FAIL: test_happy_path_pushes_encrypted_to_operator_and_writes_signal (__main__.Push.test_happy_path_pushes_encrypted_to_operator_and_writes_signal) +AssertionError: Lists differ: ['hub.db'] != [] +Ran 1 test in 0.200s +FAILED (failures=1) + +P4 and P5 above: P4 did NOT convict (masked by integrity_check), P5 ERRORED rather than failed. Tests strengthened; re-runs: + +### P4 (re-run) push: size check removed +test_truncated_copy_is_not_pushed (__main__.Push.test_truncated_copy_is_not_pushed) ... FAIL +FAIL: test_truncated_copy_is_not_pushed (__main__.Push.test_truncated_copy_is_not_pushed) +AssertionError: "bytes, the pod's file is" not found in 'felhom-hub-db-backup: FAILED: integrity_check: Error: in prepare, database disk image is malformed (11)\n' +Ran 1 test in 0.146s +FAILED (failures=1) + +### P5 (re-run) push: no encryption flag +test_happy_path_pushes_encrypted_to_operator_and_writes_signal (__main__.Push.test_happy_path_pushes_encrypted_to_operator_and_writes_signal) ... FAIL +FAIL: test_happy_path_pushes_encrypted_to_operator_and_writes_signal (__main__.Push.test_happy_path_pushes_encrypted_to_operator_and_writes_signal) +AssertionError: '--crypt-mode' not found in ['backup', 'hubdb.pxar:/tmp/tmpq980kr66/state/stage', '--ns', 'operator', '--backup-type', 'host', '--backup-id', 'dooplex-hub', '--keyfile', '/tmp/tmpq980kr66/conf/enc.key', '--repository', 'dooplex-hub@pbs!push@127.0.0.1:18007:felhom-offsite'] : push without --crypt-mode +Ran 1 test in 0.200s +FAILED (failures=1) + diff --git a/scripts/hub-db-backup/felhom-hub-db-backup b/scripts/hub-db-backup/felhom-hub-db-backup new file mode 100755 index 00000000..3a9ee52a --- /dev/null +++ b/scripts/hub-db-backup/felhom-hub-db-backup @@ -0,0 +1,65 @@ +#!/bin/sh +# felhom-hub-db-backup — push the hub's newest nightly snapshot to ep0's PBS, encrypted (R-173, decision A). +# Runs on DooPlex as root from felhom-hub-db-backup.timer (02:30; the hub writes the snapshot at 02:00). +# Runbook: documentation/runbooks/RUNBOOK-hub-db-offsite-backup.md Step 4. Pinned by test_hub_db_backup.py. +# +# Refuses to push — and so never writes the success signal — when: no snapshot exists; the newest is older than +# MAX_AGE_H (the hub stopped snapshotting: pushing yesterday's copy again would read as success); the copied bytes +# differ in size from the pod's file; PRAGMA integrity_check is not "ok"; the copy holds no hosts. The success +# timestamp is written ONLY after the push returns 0 (CLAUDE.md "presence is not success"). +set -eu +CONF=${FELHOM_HUBBK_CONF:-/etc/felhom-hub-backup} +STATE=${FELHOM_HUBBK_STATE:-/var/lib/felhom-hub-backup} +TEXTFILE_DIR=${FELHOM_HUBBK_TEXTFILE_DIR:-/var/lib/node_exporter/textfile_collector} +MAX_AGE_H=${FELHOM_HUBBK_MAX_AGE_H:-26} +NOW=${FELHOM_HUBBK_NOW:-$(date +%s)} +. "$CONF/env" # PBS_REPOSITORY_PUSH, PBS_FINGERPRINT (no secrets in this file) + +log() { echo "felhom-hub-db-backup: $*"; } +die() { echo "felhom-hub-db-backup: FAILED: $*" >&2; exit 1; } + +umask 077 +STAGE="$STATE/stage" +mkdir -p "$STAGE"; chmod 700 "$STATE" "$STAGE" +rm -f "$STAGE"/* +trap 'if [ -f "$STAGE/hub.db" ]; then shred -u "$STAGE/hub.db" 2>/dev/null || rm -f "$STAGE/hub.db"; fi' EXIT + +SNAP=$(kubectl -n felhom-system exec deploy/hub -- sh -c 'ls -1 /data/snapshots/hub-*.db 2>/dev/null | tail -n 1') || die "listing snapshots in the hub pod" +[ -n "$SNAP" ] || die "no snapshot in the hub pod's /data/snapshots" +NAME=${SNAP##*/} +STAMP=${NAME#hub-}; STAMP=${STAMP%.db} # 20261005T020000Z +case "$STAMP" in [0-9][0-9][0-9][0-9][0-9][0-9][0-9][0-9]T[0-9][0-9][0-9][0-9][0-9][0-9]Z) ;; *) die "unexpected snapshot name $NAME" ;; esac +ISO=$(echo "$STAMP" | sed -E 's/^(....)(..)(..)T(..)(..)(..)Z$/\1-\2-\3T\4:\5:\6Z/') +SNAP_EPOCH=$(date -u -d "$ISO" +%s) || die "cannot parse snapshot time $ISO" +AGE=$((NOW - SNAP_EPOCH)) +[ "$AGE" -le $((MAX_AGE_H * 3600)) ] || die "newest snapshot $NAME is $((AGE / 3600)) h old (limit ${MAX_AGE_H} h) — the hub stopped snapshotting" +log "snapshot $NAME, $((AGE / 60)) min old" + +WANT=$(kubectl -n felhom-system exec deploy/hub -- sh -c "wc -c < '$SNAP'" | tr -d ' \r\n') || die "sizing $NAME" +kubectl -n felhom-system exec deploy/hub -- cat "$SNAP" > "$STAGE/hub.db" || die "copying $NAME out of the pod" +GOT=$(wc -c < "$STAGE/hub.db" | tr -d ' ') +[ "$GOT" = "$WANT" ] || die "copy is $GOT bytes, the pod's file is $WANT" + +IC=$(sqlite3 -readonly "$STAGE/hub.db" 'PRAGMA integrity_check;' 2>&1 | head -n 5) || true +[ "$IC" = "ok" ] || die "integrity_check: $IC" +HOSTS=$(sqlite3 -readonly "$STAGE/hub.db" 'SELECT COUNT(*) FROM hosts;' 2>/dev/null) || die "cannot count hosts" +[ "${HOSTS:-0}" -gt 0 ] || die "the copy holds no hosts" +log "checked: $GOT bytes, integrity ok, $HOSTS host(s)" + +START=$(date +%s) +PBS_PASSWORD_FILE="$CONF/token-push" PBS_FINGERPRINT="$PBS_FINGERPRINT" \ + proxmox-backup-client backup hubdb.pxar:"$STAGE" --ns operator --backup-type host --backup-id dooplex-hub \ + --keyfile "$CONF/enc.key" --crypt-mode encrypt --repository "$PBS_REPOSITORY_PUSH" \ + || die "proxmox-backup-client backup" +log "pushed $NAME to ep0 (ns operator) in $(( $(date +%s) - START )) s" + +TMP="$TEXTFILE_DIR/felhom_hub_db_backup.prom.$$" +{ + echo "# HELP felhom_hub_db_backup_last_success_timestamp_seconds Last successful push of the hub DB snapshot to ep0 (R-173)." + echo "# TYPE felhom_hub_db_backup_last_success_timestamp_seconds gauge" + echo "felhom_hub_db_backup_last_success_timestamp_seconds $(date +%s)" + echo "felhom_hub_db_backup_last_success_bytes $GOT" +} > "$TMP" +chmod 644 "$TMP" +mv "$TMP" "$TEXTFILE_DIR/felhom_hub_db_backup.prom" +log "success signal written" diff --git a/scripts/hub-db-backup/felhom-hub-db-backup.service b/scripts/hub-db-backup/felhom-hub-db-backup.service new file mode 100644 index 00000000..2f63fc5f --- /dev/null +++ b/scripts/hub-db-backup/felhom-hub-db-backup.service @@ -0,0 +1,19 @@ +# Versioned in felhom.eu/scripts/hub-db-backup/ (R-231); installed by install.sh. Runbook: RUNBOOK-hub-db-offsite-backup.md. +[Unit] +Description=Felhom: push the hub DB snapshot to ep0 (R-173) +Wants=network-online.target felhom-ep0-pbs-tunnel.service +After=network-online.target felhom-ep0-pbs-tunnel.service + +[Service] +Type=oneshot +ExecStart=/usr/local/sbin/felhom-hub-db-backup +Environment=HOME=/var/lib/felhom-hub-backup KUBECONFIG=/etc/rancher/k3s/k3s.yaml +UMask=0077 +TimeoutStartSec=45min +Nice=10 +IOSchedulingClass=idle +PrivateTmp=yes +ProtectSystem=strict +ProtectHome=yes +ReadWritePaths=/var/lib/felhom-hub-backup /var/lib/node_exporter/textfile_collector +NoNewPrivileges=yes diff --git a/scripts/hub-db-backup/felhom-hub-db-backup.timer b/scripts/hub-db-backup/felhom-hub-db-backup.timer new file mode 100644 index 00000000..004726bd --- /dev/null +++ b/scripts/hub-db-backup/felhom-hub-db-backup.timer @@ -0,0 +1,11 @@ +# Versioned in felhom.eu/scripts/hub-db-backup/ (R-231); installed by install.sh. +[Unit] +Description=Felhom: push the hub DB snapshot to ep0 (R-173) — schedule + +[Timer] +OnCalendar=*-*-* 02:30:00 +Persistent=true +RandomizedDelaySec=2min + +[Install] +WantedBy=timers.target diff --git a/scripts/hub-db-backup/felhom-hub-db-restore-test b/scripts/hub-db-backup/felhom-hub-db-restore-test new file mode 100755 index 00000000..8e4e3021 --- /dev/null +++ b/scripts/hub-db-backup/felhom-hub-db-restore-test @@ -0,0 +1,64 @@ +#!/bin/sh +# felhom-hub-db-restore-test — restore the newest hub DB copy from ep0 with the READ-ONLY token and check it (R-173). +# Runs on DooPlex as root from felhom-hub-db-restore-test.timer (Sun 04:30). Runbook Step 5. Pinned by +# test_hub_db_backup.py. +# +# The success timestamp is written ONLY when: the newest copy on ep0 is at most MAX_AGE_H old; it restores and +# decrypts; PRAGMA integrity_check is "ok"; it holds at least one host; and NO console password is stored readable +# (every non-empty host_recovery.secret starts with "enc:v1:", the hub's seal, 05 §16.2). +set -eu +CONF=${FELHOM_HUBBK_CONF:-/etc/felhom-hub-backup} +STATE=${FELHOM_HUBBK_STATE:-/var/lib/felhom-hub-backup} +TEXTFILE_DIR=${FELHOM_HUBBK_TEXTFILE_DIR:-/var/lib/node_exporter/textfile_collector} +MAX_AGE_H=${FELHOM_HUBBK_RESTORE_MAX_AGE_H:-50} +NOW=${FELHOM_HUBBK_NOW:-$(date +%s)} +. "$CONF/env" # PBS_REPOSITORY_RESTORE, PBS_FINGERPRINT + +log() { echo "felhom-hub-db-restore-test: $*"; } +die() { echo "felhom-hub-db-restore-test: FAILED: $*" >&2; exit 1; } + +umask 077 +mkdir -p "$STATE"; chmod 700 "$STATE" +T=$(mktemp -d "$STATE/restore.XXXXXX") +trap 'find "$T" -type f -exec shred -u {} + 2>/dev/null; rm -rf "$T"' EXIT +export PBS_PASSWORD_FILE="$CONF/token-restore" PBS_FINGERPRINT + +LIST=$(proxmox-backup-client snapshot list host/dooplex-hub --ns operator --output-format json --repository "$PBS_REPOSITORY_RESTORE") \ + || die "listing snapshots on ep0" +NEWEST=$(printf '%s' "$LIST" | python3 -c ' +import json, sys +s = [x for x in json.load(sys.stdin) if x.get("backup-type") == "host" and x.get("backup-id") == "dooplex-hub"] +if s: + n = max(s, key=lambda x: x["backup-time"]) + print(n["backup-time"]) +') || die "reading the snapshot list" +[ -n "$NEWEST" ] || die "no hub DB copy on ep0" +AGE=$((NOW - NEWEST)) +[ "$AGE" -le $((MAX_AGE_H * 3600)) ] || die "newest copy on ep0 is $((AGE / 3600)) h old (limit ${MAX_AGE_H} h)" +SNAPSHOT="host/dooplex-hub/$(date -u -d "@$NEWEST" +%Y-%m-%dT%H:%M:%SZ)" +log "restoring $SNAPSHOT" + +proxmox-backup-client restore "$SNAPSHOT" hubdb.pxar "$T/out" --ns operator \ + --keyfile "$CONF/enc.key" --repository "$PBS_REPOSITORY_RESTORE" || die "restore of $SNAPSHOT" +DB="$T/out/hub.db" +[ -s "$DB" ] || die "the restored archive holds no hub.db" + +IC=$(sqlite3 -readonly "$DB" 'PRAGMA integrity_check;' 2>&1 | head -n 5) || true +[ "$IC" = "ok" ] || die "integrity_check: $IC" +HOSTS=$(sqlite3 -readonly "$DB" 'SELECT COUNT(*) FROM hosts;' 2>/dev/null) || die "cannot count hosts" +[ "${HOSTS:-0}" -gt 0 ] || die "the restored copy holds no hosts" +PLAIN=$(sqlite3 -readonly "$DB" "SELECT COUNT(*) FROM host_recovery WHERE COALESCE(secret,'') <> '' AND secret NOT LIKE 'enc:v1:%';" 2>/dev/null) \ + || die "cannot read host_recovery" +[ "$PLAIN" -eq 0 ] || die "$PLAIN console password(s) stored readable" +SEALED=$(sqlite3 -readonly "$DB" "SELECT COUNT(*) FROM host_recovery WHERE secret LIKE 'enc:v1:%';") +log "checked: integrity ok, $HOSTS host(s), $SEALED sealed console password(s), 0 readable" + +TMP="$TEXTFILE_DIR/felhom_hub_db_restore.prom.$$" +{ + echo "# HELP felhom_hub_db_restore_test_last_success_timestamp_seconds Last successful restore test of the hub DB copy on ep0 (R-173)." + echo "# TYPE felhom_hub_db_restore_test_last_success_timestamp_seconds gauge" + echo "felhom_hub_db_restore_test_last_success_timestamp_seconds $(date +%s)" +} > "$TMP" +chmod 644 "$TMP" +mv "$TMP" "$TEXTFILE_DIR/felhom_hub_db_restore.prom" +log "success signal written" diff --git a/scripts/hub-db-backup/felhom-hub-db-restore-test.service b/scripts/hub-db-backup/felhom-hub-db-restore-test.service new file mode 100644 index 00000000..8217f282 --- /dev/null +++ b/scripts/hub-db-backup/felhom-hub-db-restore-test.service @@ -0,0 +1,19 @@ +# Versioned in felhom.eu/scripts/hub-db-backup/ (R-231); installed by install.sh. Runbook: RUNBOOK-hub-db-offsite-backup.md. +[Unit] +Description=Felhom: restore-test the hub DB copy on ep0 (R-173) +Wants=network-online.target felhom-ep0-pbs-tunnel.service +After=network-online.target felhom-ep0-pbs-tunnel.service + +[Service] +Type=oneshot +ExecStart=/usr/local/sbin/felhom-hub-db-restore-test +Environment=HOME=/var/lib/felhom-hub-backup KUBECONFIG=/etc/rancher/k3s/k3s.yaml +UMask=0077 +TimeoutStartSec=45min +Nice=10 +IOSchedulingClass=idle +PrivateTmp=yes +ProtectSystem=strict +ProtectHome=yes +ReadWritePaths=/var/lib/felhom-hub-backup /var/lib/node_exporter/textfile_collector +NoNewPrivileges=yes diff --git a/scripts/hub-db-backup/felhom-hub-db-restore-test.timer b/scripts/hub-db-backup/felhom-hub-db-restore-test.timer new file mode 100644 index 00000000..ee3344ad --- /dev/null +++ b/scripts/hub-db-backup/felhom-hub-db-restore-test.timer @@ -0,0 +1,11 @@ +# Versioned in felhom.eu/scripts/hub-db-backup/ (R-231); installed by install.sh. +[Unit] +Description=Felhom: restore-test the hub DB copy on ep0 (R-173) — schedule + +[Timer] +OnCalendar=Sun *-*-* 04:30:00 +Persistent=true +RandomizedDelaySec=2min + +[Install] +WantedBy=timers.target diff --git a/scripts/hub-db-backup/install.sh b/scripts/hub-db-backup/install.sh new file mode 100755 index 00000000..6be087ec --- /dev/null +++ b/scripts/hub-db-backup/install.sh @@ -0,0 +1,27 @@ +#!/bin/sh +# install.sh — install the hub DB off-site backup units on DooPlex (R-173). Root. Idempotent. +# Installs the two scripts and four units, writes /etc/felhom-hub-backup/env (no secrets) when absent, and does NOT +# enable the timers — enable them by hand after the first manual run (runbook Step 7): +# systemctl enable --now felhom-hub-db-backup.timer felhom-hub-db-restore-test.timer +# The tokens (token-push, token-restore) and enc.key are created separately, file to file, never by this script. +set -eu +HERE=$(cd "$(dirname "$0")" && pwd) +[ "$(id -u)" = 0 ] || { echo "install.sh: run as root" >&2; exit 1; } +install -m 0755 "$HERE/felhom-hub-db-backup" /usr/local/sbin/felhom-hub-db-backup +install -m 0755 "$HERE/felhom-hub-db-restore-test" /usr/local/sbin/felhom-hub-db-restore-test +for u in felhom-hub-db-backup.service felhom-hub-db-backup.timer felhom-hub-db-restore-test.service felhom-hub-db-restore-test.timer; do + install -m 0644 "$HERE/$u" "/etc/systemd/system/$u" +done +install -d -m 0700 /etc/felhom-hub-backup /var/lib/felhom-hub-backup +if [ ! -f /etc/felhom-hub-backup/env ]; then + umask 077 + cat > /etc/felhom-hub-backup/env <<'ENV' +# Not secret. The tokens are in token-push / token-restore (0600), the key in enc.key (0600). +PBS_REPOSITORY_PUSH='dooplex-hub@pbs!push@127.0.0.1:18007:felhom-offsite' +PBS_REPOSITORY_RESTORE='dooplex-hub@pbs!restore@127.0.0.1:18007:felhom-offsite' +# ep0's PBS certificate, the same pin DooPlex's PBS remote "ep0" uses (/etc/proxmox-backup/remote.cfg) +PBS_FINGERPRINT='c6:07:28:3f:5b:7b:5a:41:90:28:d7:ca:4f:37:14:70:56:39:2e:2f:0b:71:e8:06:ca:60:4a:d5:56:5f:3c:fd' +ENV +fi +systemctl daemon-reload +echo "install.sh: installed; timers NOT enabled (see the header)" diff --git a/scripts/hub-db-backup/test_hub_db_backup.py b/scripts/hub-db-backup/test_hub_db_backup.py new file mode 100644 index 00000000..6b092071 --- /dev/null +++ b/scripts/hub-db-backup/test_hub_db_backup.py @@ -0,0 +1,259 @@ +#!/usr/bin/env python3 +"""Tests for felhom-hub-db-backup and felhom-hub-db-restore-test (R-173). + +No test reaches the hub, PBS or ep0: `kubectl` and `proxmox-backup-client` are fakes on PATH (the pod's +/data/snapshots is a temp dir; the PBS "server" is a temp dir). `sqlite3`, `date`, `shred` are the real tools. +Each test asserts the CONSEQUENCE: whether a push happened and whether the success signal (the file the alarm reads) +was written. Run: python3 scripts/hub-db-backup/test_hub_db_backup.py +""" +import json +import os +import shutil +import sqlite3 +import stat +import subprocess +import tempfile +import time +import unittest + +HERE = os.path.dirname(os.path.abspath(__file__)) +PUSH = os.path.join(HERE, "felhom-hub-db-backup") +RESTORE = os.path.join(HERE, "felhom-hub-db-restore-test") + +FAKE_KUBECTL = r'''#!/usr/bin/env python3 +import os, subprocess, sys +a = sys.argv[1:] +pod = os.environ["FAKE_POD_DATA"] +i = a.index("--") +cmd = a[i + 1:] +def m(p): return p.replace("/data", pod, 1) +if cmd[:2] == ["sh", "-c"]: + sys.exit(subprocess.call(["sh", "-c", cmd[2].replace("/data", pod)])) +if cmd[0] == "cat": + if os.environ.get("FAKE_TRUNCATE"): + data = open(m(cmd[1]), "rb").read() + sys.stdout.buffer.write(data[: len(data) // 2]); sys.exit(0) + sys.exit(subprocess.call(["cat", m(cmd[1])])) +sys.exit(97) +''' + +FAKE_PBS = r'''#!/usr/bin/env python3 +import json, os, shutil, sys, time +a = sys.argv[1:] +srv = os.environ["FAKE_PBS_DIR"] +open(os.path.join(srv, "calls.log"), "a").write(json.dumps({"argv": a, "pw": os.environ.get("PBS_PASSWORD_FILE", ""), "fp": os.environ.get("PBS_FINGERPRINT", "")}) + "\n") +if a[0] == "backup": + if os.environ.get("FAKE_PBS_FAIL"): sys.exit(1) + src = a[1].split(":", 1)[1] + t = int(time.time()) + d = os.path.join(srv, "snaps", str(t)); os.makedirs(d, exist_ok=True) + shutil.copy(os.path.join(src, "hub.db"), os.path.join(d, "hub.db")) + sys.exit(0) +if a[0] == "snapshot" and a[1] == "list": + base = os.path.join(srv, "snaps") + out = [{"backup-type": "host", "backup-id": "dooplex-hub", "backup-time": int(x)} for x in (os.listdir(base) if os.path.isdir(base) else [])] + print(json.dumps(out)); sys.exit(0) +if a[0] == "restore": + if os.environ.get("FAKE_PBS_FAIL"): sys.exit(1) + import calendar + t = calendar.timegm(time.strptime(a[1].split("/")[-1], "%Y-%m-%dT%H:%M:%SZ")) + os.makedirs(a[3], exist_ok=True) + shutil.copy(os.path.join(srv, "snaps", str(t), "hub.db"), os.path.join(a[3], "hub.db")) + sys.exit(0) +sys.exit(98) +''' + + +def make_db(path, hosts=2, recovery=("enc:v1:abc", "enc:v1:def"), corrupt=False): + db = sqlite3.connect(path) + db.execute("CREATE TABLE hosts (host_id TEXT)") + db.execute("CREATE TABLE host_recovery (host_id TEXT, secret TEXT)") + db.executemany("INSERT INTO hosts VALUES (?)", [("h%d" % i,) for i in range(hosts)]) + db.executemany("INSERT INTO host_recovery VALUES ('h', ?)", [(r,) for r in recovery]) + db.execute("CREATE TABLE filler (x BLOB)") + db.executemany("INSERT INTO filler VALUES (randomblob(3000))", [()] * 40) + db.execute("CREATE INDEX filler_x ON filler(x)") + db.commit(); db.close() + if corrupt: # overwrite a late page (index b-tree) so integrity_check reports errors but the file still opens + size = os.path.getsize(path) + with open(path, "r+b") as f: + f.seek(size - 4096 + 100); f.write(b"\xff" * 2000) + + +def stamp(epoch): + return time.strftime("%Y%m%dT%H%M%SZ", time.gmtime(epoch)) + + +class Base(unittest.TestCase): + def setUp(self): + self.t = tempfile.mkdtemp() + j = lambda *p: os.path.join(self.t, *p) + for d in ("bin", "pod/snapshots", "conf", "state", "textfile", "pbs"): + os.makedirs(j(d), exist_ok=True) + for name, body in (("kubectl", FAKE_KUBECTL), ("proxmox-backup-client", FAKE_PBS)): + p = j("bin", name); open(p, "w").write(body); os.chmod(p, 0o755) + open(j("conf", "env"), "w").write( + "PBS_REPOSITORY_PUSH='dooplex-hub@pbs!push@127.0.0.1:18007:felhom-offsite'\n" + "PBS_REPOSITORY_RESTORE='dooplex-hub@pbs!restore@127.0.0.1:18007:felhom-offsite'\n" + "PBS_FINGERPRINT='aa:bb'\n") + for f in ("token-push", "token-restore", "enc.key"): + open(j("conf", f), "w").write("x") + self.env = dict(os.environ, PATH=j("bin") + ":/usr/bin:/bin", FAKE_POD_DATA=j("pod"), FAKE_PBS_DIR=j("pbs"), + FELHOM_HUBBK_CONF=j("conf"), FELHOM_HUBBK_STATE=j("state"), FELHOM_HUBBK_TEXTFILE_DIR=j("textfile")) + self.j = j + + def tearDown(self): + shutil.rmtree(self.t, ignore_errors=True) + + def snapshot(self, age_s=600, **kw): + p = self.j("pod", "snapshots", "hub-%s.db" % stamp(int(time.time()) - age_s)) + make_db(p, **kw) + return p + + def run_script(self, script, **extra): + env = dict(self.env, **extra) + return subprocess.run([script], env=env, capture_output=True, text=True, timeout=60) + + def pushed(self): + d = self.j("pbs", "snaps") + return sorted(os.listdir(d)) if os.path.isdir(d) else [] + + def signal(self, name): + return os.path.exists(self.j("textfile", name)) + + def calls(self): + p = self.j("pbs", "calls.log") + return [json.loads(l) for l in open(p)] if os.path.exists(p) else [] + + +class Push(Base): + def test_happy_path_pushes_encrypted_to_operator_and_writes_signal(self): + self.snapshot() + r = self.run_script(PUSH) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertEqual(len(self.pushed()), 1) + self.assertTrue(self.signal("felhom_hub_db_backup.prom")) + c = [x for x in self.calls() if x["argv"][0] == "backup"][0] + a = c["argv"] + for flag in ("--ns", "--backup-id", "--crypt-mode", "--keyfile", "--repository"): + self.assertIn(flag, a, "push without %s" % flag) + for flag, val in (("--ns", "operator"), ("--backup-id", "dooplex-hub"), ("--crypt-mode", "encrypt")): + self.assertEqual(a[a.index(flag) + 1], val) + self.assertTrue(a[a.index("--keyfile") + 1].endswith("/enc.key")) + self.assertIn("!push@", a[a.index("--repository") + 1]) + self.assertTrue(c["pw"].endswith("/token-push")) + self.assertNotIn("x", " ".join(a).split()) # the token's value never on the command line + self.assertEqual(os.listdir(self.j("state", "stage")), [], "the staged plaintext copy is left behind") + txt = open(self.j("textfile", "felhom_hub_db_backup.prom")).read() + self.assertIn("felhom_hub_db_backup_last_success_timestamp_seconds ", txt) + + def test_corrupt_copy_is_never_pushed(self): + self.snapshot(corrupt=True) + r = self.run_script(PUSH) + self.assertNotEqual(r.returncode, 0) + self.assertIn("integrity_check", r.stderr) + self.assertEqual(self.pushed(), []) + self.assertFalse(self.signal("felhom_hub_db_backup.prom")) + + def test_stale_snapshot_is_not_pushed_again(self): + self.snapshot(age_s=27 * 3600) + r = self.run_script(PUSH) + self.assertNotEqual(r.returncode, 0) + self.assertIn("stopped snapshotting", r.stderr) + self.assertEqual(self.pushed(), []) + self.assertFalse(self.signal("felhom_hub_db_backup.prom")) + + def test_truncated_copy_is_not_pushed(self): + self.snapshot() + r = self.run_script(PUSH, FAKE_TRUNCATE="1") + self.assertNotEqual(r.returncode, 0) + # the SIZE guard must be the one that refuses (half a file usually fails integrity_check too, which would + # mask a missing size check — red-proof P4's first run did exactly that) + self.assertIn("bytes, the pod's file is", r.stderr) + self.assertEqual(self.pushed(), []) + self.assertFalse(self.signal("felhom_hub_db_backup.prom")) + + def test_failed_push_writes_no_signal(self): + self.snapshot() + r = self.run_script(PUSH, FAKE_PBS_FAIL="1") + self.assertNotEqual(r.returncode, 0) + self.assertFalse(self.signal("felhom_hub_db_backup.prom")) + self.assertEqual(os.listdir(self.j("state", "stage")), []) + + def test_no_snapshot_fails(self): + r = self.run_script(PUSH) + self.assertNotEqual(r.returncode, 0) + self.assertFalse(self.signal("felhom_hub_db_backup.prom")) + + def test_empty_hosts_is_not_pushed(self): + self.snapshot(hosts=0) + r = self.run_script(PUSH) + self.assertNotEqual(r.returncode, 0) + self.assertEqual(self.pushed(), []) + + def test_newest_snapshot_is_the_one_pushed(self): + self.snapshot(age_s=25 * 3600, hosts=1) + self.snapshot(age_s=600, hosts=3) + self.assertEqual(self.run_script(PUSH).returncode, 0) + db = os.path.join(self.j("pbs", "snaps"), self.pushed()[0], "hub.db") + self.assertEqual(sqlite3.connect(db).execute("SELECT COUNT(*) FROM hosts").fetchone()[0], 3) + + +class RestoreTest(Base): + def push_one(self, **kw): + self.snapshot(**kw) + r = self.run_script(PUSH) + self.assertEqual(r.returncode, 0, r.stderr) + + def test_happy_path_writes_signal_with_the_read_only_token(self): + self.push_one() + r = self.run_script(RESTORE) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertTrue(self.signal("felhom_hub_db_restore.prom")) + for c in self.calls(): + if c["argv"][0] in ("restore", "snapshot"): + self.assertTrue(c["pw"].endswith("/token-restore"), c) + self.assertIn("!restore@", c["argv"][c["argv"].index("--repository") + 1]) + self.assertEqual([x for x in os.listdir(self.j("state")) if x.startswith("restore.")], [], "restored copy left behind") + + def test_readable_console_password_fails(self): + self.push_one(recovery=("enc:v1:abc", "hunter2")) + r = self.run_script(RESTORE) + self.assertNotEqual(r.returncode, 0) + self.assertIn("stored readable", r.stderr) + self.assertNotIn("hunter2", r.stderr + r.stdout) + self.assertFalse(self.signal("felhom_hub_db_restore.prom")) + + def test_no_copy_on_ep0_fails(self): + r = self.run_script(RESTORE) + self.assertNotEqual(r.returncode, 0) + self.assertFalse(self.signal("felhom_hub_db_restore.prom")) + + def test_old_copy_fails(self): + self.push_one() + r = self.run_script(RESTORE, FELHOM_HUBBK_NOW=str(int(time.time()) + 51 * 3600)) + self.assertNotEqual(r.returncode, 0) + self.assertFalse(self.signal("felhom_hub_db_restore.prom")) + + def test_failed_restore_fails(self): + self.push_one() + r = self.run_script(RESTORE, FAKE_PBS_FAIL="1") + self.assertNotEqual(r.returncode, 0) + self.assertFalse(self.signal("felhom_hub_db_restore.prom")) + + +class Units(unittest.TestCase): + def read(self, n): + return open(os.path.join(HERE, n)).read() + + def test_schedules(self): + self.assertIn("OnCalendar=*-*-* 02:30:00", self.read("felhom-hub-db-backup.timer")) + self.assertIn("OnCalendar=Sun *-*-* 04:30:00", self.read("felhom-hub-db-restore-test.timer")) + + def test_units_run_the_installed_scripts(self): + self.assertIn("ExecStart=/usr/local/sbin/felhom-hub-db-backup\n", self.read("felhom-hub-db-backup.service")) + self.assertIn("ExecStart=/usr/local/sbin/felhom-hub-db-restore-test\n", self.read("felhom-hub-db-restore-test.service")) + + +if __name__ == "__main__": + unittest.main(verbosity=2)