From f5e11277f6cc464fe5a907606d941c4c87162e58 Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Thu, 24 Sep 2026 16:26:28 +0200 Subject: [PATCH] r672-2026-09-24: evidence (restore test off, 9201 repair, red-proofs, live cases) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- .../r672-2026-09-24/A-restore-test-off.txt | 16 ++ .../audits/r672-2026-09-24/B1-before.txt | 21 +++ .../r672-2026-09-24/B2-9201-apps-before.txt | 23 +++ .../audits/r672-2026-09-24/B3-stop.txt | 7 + .../audits/r672-2026-09-24/B4-fsck.txt | 43 +++++ .../audits/r672-2026-09-24/B5-after.txt | 27 +++ .../audits/r672-2026-09-24/B5-start.txt | 4 + .../B6-box-stopped-redis-apps.log | 162 ++++++++++++++++++ .../r672-2026-09-24/B7-redis-aof-fix.txt | 126 ++++++++++++++ .../r672-2026-09-24/B8-start-redis-apps.txt | 10 ++ .../audits/r672-2026-09-24/B9-hub-online.txt | 1 + .../r672-2026-09-24/C5-r673-vzdump-logs.txt | 40 +++++ .../r672-2026-09-24/C7a-high-margin.txt | 27 +++ .../r672-2026-09-24/C7b-normal-margin.txt | 22 +++ .../redproofs/C-hub-storage-cooldown.txt | 5 + .../redproofs/C-hub-thin-generic.txt | 8 + .../redproofs/C-janitor-no-gate.txt | 6 + .../redproofs/C-retry-no-timer.txt | 8 + .../redproofs/C-skip-dropped.txt | 4 + .../redproofs/C-space-1-no-preflight.txt | 10 ++ .../redproofs/C-space-2-file-size.txt | 6 + .../redproofs/C-thinhigh-no-edge.txt | 5 + 22 files changed, 581 insertions(+) create mode 100644 documentation/audits/r672-2026-09-24/A-restore-test-off.txt create mode 100644 documentation/audits/r672-2026-09-24/B1-before.txt create mode 100644 documentation/audits/r672-2026-09-24/B2-9201-apps-before.txt create mode 100644 documentation/audits/r672-2026-09-24/B3-stop.txt create mode 100644 documentation/audits/r672-2026-09-24/B4-fsck.txt create mode 100644 documentation/audits/r672-2026-09-24/B5-after.txt create mode 100644 documentation/audits/r672-2026-09-24/B5-start.txt create mode 100644 documentation/audits/r672-2026-09-24/B6-box-stopped-redis-apps.log create mode 100644 documentation/audits/r672-2026-09-24/B7-redis-aof-fix.txt create mode 100644 documentation/audits/r672-2026-09-24/B8-start-redis-apps.txt create mode 100644 documentation/audits/r672-2026-09-24/B9-hub-online.txt create mode 100644 documentation/audits/r672-2026-09-24/C5-r673-vzdump-logs.txt create mode 100644 documentation/audits/r672-2026-09-24/C7a-high-margin.txt create mode 100644 documentation/audits/r672-2026-09-24/C7b-normal-margin.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-hub-storage-cooldown.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-hub-thin-generic.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-janitor-no-gate.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-retry-no-timer.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-skip-dropped.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-space-1-no-preflight.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-space-2-file-size.txt create mode 100644 documentation/audits/r672-2026-09-24/redproofs/C-thinhigh-no-edge.txt diff --git a/documentation/audits/r672-2026-09-24/A-restore-test-off.txt b/documentation/audits/r672-2026-09-24/A-restore-test-off.txt new file mode 100644 index 00000000..d881a593 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/A-restore-test-off.txt @@ -0,0 +1,16 @@ +== demo-hp 15:33:51 +active +Sep 24 15:33:52 demo-hp felhom-agent[3134463]: time=2026-09-24T15:33:52.034+02:00 level=INFO msg="backup: restore-test scheduler shutting down" reason="context canceled" +Sep 24 15:33:52 demo-hp felhom-agent[3717779]: time=2026-09-24T15:33:52.069+02:00 level=INFO msg="felhom-agent daemon starting" version=0.132.0 host_id=demo-hp-bb76ea hub_url=https://hub.felhom.eu interval_s=900 +Sep 24 15:33:53 demo-hp felhom-agent[3717779]: time=2026-09-24T15:33:53.084+02:00 level=INFO msg="backup: restore-test cadence disabled" +Sep 24 15:33:53 demo-hp felhom-agent[3717779]: time=2026-09-24T15:33:53.084+02:00 level=INFO msg="storage: watchdog starting" interval=5s debounce=15s +Sep 24 15:33:53 demo-hp felhom-agent[3717779]: time=2026-09-24T15:33:53.084+02:00 level=INFO msg="pbs: verify loop starting" cadence=6h0m0s +Sep 24 15:33:53 demo-hp felhom-agent[3717779]: time=2026-09-24T15:33:53.864+02:00 level=INFO msg="selfheal: node watchdog starting" mode=appliance interval_s=60 +== felhom-pve 15:34:00 +active +Sep 24 15:34:00 demo-felhom felhom-agent[3550256]: time=2026-09-24T15:34:00.464+02:00 level=INFO msg="backup: restore-test scheduler shutting down" reason="context canceled" +Sep 24 15:34:00 demo-felhom felhom-agent[3559458]: time=2026-09-24T15:34:00.487+02:00 level=INFO msg="felhom-agent daemon starting" version=0.132.0 host_id=demo-felhom-8363b5 hub_url=https://hub.felhom.eu interval_s=900 +Sep 24 15:34:01 demo-felhom felhom-agent[3559458]: time=2026-09-24T15:34:01.093+02:00 level=INFO msg="pbs: verify loop starting" cadence=6h0m0s +Sep 24 15:34:01 demo-felhom felhom-agent[3559458]: time=2026-09-24T15:34:01.093+02:00 level=INFO msg="storage: watchdog starting" interval=5s debounce=15s +Sep 24 15:34:01 demo-felhom felhom-agent[3559458]: time=2026-09-24T15:34:01.094+02:00 level=INFO msg="backup: restore-test cadence disabled" +Sep 24 15:34:02 demo-felhom felhom-agent[3559458]: time=2026-09-24T15:34:02.268+02:00 level=INFO msg="selfheal: node watchdog starting" mode=appliance interval_s=60 diff --git a/documentation/audits/r672-2026-09-24/B1-before.txt b/documentation/audits/r672-2026-09-24/B1-before.txt new file mode 100644 index 00000000..f0f08ad3 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B1-before.txt @@ -0,0 +1,21 @@ +Thu Sep 24 15:34:15 CEST 2026 +Name Type Status Total (KiB) Used (KiB) Available (KiB) % +felhom-pbs pbs active 0 0 0 0.00% +local dir active 40453376 34137828 4228432 84.39% +local-lvm lvmthin active 56487936 33277043 23210892 58.91% +nvme-scratch dir active 983379700 74629276 858723812 7.59% + LV LSize Data% Meta% + data 53.87g 58.91 2.65 + root <39.50g + swap 8.00g + vm-9201-disk-0 32.00g 5.39 + vm-9201-disk-1 70.00g 42.87 +hostname: demo-hp +mp0: local-lvm:vm-9201-disk-1,mp=/var/lib/felhom,backup=1,size=70G +mp8: /mnt/felhom-drives,mp=/mnt/felhom-drives +mp9: /var/lib/felhom-agent/guests/9201/bootstrap,mp=/etc/felhom-bootstrap,ro=1 +rootfs: local-lvm:vm-9201-disk-0,size=32G +unprivileged: 1 +VMID Status Lock Name +9201 running demo-hp +9202 running demo-hp-scratch diff --git a/documentation/audits/r672-2026-09-24/B2-9201-apps-before.txt b/documentation/audits/r672-2026-09-24/B2-9201-apps-before.txt new file mode 100644 index 00000000..163d8bc4 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B2-9201-apps-before.txt @@ -0,0 +1,23 @@ +felhom-controller Up 7 hours (healthy) +romm Up 7 hours (healthy) +romm-db Up 7 hours (healthy) +romm-redis Up 7 hours (healthy) +privatebin Up 7 hours (healthy) +opengist Up 7 hours (healthy) +kimai Up 7 hours (healthy) +kimai-db Up 7 hours (healthy) +docmost Up 7 hours (healthy) +docmost-postgres Up 7 hours (healthy) +docmost-redis Up 7 hours (healthy) +calibre-web Up 7 hours (healthy) +bookstack Up 7 hours (unhealthy) +bookstack-db Up 7 hours (healthy) +bentopdf Up 7 hours (healthy) +adventurelog Up 7 hours (unhealthy) +adventurelog-postgres Up 7 hours (healthy) +adventurelog-frontend Up 7 hours (healthy) +filebrowser Up 3 weeks (healthy) +cloudflared Up 3 weeks +traefik Up 3 weeks +touch: cannot touch '/root/.w': Read-only file system +touch: cannot touch '/var/lib/felhom/.w': Read-only file system diff --git a/documentation/audits/r672-2026-09-24/B3-stop.txt b/documentation/audits/r672-2026-09-24/B3-stop.txt new file mode 100644 index 00000000..c351930f --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B3-stop.txt @@ -0,0 +1,7 @@ +15:34:31 +rc=0 + +real 0m4.823s +user 0m0.846s +sys 0m0.226s +status: stopped diff --git a/documentation/audits/r672-2026-09-24/B4-fsck.txt b/documentation/audits/r672-2026-09-24/B4-fsck.txt new file mode 100644 index 00000000..de375e33 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B4-fsck.txt @@ -0,0 +1,43 @@ +== pct fsck 9201 (rootfs) +fsck from util-linux 2.41 +MMP interval is 10 seconds and total wait time is 42 seconds. Please wait... +/dev/mapper/pve-vm--9201--disk--0: recovering journal +/dev/mapper/pve-vm--9201--disk--0: 24869/2097152 files (0.1% non-contiguous), 455009/8388608 blocks +command 'fsck -a -l /dev/pve/vm-9201-disk-0' failed: exit code 1 +rc=1 +== pct fsck 9201 --device mp0 +fsck from util-linux 2.41 +MMP interval is 10 seconds and total wait time is 42 seconds. Please wait... +/dev/mapper/pve-vm--9201--disk--1: recovering journal +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4063400 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4063401 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4063402 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4063403 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4064117 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4064123 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4065627 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4065628 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4065629 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4065630 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066048 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066185 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066186 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066187 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066188 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066277 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Deleted inode 4066278 has zero dtime. FIXED. +/dev/mapper/pve-vm--9201--disk--1: Orphan file (inode 12) block 13 is not clean. +CLEARED. +/dev/mapper/pve-vm--9201--disk--1: 616408/4587520 files (0.3% non-contiguous), 6139421/18350080 blocks +command 'fsck -a -l /dev/pve/vm-9201-disk-1' failed: exit code 1 +rc=1 +== second pass rootfs +fsck from util-linux 2.41 +MMP interval is 10 seconds and total wait time is 42 seconds. Please wait... +/dev/mapper/pve-vm--9201--disk--0: clean, 24869/2097152 files, 455009/8388608 blocks +rc=0 +== second pass mp0 +fsck from util-linux 2.41 +MMP interval is 10 seconds and total wait time is 42 seconds. Please wait... +/dev/mapper/pve-vm--9201--disk--1: clean, 616408/4587520 files, 6139421/18350080 blocks +rc=0 diff --git a/documentation/audits/r672-2026-09-24/B5-after.txt b/documentation/audits/r672-2026-09-24/B5-after.txt new file mode 100644 index 00000000..5a9478af --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B5-after.txt @@ -0,0 +1,27 @@ +15:37:16 +adventurelog ghcr.io/seanmorley15/adventurelog-backend:v0.12.1 Up 35 seconds (healthy) +adventurelog-frontend ghcr.io/seanmorley15/adventurelog-frontend:v0.12.1 Up 35 seconds (healthy) +adventurelog-postgres postgis/postgis:16-3.5-alpine Up 35 seconds (healthy) +bentopdf ghcr.io/alam00000/bentopdf:v2.8.6 Up 35 seconds (healthy) +bookstack lscr.io/linuxserver/bookstack:26.05.5 Up 35 seconds (healthy) +bookstack-db mariadb:12.3 Up 35 seconds (healthy) +calibre-web crocodilestick/calibre-web-automated:v4.0.6 Up 15 seconds (health: starting) +cloudflared cloudflare/cloudflared:2026.6.0 Up 35 seconds +docmost docmost/docmost:0.96.0 Up 35 seconds (health: starting) +docmost-postgres postgres:16-alpine Up 35 seconds (healthy) +docmost-redis redis:7-alpine Restarting (1) 2 seconds ago +felhom-controller gitea.dooplex.hu/admin/felhom-controller:0.269.1 Up 8 seconds (healthy) +filebrowser gtstef/filebrowser:1.3.3-stable Up 35 seconds (healthy) +kimai kimai/kimai2:apache-2.57.0 Up 35 seconds (health: starting) +kimai-db mariadb:11.8 Up 35 seconds (healthy) +opengist ghcr.io/thomiceli/opengist:1.15 Up 35 seconds (healthy) +privatebin privatebin/pdo:2.0.6 Up 35 seconds (healthy) +romm rommapp/romm:5.3.1 Up 35 seconds (health: starting) +romm-db mariadb:11.8 Up 35 seconds (healthy) +romm-redis redis:7-alpine Restarting (1) 2 seconds ago +traefik traefik:v3.6.7 Up 35 seconds +ROOTFS_WRITABLE +DATA_WRITABLE +2026/09/24 13:37:10 offsiteapply.go:151: [INFO] [offsite-apply] settle-gate: awaiting floor knowledge (first report ACK) before offsite apply +2026/09/24 13:37:16 updater.go:102: [DEBUG] [selfupdate] SetFloor: floor "" → "0.269.1" +2026/09/24 13:37:16 updater.go:102: [DEBUG] [selfupdate] maybeAutoUpdate: current 0.269.1 >= floor 0.269.1 — no action diff --git a/documentation/audits/r672-2026-09-24/B5-start.txt b/documentation/audits/r672-2026-09-24/B5-start.txt new file mode 100644 index 00000000..b973e758 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B5-start.txt @@ -0,0 +1,4 @@ +15:36:31 +start_rc=0 +15:36:56 +controller: gitea.dooplex.hu/admin/felhom-controller:0.268.0 Up 11 seconds (healthy) diff --git a/documentation/audits/r672-2026-09-24/B6-box-stopped-redis-apps.log b/documentation/audits/r672-2026-09-24/B6-box-stopped-redis-apps.log new file mode 100644 index 00000000..8cf29cc4 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B6-box-stopped-redis-apps.log @@ -0,0 +1,162 @@ +2026/09/24 13:37:10 scheduler.go:102: [INFO] [scheduler] Registered periodic job: deadapp-check (every 30s) +2026/09/24 13:37:10 notifier.go:346: [INFO] Event pushed: controller_updated (info) [hu-only] — Controller frissítve: 0.268.0 → 0.269.1 +386ae33fee09 romm romm rommapp/romm:5.3.1 +e65ba7c8b78c romm-db romm mariadb:11.8 +2026/09/24 13:37:10 recovery_unit.go:224: [INFO] [backup] Recovery unit captured for docmost → /mnt/sys_drive/felhom-data/backups/primary/docmost (images=3, secrets-referenced=2, data_keys=0, portable-carried=2/2, withheld=0) +2026/09/24 13:37:11 recovery_unit.go:224: [INFO] [backup] Recovery unit captured for romm → /mnt/felhom-drives/hdd_1/backups/primary/romm (images=3, secrets-referenced=3, data_keys=0, portable-carried=3/3, withheld=0) +2026/09/24 13:37:15 notifier.go:346: [INFO] Event pushed: controller_started (info) [hu-only] — Controller elindult (0.269.1) +2026/09/24 13:37:35 intermediary.go:462: [INFO] [gate] boot 1787327067-292992616: live bind confirmed — recreating drive-backed app romm (state=starting) onto /mnt/felhom-drives/hdd_1 +2026/09/24 13:37:35 manager.go:1286: [INFO] [stacks] Stopping stack: romm +2026/09/24 13:37:38 manager.go:1295: [INFO] [stacks] Stack romm stopped successfully (took 3.5s) +2026/09/24 13:37:38 manager.go:1197: [INFO] [stacks] Starting stack: romm +2026/09/24 13:37:44 manager.go:1472: [ERROR] [stacks] Command failed: docker compose up -d (in /opt/docker/stacks/romm) — exit code 1 (took 5.6s) +2026/09/24 13:37:44 manager.go:1478: [ERROR] [stacks] stderr: Network romm_romm-internal Creating + Network romm_romm-internal Creating + Network romm_romm-internal Created + Network romm_romm-internal Created + Container romm-db Creating + Container romm-redis Creating + Container romm-redis Created + Container romm-db Created + Container romm Creating + Container romm Created + Container romm-redis Starting + Container romm-db Starting + Container romm-db Started + Container romm-redis Started + Container romm-redis Waiting + Container romm-db Waiting +2026/09/24 13:37:44 manager.go:1208: [ERROR] [stacks] Stack romm start failed after 5.6s: exit code 1 +stderr: Network romm_romm-internal Creating + Network romm_romm-internal Creating + Network romm_romm-internal Created + Network romm_romm-internal Created + Container romm-db Creating + Container romm-redis Creating + Container romm-redis Created + Container romm-db Created + Container romm Creating + Container romm Created + Container romm-redis Starting + Container romm-db Starting + Container romm-db Started + Container romm-redis Started + Container romm-redis Waiting + Container romm-db Waiting +2026/09/24 13:37:44 intermediary.go:468: [WARN] [gate] boot recreate romm: starting stack romm: exit code 1 +stderr: Network romm_romm-internal Creating + Network romm_romm-internal Creating + Network romm_romm-internal Created + Network romm_romm-internal Created + Container romm-db Creating + Container romm-redis Creating + Container romm-redis Created + Container romm-db Created + Container romm Creating + Container romm Created + Container romm-redis Starting + Container romm-db Starting + Container romm-db Started + Container romm-redis Started + Container romm-redis Waiting + Container romm-db Waiting +2026/09/24 13:37:55 bootrecon.go:259: [INFO] [bootrecon] Boot reconciliation: 1 boot-orphaned app(s) found: [romm] — up to 2 attempt(s) +2026/09/24 13:37:55 manager.go:1197: [INFO] [stacks] Starting stack: romm +2026/09/24 13:37:56 manager.go:1472: [ERROR] [stacks] Command failed: docker compose up -d (in /opt/docker/stacks/romm) — exit code 1 (took 0.7s) +2026/09/24 13:37:56 manager.go:1478: [ERROR] [stacks] stderr: Container romm-db Running + Container romm-redis Starting + Container romm-redis Started + Container romm-redis Waiting + Container romm-db Waiting + Container romm-redis Error dependency romm-redis failed to start + Container romm-db Healthy +dependency failed to start: container romm-redis is unhealthy +2026/09/24 13:37:56 manager.go:1208: [ERROR] [stacks] Stack romm start failed after 0.7s: exit code 1 +stderr: Container romm-db Running + Container romm-redis Starting + Container romm-redis Started + Container romm-redis Waiting + Container romm-db Waiting + Container romm-redis Error dependency romm-redis failed to start + Container romm-db Healthy +dependency failed to start: container romm-redis is unhealthy +2026/09/24 13:37:56 bootrecon.go:270: [WARN] [bootrecon] Boot reconciliation attempt 1/2: start "romm" failed after 0.7s: starting stack romm: exit code 1 +stderr: Container romm-db Running + Container romm-redis Starting + Container romm-redis Started + Container romm-redis Waiting + Container romm-db Waiting + Container romm-redis Error dependency romm-redis failed to start + Container romm-db Healthy +dependency failed to start: container romm-redis is unhealthy +2026/09/24 13:38:20 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:38:26 manager.go:1197: [INFO] [stacks] Starting stack: romm +2026/09/24 13:38:27 manager.go:1472: [ERROR] [stacks] Command failed: docker compose up -d (in /opt/docker/stacks/romm) — exit code 1 (took 0.6s) +2026/09/24 13:38:27 manager.go:1478: [ERROR] [stacks] stderr: Container romm-db Running + Container romm-redis Starting + Container romm-redis Started + Container romm-db Waiting + Container romm-redis Waiting + Container romm-db Healthy + Container romm-redis Error dependency romm-redis failed to start +dependency failed to start: container romm-redis is unhealthy +2026/09/24 13:38:27 manager.go:1208: [ERROR] [stacks] Stack romm start failed after 0.6s: exit code 1 +stderr: Container romm-db Running + Container romm-redis Starting + Container romm-redis Started + Container romm-db Waiting + Container romm-redis Waiting + Container romm-db Healthy + Container romm-redis Error dependency romm-redis failed to start +dependency failed to start: container romm-redis is unhealthy +2026/09/24 13:38:27 bootrecon.go:270: [WARN] [bootrecon] Boot reconciliation attempt 2/2: start "romm" failed after 0.6s: starting stack romm: exit code 1 +stderr: Container romm-db Running + Container romm-redis Starting + Container romm-redis Started + Container romm-db Waiting + Container romm-redis Waiting + Container romm-db Healthy + Container romm-redis Error dependency romm-redis failed to start +dependency failed to start: container romm-redis is unhealthy +2026/09/24 13:38:27 bootrecon.go:308: [WARN] [bootrecon] Boot reconciliation gave up after 2 attempt(s): recovered=[] still down=[romm] (the dead-app alarm now owns these) +2026/09/24 13:38:40 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:38:40 notifier.go:346: [INFO] Event pushed: app_start_failed (warning) [hu-only] — Telepített alkalmazás nem fut: RomM +2026/09/24 13:39:00 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:39:10 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:39:20 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:39:40 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:40:00 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:40:20 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:40:30 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:40:50 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:41:10 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:41:30 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:41:50 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:42:10 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:42:30 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:42:50 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:42:50 healthprobe.go:190: [WARN] Health probe romm: HTTP GET :8080/ → Get "http://romm-db:8080/": dial tcp: lookup romm-db on 127.0.0.11:53: no such host +2026/09/24 13:43:00 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:43:00 healthprobe.go:190: [WARN] Health probe romm: HTTP GET :8080/ → Get "http://romm-db:8080/": dial tcp: lookup romm-db on 127.0.0.11:53: no such host +2026/09/24 13:43:10 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:43:10 healthprobe.go:190: [WARN] Health probe romm: HTTP GET :8080/ → Get "http://romm-db:8080/": dial tcp: lookup romm-db on 127.0.0.11:53: no such host +2026/09/24 13:43:30 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:43:30 healthprobe.go:190: [WARN] Health probe romm: HTTP GET :8080/ → Get "http://romm-db:8080/": dial tcp: lookup romm-db on 127.0.0.11:53: no such host +2026/09/24 13:43:40 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:43:40 healthprobe.go:190: [WARN] Health probe romm: HTTP GET :8080/ → Get "http://romm-db:8080/": dial tcp: lookup romm-db on 127.0.0.11:53: no such host +2026/09/24 13:44:00 healthprobe.go:190: [WARN] Health probe docmost: HTTP GET :3000/ → Get "http://docmost:3000/": dial tcp 172.18.0.15:3000: connect: connection refused +2026/09/24 13:44:00 healthprobe.go:190: [WARN] Health probe romm: HTTP GET :8080/ → Get "http://romm-db:8080/": dial tcp: lookup romm-db on 127.0.0.11:53: no such host +2026/09/24 13:44:10 main.go:3671: [WARN] [deadapp] docmost: crash_loop — 6 in 10m0s; STOPPING it (decision 28) +2026/09/24 13:44:10 manager.go:1286: [INFO] [stacks] Stopping stack: docmost +2026/09/24 13:44:11 manager.go:1295: [INFO] [stacks] Stack docmost stopped successfully (took 1.1s) +2026/09/24 13:44:11 settings.go:1814: [WARN] [settings] restore hold SET for docmost — the app stays stopped until it is cleared +2026/09/24 13:44:11 update_guard.go:776: [WARN] [backup] docmost is STOPPED by the box: crash_loop (trip 1 within 24h0m0s) — Start gives it one more try (decision 28) +2026/09/24 13:44:11 main.go:3671: [WARN] [deadapp] romm: crash_loop — 6 in 10m0s; STOPPING it (decision 28) +2026/09/24 13:44:11 manager.go:1286: [INFO] [stacks] Stopping stack: romm +2026/09/24 13:44:11 notifier.go:346: [INFO] Event pushed: app_stopped_unhealthy (warning) [hu-only] — A(z) docmost alkalmazást leállítottuk, mert újra és újra összeomlott (vagy elfogyott a memóriája). Az Indítás gombbal újra megpróbálhatod. +2026/09/24 13:44:12 manager.go:1295: [INFO] [stacks] Stack romm stopped successfully (took 0.5s) +2026/09/24 13:44:12 settings.go:1814: [WARN] [settings] restore hold SET for romm — the app stays stopped until it is cleared +2026/09/24 13:44:12 update_guard.go:776: [WARN] [backup] romm is STOPPED by the box: crash_loop (trip 1 within 24h0m0s) — Start gives it one more try (decision 28) +2026/09/24 13:44:12 notifier.go:346: [INFO] Event pushed: app_stopped_unhealthy (warning) [hu-only] — A(z) romm alkalmazást leállítottuk, mert újra és újra összeomlott (vagy elfogyott a memóriája). Az Indítás gombbal újra megpróbálhatod. +2026/09/24 13:48:10 main.go:1981: [INFO] [deadapp] check alive: 20 scans since boot, 10 deployed app(s) evaluated, 0 currently down +2026/09/24 13:58:10 main.go:1981: [INFO] [deadapp] check alive: 40 scans since boot, 10 deployed app(s) evaluated, 0 currently down diff --git a/documentation/audits/r672-2026-09-24/B7-redis-aof-fix.txt b/documentation/audits/r672-2026-09-24/B7-redis-aof-fix.txt new file mode 100644 index 00000000..fb122e99 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B7-redis-aof-fix.txt @@ -0,0 +1,126 @@ + LANG = "en_US.UTF-8" + LANG = "en_US.UTF-8" +== docmost_docmost_redis_data +total 43996 +drwx------ 2 999 1000 4096 Sep 23 02:37 . +drwxr-xr-x 3 999 1000 4096 Sep 24 08:39 .. +-rw------- 1 999 1000 205432 Sep 23 02:37 appendonly.aof.4.base.rdb +-rw------- 1 999 1000 44828283 Sep 24 08:37 appendonly.aof.4.incr.aof +-rw------- 1 999 1000 88 Sep 23 02:37 appendonly.aof.manifest +-- copy kept as volume docmost_docmost_redis_data.pre-aof-fix-20260924T135937Z +total 44000 +drwx------ 2 redis redis 4096 Sep 23 02:37 . +drwxr-xr-x 3 redis redis 4096 Sep 24 08:39 .. +-rw------- 1 redis redis 205432 Sep 23 02:37 appendonly.aof.4.base.rdb +-rw------- 1 redis redis 44828283 Sep 24 08:37 appendonly.aof.4.incr.aof +-rw------- 1 redis redis 88 Sep 23 02:37 appendonly.aof.manifest +Start checking Multi Part AOF +Start to check BASE AOF (RDB format). +[offset 0] Checking RDB file ./appendonly.aof.4.base.rdb +[offset 27] AUX FIELD redis-ver = '7.4.11' +[offset 41] AUX FIELD redis-bits = '64' +[offset 53] AUX FIELD ctime = '1790131053' +[offset 68] AUX FIELD used-mem = '3686984' +[offset 80] AUX FIELD aof-base = '1' +[offset 82] Selecting DB ID 0 +[offset 205432] Checksum OK +[offset 205432] \o/ RDB looks OK! \o/ +[info] 52 keys read +[info] 12 expires +[info] 12 already expired +[info] 0 subexpires +RDB preamble is OK, proceeding with AOF tail... +AOF analyzed: filename=appendonly.aof.4.base.rdb, size=205432, ok_up_to=205432, ok_up_to_line=1, diff=0 +BASE AOF appendonly.aof.4.base.rdb is valid +Start to check INCR files. +0x 2abfffc: Expected \r\n, got: 0d00 +AOF analyzed: filename=appendonly.aof.4.incr.aof, size=44828283, ok_up_to=44825340, ok_up_to_line=4414097, diff=2943 +This will shrink the AOF appendonly.aof.4.incr.aof from 44828283 bytes, with 2943 bytes, to 44825340 bytes +Continue? [y/N]: Successfully truncated AOF appendonly.aof.4.incr.aof +All AOF files and manifest are valid +fix_rc=0 +Start checking Multi Part AOF +Start to check BASE AOF (RDB format). +[offset 0] Checking RDB file ./appendonly.aof.4.base.rdb +[offset 27] AUX FIELD redis-ver = '7.4.11' +[offset 41] AUX FIELD redis-bits = '64' +[offset 53] AUX FIELD ctime = '1790131053' +[offset 68] AUX FIELD used-mem = '3686984' +[offset 80] AUX FIELD aof-base = '1' +[offset 82] Selecting DB ID 0 +[offset 205432] Checksum OK +[offset 205432] \o/ RDB looks OK! \o/ +[info] 52 keys read +[info] 12 expires +[info] 12 already expired +[info] 0 subexpires +RDB preamble is OK, proceeding with AOF tail... +AOF analyzed: filename=appendonly.aof.4.base.rdb, size=205432, ok_up_to=205432, ok_up_to_line=1, diff=0 +BASE AOF appendonly.aof.4.base.rdb is valid +Start to check INCR files. +AOF analyzed: filename=appendonly.aof.4.incr.aof, size=44825340, ok_up_to=44825340, ok_up_to_line=4413969, diff=0 +INCR AOF appendonly.aof.4.incr.aof is valid +All AOF files and manifest are valid +check_rc=0 +== romm_romm_redis_data +total 52912 +drwx------ 2 999 1000 4096 Sep 23 15:33 . +drwxr-xr-x 3 999 1000 4096 Sep 24 08:35 .. +-rw------- 1 999 1000 6778429 Sep 23 15:33 appendonly.aof.3.base.rdb +-rw------- 1 999 1000 47389115 Sep 24 08:38 appendonly.aof.3.incr.aof +-rw------- 1 999 1000 88 Sep 23 15:33 appendonly.aof.manifest +-- copy kept as volume romm_romm_redis_data.pre-aof-fix-20260924T135937Z +total 52916 +drwx------ 2 redis redis 4096 Sep 23 15:33 . +drwxr-xr-x 3 redis redis 4096 Sep 24 08:35 .. +-rw------- 1 redis redis 6778429 Sep 23 15:33 appendonly.aof.3.base.rdb +-rw------- 1 redis redis 47389115 Sep 24 08:38 appendonly.aof.3.incr.aof +-rw------- 1 redis redis 88 Sep 23 15:33 appendonly.aof.manifest +Start checking Multi Part AOF +Start to check BASE AOF (RDB format). +[offset 0] Checking RDB file ./appendonly.aof.3.base.rdb +[offset 27] AUX FIELD redis-ver = '7.4.11' +[offset 41] AUX FIELD redis-bits = '64' +[offset 53] AUX FIELD ctime = '1790177587' +[offset 68] AUX FIELD used-mem = '12936648' +[offset 80] AUX FIELD aof-base = '1' +[offset 82] Selecting DB ID 0 +[offset 6778429] Checksum OK +[offset 6778429] \o/ RDB looks OK! \o/ +[info] 194 keys read +[info] 165 expires +[info] 146 already expired +[info] 0 subexpires +RDB preamble is OK, proceeding with AOF tail... +AOF analyzed: filename=appendonly.aof.3.base.rdb, size=6778429, ok_up_to=6778429, ok_up_to_line=1, diff=0 +BASE AOF appendonly.aof.3.base.rdb is valid +Start to check INCR files. +0x 2d2fff6: Expected \r\n, got: 0000 +AOF analyzed: filename=appendonly.aof.3.incr.aof, size=47389115, ok_up_to=47382484, ok_up_to_line=4501637, diff=6631 +This will shrink the AOF appendonly.aof.3.incr.aof from 47389115 bytes, with 6631 bytes, to 47382484 bytes +Continue? [y/N]: Successfully truncated AOF appendonly.aof.3.incr.aof +All AOF files and manifest are valid +fix_rc=0 +Start checking Multi Part AOF +Start to check BASE AOF (RDB format). +[offset 0] Checking RDB file ./appendonly.aof.3.base.rdb +[offset 27] AUX FIELD redis-ver = '7.4.11' +[offset 41] AUX FIELD redis-bits = '64' +[offset 53] AUX FIELD ctime = '1790177587' +[offset 68] AUX FIELD used-mem = '12936648' +[offset 80] AUX FIELD aof-base = '1' +[offset 82] Selecting DB ID 0 +[offset 6778429] Checksum OK +[offset 6778429] \o/ RDB looks OK! \o/ +[info] 194 keys read +[info] 165 expires +[info] 146 already expired +[info] 0 subexpires +RDB preamble is OK, proceeding with AOF tail... +AOF analyzed: filename=appendonly.aof.3.base.rdb, size=6778429, ok_up_to=6778429, ok_up_to_line=1, diff=0 +BASE AOF appendonly.aof.3.base.rdb is valid +Start to check INCR files. +AOF analyzed: filename=appendonly.aof.3.incr.aof, size=47382484, ok_up_to=47382484, ok_up_to_line=4501630, diff=0 +INCR AOF appendonly.aof.3.incr.aof is valid +All AOF files and manifest are valid +check_rc=0 diff --git a/documentation/audits/r672-2026-09-24/B8-start-redis-apps.txt b/documentation/audits/r672-2026-09-24/B8-start-redis-apps.txt new file mode 100644 index 00000000..af91e3f7 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B8-start-redis-apps.txt @@ -0,0 +1,10 @@ +docmost before: stopped unhealthy_stop +docmost Start -> 200 {'ok': True, 'data': {'state': 'starting'}, 'message': 'Stack docmost start requested — state now: starting'} +romm before: stopped unhealthy_stop +romm Start -> 200 {'ok': True, 'data': {'state': 'starting'}, 'message': 'Stack romm start requested — state now: starting'} +after 90 s: {'docmost': ('running', None), 'romm': ('running', None)} +/docmost-redis running RestartCount=0 +1:M 24 Sep 2026 16:01:58.128 * Background saving terminated with success +/romm-redis running RestartCount=0 +1:M 24 Sep 2026 16:02:10.238 * Background saving terminated with success + diff --git a/documentation/audits/r672-2026-09-24/B9-hub-online.txt b/documentation/audits/r672-2026-09-24/B9-hub-online.txt new file mode 100644 index 00000000..e5e7acf9 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/B9-hub-online.txt @@ -0,0 +1 @@ +demo-hp-bb76ea'" style="cursor: pointer;"> demo-hp-bb76ea Demo HP 0.132.0 floor: declared MinAgent ONLINE 1/2 12% 15% 84% inactive 84% local drill-r50-0a4f9a drill-r50 0.129.0 floor held DOWN 1/1 20% 20% 10% inactive 10% l diff --git a/documentation/audits/r672-2026-09-24/C5-r673-vzdump-logs.txt b/documentation/audits/r672-2026-09-24/C5-r673-vzdump-logs.txt new file mode 100644 index 00000000..5cfc7bbf --- /dev/null +++ b/documentation/audits/r672-2026-09-24/C5-r673-vzdump-logs.txt @@ -0,0 +1,40 @@ +== /var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-06_59_28.log +2026-09-24 06:59:28 INFO: Starting Backup of VM 9201 (lxc) +2026-09-24 06:59:28 INFO: status = running +2026-09-24 06:59:28 INFO: CT Name: demo-hp +2026-09-24 06:59:28 INFO: including mount point rootfs ('/') in backup +2026-09-24 06:59:28 INFO: including mount point mp0 ('/var/lib/felhom') in backup +2026-09-24 06:59:28 INFO: excluding bind mount point mp8 ('/mnt/felhom-drives') from backup (not a volume) +2026-09-24 06:59:28 INFO: excluding bind mount point mp9 ('/etc/felhom-bootstrap') from backup (not a volume) +2026-09-24 06:59:28 INFO: backup mode: snapshot +2026-09-24 06:59:28 INFO: ionice priority: 7 +2026-09-24 06:59:28 INFO: suspend vm to make snapshot +2026-09-24 06:59:28 INFO: create storage snapshot 'vzdump' +2026-09-24 06:59:29 INFO: resume vm +2026-09-24 06:59:29 INFO: guest is online again after 1 seconds +2026-09-24 06:59:29 INFO: creating vzdump archive '/var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-06_59_28.tar.zst' +2026-09-24 07:05:46 INFO: zstd: error 70 : Write error : cannot write block : No space left on device +2026-09-24 07:05:50 INFO: cleanup temporary 'vzdump' snapshot +2026-09-24 07:05:51 ERROR: Backup of VM 9201 failed - command 'set -o pipefail && lxc-usernsexec -m u:0:100000:65536 -m g:0:100000:65536 -- tar cpf - --totals --one-file-system -p --sparse --numeric-owner --acls --xattrs '--xattrs-include=user.*' '--xattrs-include=security.capability' '--warning=no-file-ignored' '--warning=no-xattr-write' --one-file-system '--warning=no-file-ignored' '--directory=/var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-06_59_28.tmp' ./etc/vzdump/pct.conf ./etc/vzdump/pct.fw '--directory=/mnt/vzsnap0' --no-anchored '--exclude=lost+found' --anchored '--exclude=./tmp/?*' '--exclude=./var/tmp/?*' '--exclude=./var/run/?*.pid' ./ ./var/lib/felhom | zstd '--threads=1' >/var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-06_59_28.tar.dat' failed: exit code 70 +== /var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-07_42_54.log +2026-09-24 07:42:54 INFO: Starting Backup of VM 9201 (lxc) +2026-09-24 07:42:54 INFO: status = running +2026-09-24 07:42:54 INFO: CT Name: demo-hp +2026-09-24 07:42:54 INFO: including mount point rootfs ('/') in backup +2026-09-24 07:42:54 INFO: including mount point mp0 ('/var/lib/felhom') in backup +2026-09-24 07:42:54 INFO: excluding bind mount point mp8 ('/mnt/felhom-drives') from backup (not a volume) +2026-09-24 07:42:54 INFO: excluding bind mount point mp9 ('/etc/felhom-bootstrap') from backup (not a volume) +2026-09-24 07:42:54 INFO: backup mode: snapshot +2026-09-24 07:42:54 INFO: ionice priority: 7 +2026-09-24 07:42:54 INFO: suspend vm to make snapshot +2026-09-24 07:42:54 INFO: create storage snapshot 'vzdump' +2026-09-24 07:42:55 INFO: resume vm +2026-09-24 07:42:55 INFO: guest is online again after 1 seconds +2026-09-24 07:42:55 INFO: creating vzdump archive '/var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-07_42_54.tar.zst' +2026-09-24 07:49:08 INFO: zstd: error 70 : Write error : cannot write block : No space left on device +2026-09-24 07:49:12 INFO: cleanup temporary 'vzdump' snapshot +2026-09-24 07:49:13 ERROR: Backup of VM 9201 failed - command 'set -o pipefail && lxc-usernsexec -m u:0:100000:65536 -m g:0:100000:65536 -- tar cpf - --totals --one-file-system -p --sparse --numeric-owner --acls --xattrs '--xattrs-include=user.*' '--xattrs-include=security.capability' '--warning=no-file-ignored' '--warning=no-xattr-write' --one-file-system '--warning=no-file-ignored' '--directory=/var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-07_42_54.tmp' ./etc/vzdump/pct.conf ./etc/vzdump/pct.fw '--directory=/mnt/vzsnap0' --no-anchored '--exclude=lost+found' --anchored '--exclude=./tmp/?*' '--exclude=./var/tmp/?*' '--exclude=./var/run/?*.pid' ./ ./var/lib/felhom | zstd '--threads=1' >/var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-07_42_54.tar.dat' failed: exit code 70 +== /var/lib/vz/dump/vzdump-lxc-9201-2026_09_24-08_22_55.log +2026-09-24 08:22:55 INFO: Starting Backup of VM 9201 (lxc) +2026-09-24 08:22:55 INFO: status = running +2026-09-24 08:22:55 ERROR: Backup of VM 9201 failed - CT is locked (snapshot-delete) diff --git a/documentation/audits/r672-2026-09-24/C7a-high-margin.txt b/documentation/audits/r672-2026-09-24/C7a-high-margin.txt new file mode 100644 index 00000000..def4c45c --- /dev/null +++ b/documentation/audits/r672-2026-09-24/C7a-high-margin.txt @@ -0,0 +1,27 @@ +== pool before: 58.99 2.65 +16:23:46 +=== felhom-agent 0.133.0-livetest selftest=restore-test === + --- recover: reaping any leaked scratch from a prior crashed test --- + recover: examined=0 scratch_destroyed=0 scratch_clean=0 + restoring local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst into scratch band [990000,990009] on local-lvm … +time=2026-09-24T16:23:47.394+02:00 level=WARN msg="restore-test SKIPPED by the space preflight (R-672) — nothing was created" archive=local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst storage=local-lvm required_bytes=231442309120 avail_bytes=23721679415 reason="not enough space on local-lvm: restoring 21.1 GiB (vzdump log: total bytes written) needs 215.5 GiB free, has 22.1 GiB" + --- restore-test record --- + { + "source_archive": "local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst", + "source_tier": "local", + "scratch_vmid": 0, + "pass": false, + "verified": "", + "error": "skipped: not enough space on local-lvm: restoring 21.1 GiB (vzdump log: total bytes written) needs 215.5 GiB free, has 22.1 GiB", + "tested_at": "2026-09-24T14:23:47Z", + "duration_seconds": 1.144703492, + "skipped": true + } + space preflight: storage=local-lvm required=231442309120 avail=23721679415 +=== selftest=restore-test SKIPPED — skipped: not enough space on local-lvm: restoring 21.1 GiB (vzdump log: total bytes written) needs 215.5 GiB free, has 22.1 GiB === +exit=4 +== pool after: 58.99 2.65 + LANG = "en_US.UTF-8" +VMID Status Lock Name +9201 running demo-hp +9202 running demo-hp-scratch diff --git a/documentation/audits/r672-2026-09-24/C7b-normal-margin.txt b/documentation/audits/r672-2026-09-24/C7b-normal-margin.txt new file mode 100644 index 00000000..16478b0d --- /dev/null +++ b/documentation/audits/r672-2026-09-24/C7b-normal-margin.txt @@ -0,0 +1,22 @@ +== pool before: 58.99 2.65 +16:23:57 +=== felhom-agent 0.133.0-livetest selftest=restore-test === + --- recover: reaping any leaked scratch from a prior crashed test --- + recover: examined=0 scratch_destroyed=0 scratch_clean=0 + restoring local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst into scratch band [990000,990009] on local-lvm … +time=2026-09-24T16:23:58.489+02:00 level=WARN msg="restore-test SKIPPED by the space preflight (R-672) — nothing was created" archive=local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst storage=local-lvm required_bytes=32497541120 avail_bytes=23721679415 reason="not enough space on local-lvm: restoring 21.1 GiB (vzdump log: total bytes written) needs 30.3 GiB free, has 22.1 GiB" + --- restore-test record --- + { + "source_archive": "local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst", + "pass": false, + "error": "skipped: not enough space on local-lvm: restoring 21.1 GiB (vzdump log: total bytes written) needs 30.3 GiB free, has 22.1 GiB", + "skipped": true + } + space preflight: storage=local-lvm required=32497541120 avail=23721679415 +=== selftest=restore-test SKIPPED — skipped: not enough space on local-lvm: restoring 21.1 GiB (vzdump log: total bytes written) needs 30.3 GiB free, has 22.1 GiB === +exit=4 +== pool after: 58.99 2.65 + LANG = "en_US.UTF-8" +VMID Status Lock Name +9201 running demo-hp +9202 running demo-hp-scratch diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-hub-storage-cooldown.txt b/documentation/audits/r672-2026-09-24/redproofs/C-hub-storage-cooldown.txt new file mode 100644 index 00000000..a13a0e4e --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-hub-storage-cooldown.txt @@ -0,0 +1,5 @@ +# Red-proof 8 (hub) — storage-fill cooldown keyed customer:type, 1 hour (hub v0.123.0) +--- FAIL: TestR672_StorageFillIsPerPoolPerSixHours (0.03s) + r672_storage_cooldown_test.go:28: a second pool filling in the same hour was silenced (sent=1) +FAIL +ok gitea.dooplex.hu/admin/felhom-hub/internal/notify 0.032s diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-hub-thin-generic.txt b/documentation/audits/r672-2026-09-24/redproofs/C-hub-thin-generic.txt new file mode 100644 index 00000000..cce21a2f --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-hub-thin-generic.txt @@ -0,0 +1,8 @@ +# Red-proof 7 (hub) — thin pools judged like any storage, data only (hub v0.123.0) +--- FAIL: TestR672_ThinPoolCriticalAt90 (0.09s) + --- FAIL: TestR672_ThinPoolCriticalAt90/data_91_% (0.02s) + r672_thinpool_test.go:52: events = [storage_fill_warning] [warning] — want one storage_fill_critical (critical) + --- FAIL: TestR672_ThinPoolCriticalAt90/metadata_92_%,_data_50_% (0.02s) + r672_thinpool_test.go:52: events = [] [] — want one storage_fill_critical (critical) + --- FAIL: TestR672_ThinPoolCriticalAt90/data_87_% (0.02s) +ok gitea.dooplex.hu/admin/felhom-hub/internal/monitor 0.122s diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-janitor-no-gate.txt b/documentation/audits/r672-2026-09-24/redproofs/C-janitor-no-gate.txt new file mode 100644 index 00000000..0f3bf467 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-janitor-no-gate.txt @@ -0,0 +1,6 @@ +# Red-proof 4 — the timer's stale-lock sweep without the one-heavy-operation gate +--- FAIL: TestJanitor_StaleLockSweepWaitsForTheHeavyGate (0.00s) + janitor_test.go:50: the sweep ran while a backup held the gate +FAIL +FAIL gitea.dooplex.hu/admin/felhom-agent/cmd/felhom-agent 0.009s +ok gitea.dooplex.hu/admin/felhom-agent/cmd/felhom-agent 0.009s diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-retry-no-timer.txt b/documentation/audits/r672-2026-09-24/redproofs/C-retry-no-timer.txt new file mode 100644 index 00000000..b8790a4b --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-retry-no-timer.txt @@ -0,0 +1,8 @@ +# Red-proof 3 — RetryScratchTeardown does nothing (v0.132.0: a leak was resolved only by Recover at agent start) +--- FAIL: TestRetry_TheTimerDestroysALeakedScratch (0.00s) + restoretest_retry_test.go:38: the leaked scratch was not destroyed by the timer: result={Examined:0 Destroyed:0 Clean:0 Failed:0 GaveUp:[]} destroys=[990000] +--- FAIL: TestRetry_OperatorToldOnceAfterThreeFailures (0.00s) + restoretest_retry_test.go:60: pass 3 gave up on [] — want the operator told exactly once, on pass 3 +FAIL +FAIL gitea.dooplex.hu/admin/felhom-agent/internal/reconcile 0.010s +ok gitea.dooplex.hu/admin/felhom-agent/internal/reconcile 0.010s diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-skip-dropped.txt b/documentation/audits/r672-2026-09-24/redproofs/C-skip-dropped.txt new file mode 100644 index 00000000..115e2325 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-skip-dropped.txt @@ -0,0 +1,4 @@ +# Red-proof 6 — the scheduler drops every skip (v0.132.0's 'if res.Skipped { return }') +--- FAIL: TestR672_SpaceSkipIsReportedNotDropped (0.00s) + r672_skip_test.go:31: a space refusal never reached the host report: [] +FAIL diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-space-1-no-preflight.txt b/documentation/audits/r672-2026-09-24/redproofs/C-space-1-no-preflight.txt new file mode 100644 index 00000000..0e814b2a --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-space-1-no-preflight.txt @@ -0,0 +1,10 @@ +# Red-proof 1 — the space preflight removed (v0.132.0's shape): PreflightRestoreSpace always OK +--- FAIL: TestSpace_The2026_09_24TestIsRefused (0.00s) + restoretest_space_test.go:80: a restore was issued into a pool that cannot take it: [{VMID:990000 Archive:local:backup/vzdump-lxc-9201-2026_09_23-06_55_25.tar.zst Storage:local-lvm Force:false Pool:felhom MountOverrides:map[mp0:local-lvm:70,mp=/var/lib/felhom,backup=1 mp8:local-lvm:1,mp=/mnt/felhom-drives,backup=0 mp9:local-lvm:1,mp=/etc/felhom-bootstrap,backup=0 rootfs:local-lvm:32] ConfigOverrides:map[onboot:0]}] +--- FAIL: TestSpace_UnknownRefuses (0.00s) + --- FAIL: TestSpace_UnknownRefuses/no_space_check_wired (0.00s) + restoretest_space_test.go:143: restores=1 pass=false skipped=false reason="" — an unknown must refuse before anything moves + --- FAIL: TestSpace_UnknownRefuses/restore_size_unknown (0.00s) + restoretest_space_test.go:143: restores=1 pass=false skipped=false reason="" — an unknown must refuse before anything moves + --- FAIL: TestSpace_UnknownRefuses/free_space_unreadable (0.00s) +ok gitea.dooplex.hu/admin/felhom-agent/internal/reconcile (cached) diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-space-2-file-size.txt b/documentation/audits/r672-2026-09-24/redproofs/C-space-2-file-size.txt new file mode 100644 index 00000000..55b80600 --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-space-2-file-size.txt @@ -0,0 +1,6 @@ +# Red-proof 2 — RestoredBytes returns the archive FILE size for a dir storage (the brief's 'archive × 1.2 + 5 GiB') +--- FAIL: TestRestoredBytes_ReadsTheUncompressedSize (0.00s) + restorespace_test.go:80: the compressed file size was used — the restore writes 3× that +FAIL +FAIL gitea.dooplex.hu/admin/felhom-agent/internal/restorespace 0.009s +ok gitea.dooplex.hu/admin/felhom-agent/internal/restorespace (cached) diff --git a/documentation/audits/r672-2026-09-24/redproofs/C-thinhigh-no-edge.txt b/documentation/audits/r672-2026-09-24/redproofs/C-thinhigh-no-edge.txt new file mode 100644 index 00000000..4351fb4e --- /dev/null +++ b/documentation/audits/r672-2026-09-24/redproofs/C-thinhigh-no-edge.txt @@ -0,0 +1,5 @@ +# Red-proof 5 — no 90 % edge on the thin-pool data path (v0.132.0: a WARN line only) +--- FAIL: TestThinHigh_RequestsOneReportPerCrossing (0.00s) + observe_thinhigh_test.go:31: after 910/1000 used: 0 report requests, want 1 +FAIL +ok gitea.dooplex.hu/admin/felhom-agent/internal/storage 0.009s