From 4c69c25b4818212aa2abad2ab84eeefd203a357c Mon Sep 17 00:00:00 2001 From: kisfenyo Date: Thu, 8 Oct 2026 07:34:19 +0200 Subject: [PATCH] R-899: no OS leg after a household press (trigger=manual); after-boot kernel reports carry the saved ring Unreleased; ships with tomorrow's release. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS --- CHANGELOG.md | 16 ++++++++++++++ internal/localapi/afterbackup_test.go | 21 ++++++++++++++---- internal/localapi/server.go | 8 ++++++- internal/osupdate/kernel.go | 28 ++++++++++++++++++----- internal/osupdate/r899_ring_test.go | 32 +++++++++++++++++++++++++++ 5 files changed, 95 insertions(+), 10 deletions(-) create mode 100644 internal/osupdate/r899_ring_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index c6f5cf1..541a303 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,19 @@ +## Unreleased (2026-10-08) — no OS leg after a household press; the after-boot kernel report carries the real ring (R-899) — ships with tomorrow's release + +**Delivery: the agent binary only** — no root file changed. + +- **R-899 (operator ruling 2026-10-08, option A):** `POST /backup?trigger=manual` (a „Mentés most" press, sent by the + controller from its next release) runs no OS leg after it: the leg belongs to the night, after the night's own copy. + Before, a press ran the leg at once (in the day; on 2026-10-07 08:51 demo-hp skipped it only by the 20-hour rule). A + request without the parameter (an older controller, the scheduled path) behaves as before. + `internal/localapi/server.go`; `TestAfterPrimaryBackup` gained two sub-cases (red-proved: ignoring `trigger`, the leg + ran once after a press). +- **The „ring 1" label after a boot:** in the first second after a reboot the agent has not fetched the hub's block, and + the after-boot kernel reports (`judging`, `good`, `revert`, `fell_back`…) said ring 1 on a ring-0 box (demo-felhom, + 2026-10-08 night). Now they read the fetched block, else the block the daemon saved on disk before the reboot (R-866), + else ring 1 as before. A label only — the hub's approval reads its own ring list. `kernelReportRing`, + `TestKernelReportRing_BeforeFirstFetch` (red-proved: ring 1, want 0). + ## v0.153.0 — ring 0 stages exactly the told kernel (R-898; `09` §3 decision 176) (2026-10-07) Released by `scripts/release-agent.sh`: binary sha256 `b204ebe6d65944f43b70dcfa4ae0dfb90da38c92e2ba6a38a1826e488eeb6630`, bundle diff --git a/internal/localapi/afterbackup_test.go b/internal/localapi/afterbackup_test.go index 4c458c4..ee5fe88 100644 --- a/internal/localapi/afterbackup_test.go +++ b/internal/localapi/afterbackup_test.go @@ -15,7 +15,7 @@ import ( // Red-proof: drop the `b.Success &&` guard and the failed-backup sub-case fails; move the call after release() // and the gate sub-case fails. func TestAfterPrimaryBackup(t *testing.T) { - run := func(t *testing.T, failErr string) (calls []int, gateHeld bool) { + run := func(t *testing.T, failErr, path string) (calls []int, gateHeld bool) { gate := &backup.InFlight{} b := &fakeBackups{failErr: failErr} srv := newTestServerS(t, &fakeGuests{}, b, &fakeStore{}, nil) @@ -34,7 +34,7 @@ func TestAfterPrimaryBackup(t *testing.T) { done <- struct{}{} }) h := srv.Handler() - if do(t, h, "POST", "/backup", "A", "").Code != http.StatusAccepted { + if do(t, h, "POST", path, "A", "").Code != http.StatusAccepted { t.Fatal("POST /backup not accepted") } select { @@ -47,7 +47,7 @@ func TestAfterPrimaryBackup(t *testing.T) { return calls, gateHeld } t.Run("success runs the leg under the gate", func(t *testing.T) { - calls, held := run(t, "") + calls, held := run(t, "", "/backup") if len(calls) != 1 { t.Fatalf("the leg ran %d time(s), want 1", len(calls)) } @@ -56,8 +56,21 @@ func TestAfterPrimaryBackup(t *testing.T) { } }) t.Run("a failed backup runs nothing", func(t *testing.T) { - if calls, _ := run(t, "vzdump exploded"); len(calls) != 0 { + if calls, _ := run(t, "vzdump exploded", "/backup"); len(calls) != 0 { t.Fatalf("the leg ran after a FAILED backup: %v", calls) } }) + // R-899: a household press is not the night's backup — no OS leg after it. Same fake, same successful backup as + // the first sub-case; only the query differs. Red-proof: make handleBackup ignore `trigger` and this sub-case + // fails with the leg run once. + t.Run("a manual press runs nothing", func(t *testing.T) { + if calls, _ := run(t, "", "/backup?trigger=manual"); len(calls) != 0 { + t.Fatalf("the OS leg ran after a manual press: %v", calls) + } + }) + t.Run("the scheduled path with the new query still runs the leg", func(t *testing.T) { + if calls, _ := run(t, "", "/backup?trigger=night"); len(calls) != 1 { + t.Fatalf("the leg ran %d time(s) after a non-manual backup, want 1", len(calls)) + } + }) } diff --git a/internal/localapi/server.go b/internal/localapi/server.go index 832de35..8819be8 100644 --- a/internal/localapi/server.go +++ b/internal/localapi/server.go @@ -825,6 +825,10 @@ func (s *Server) handleBackup(w http.ResponseWriter, r *http.Request, vmid int) return } key := backupJobKey{vmid: vmid, target: tier.TargetID} + // R-899 (operator ruling 2026-10-08): a household press („Mentés most") is not the night's backup. A controller + // that knows sends `trigger=manual`; then the OS leg does not follow (it belongs to the night, after the night's + // own copy). An older controller sends nothing and keeps the old behaviour. + manual := r.URL.Query().Get("trigger") == "manual" // ONE BACKUP AT A TIME PER GUEST, ACROSS ALL TIERS (operator ruling 2026-07-26: "other backup // shouldn't start until finished"). vzdump takes a guest lock, so a concurrent second backup @@ -926,7 +930,9 @@ func (s *Server) handleBackup(w http.ResponseWriter, r *http.Request, vmid int) } s.finishJob(key, jobID, b) // OS leg (agent v0.140.0): after the night's whole-guest copy exists, still holding the heavy-op gate. - if b.Success && tier.Primary && s.afterPrimaryBackup != nil { + if b.Success && tier.Primary && s.afterPrimaryBackup != nil && manual { + s.logger.Info("local-api: no OS leg after a manual backup — it follows the night's own backup (R-899)", "vmid", vmid, "job", jobID) + } else if b.Success && tier.Primary && s.afterPrimaryBackup != nil { s.afterPrimaryBackup(base, vmid) } }() diff --git a/internal/osupdate/kernel.go b/internal/osupdate/kernel.go index 56cd1f1..4219b37 100644 --- a/internal/osupdate/kernel.go +++ b/internal/osupdate/kernel.go @@ -51,6 +51,24 @@ type KernelView struct { VMID int `json:"vmid"` // the customer guest the step was staged for (the health rule's guest) } +// kernelReportRing is the ring an after-boot report carries. In the first second after a boot the agent has not +// fetched the hub's block yet, and Block() then answers ring 1 — so a ring-0 box's „judging" report said ring 1 +// (seen on demo-felhom, 2026-10-08 night, `audits/kernel-night-2026-10-07/readback/`). Order: the fetched block; the +// block the daemon saved on disk before the reboot (R-866); ring 1 as before. A label only: the hub's approval reads its +// own ring list. Pinned by TestKernelReportRing_BeforeFirstFetch. +func (l *Leg) kernelReportRing() int { + l.mu.Lock() + fetched := l.block + l.mu.Unlock() + if fetched != nil { + return fetched.Ring + } + if b, _, ok := LoadSavedBlock(l.planDir()); ok && b != nil { + return b.Ring + } + return 1 +} + func parseKernel(raw json.RawMessage) KernelView { var v KernelView _ = json.Unmarshal(raw, &v) @@ -190,7 +208,7 @@ func (l *Leg) KernelAfterBoot(ctx context.Context, vmid int, j KernelJudge) Repo if vmid <= 0 { vmid = v.VMID // after a boot the guest may not run yet — the step's own record names it } - rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, Mode: "kernel-boot", + rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-boot", ReleaseID: v.To, Kernel: rawOrNil(wr.Kernel)} switch wr.KernelEvent { case "fell_back": @@ -240,7 +258,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV for { if !hubReached && l.Hub != nil { // the hub's reachability IS this report reaching it (and the operator sees the box is back on the new kernel) - body, _ := json.Marshal(Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, + body, _ := json.Marshal(Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-boot", ReleaseID: v.To, Outcome: "judging", Kernel: mustRaw(v)}) rctx, cancel := context.WithTimeout(ctx, 30*time.Second) if err := l.Hub.PostOSReport(rctx, body); err == nil { @@ -280,7 +298,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV return Report{} } // not healthy by the deadline: tell the hub (best effort), then ONE self-revert into the old kernel - rep := l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, + rep := l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-revert", ReleaseID: v.To, Outcome: "health_failed", HealthReason: why + " — reverting to " + v.From, Kernel: mustRaw(v)}) lg.Error("osupdate: kernel step — the one-shot boot is NOT healthy; restarting ONCE into the old kernel", "reason", why, @@ -289,7 +307,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV if err != nil || wr.refused() || wr.failed() { lg.Error("osupdate: kernel self-revert did not start — the box stays on the new kernel; the operator decides", "err", err, "refused", string(firstRaw(wr.Refused, wr.Failed))) - return l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, + return l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-revert", ReleaseID: v.To, Outcome: "revert_failed", Refused: firstRaw(wr.Refused, wr.Failed), HealthReason: "the self-revert did not start"}) } @@ -298,7 +316,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV func (l *Leg) kernelGood(ctx context.Context, runID string, vmid int, v KernelView, start time.Time, lg *slog.Logger) Report { wr, err := l.call(ctx, runID, kernelPlan("kernel-good", vmid, nil)) - rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, Mode: "kernel-good", + rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-good", ReleaseID: v.To} switch { case err != nil: diff --git a/internal/osupdate/r899_ring_test.go b/internal/osupdate/r899_ring_test.go new file mode 100644 index 0000000..08e9393 --- /dev/null +++ b/internal/osupdate/r899_ring_test.go @@ -0,0 +1,32 @@ +package osupdate + +import ( + "os" + "path/filepath" + "testing" + + "gitea.dooplex.hu/admin/felhom-agent/internal/hub" +) + +// After a boot the agent has not fetched the hub's block yet; the after-boot kernel report must still carry the +// box's real ring (seen 2026-10-08: a ring-0 box's „judging" report said ring 1). Red-proof: return Block().Ring +// from kernelReportRing and the saved-block case fails with 1. +func TestKernelReportRing_BeforeFirstFetch(t *testing.T) { + dir := t.TempDir() + l := &Leg{PlanDir: dir} + if got := l.kernelReportRing(); got != 1 { + t.Fatalf("nothing fetched, nothing saved: ring %d, want 1 (the old default)", got) + } + l.saveBlock(&hub.WireOSUpdate{Ring: 0, Enabled: true}) + l2 := &Leg{PlanDir: dir} // a fresh daemon after the reboot: no block fetched yet + if got := l2.kernelReportRing(); got != 0 { + t.Fatalf("ring-0 block saved before the reboot: ring %d, want 0", got) + } + l2.SetBlock(&hub.WireOSUpdate{Ring: 1, Enabled: true}) + if got := l2.kernelReportRing(); got != 1 { + t.Fatalf("a fetched block wins over the saved one: ring %d, want 1", got) + } + if _, err := os.Stat(filepath.Join(dir, SavedBlockFile)); err != nil { + t.Fatalf("saved block missing: %v", err) + } +}