R-899: no OS leg after a household press (trigger=manual); after-boot kernel reports carry the saved ring
gates / gates (push) Successful in 50s
gates / gates (push) Successful in 50s
Unreleased; ships with tomorrow's release. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -1,3 +1,19 @@
|
||||
## Unreleased (2026-10-08) — no OS leg after a household press; the after-boot kernel report carries the real ring (R-899) — ships with tomorrow's release
|
||||
|
||||
**Delivery: the agent binary only** — no root file changed.
|
||||
|
||||
- **R-899 (operator ruling 2026-10-08, option A):** `POST /backup?trigger=manual` (a „Mentés most" press, sent by the
|
||||
controller from its next release) runs no OS leg after it: the leg belongs to the night, after the night's own copy.
|
||||
Before, a press ran the leg at once (in the day; on 2026-10-07 08:51 demo-hp skipped it only by the 20-hour rule). A
|
||||
request without the parameter (an older controller, the scheduled path) behaves as before.
|
||||
`internal/localapi/server.go`; `TestAfterPrimaryBackup` gained two sub-cases (red-proved: ignoring `trigger`, the leg
|
||||
ran once after a press).
|
||||
- **The „ring 1" label after a boot:** in the first second after a reboot the agent has not fetched the hub's block, and
|
||||
the after-boot kernel reports (`judging`, `good`, `revert`, `fell_back`…) said ring 1 on a ring-0 box (demo-felhom,
|
||||
2026-10-08 night). Now they read the fetched block, else the block the daemon saved on disk before the reboot (R-866),
|
||||
else ring 1 as before. A label only — the hub's approval reads its own ring list. `kernelReportRing`,
|
||||
`TestKernelReportRing_BeforeFirstFetch` (red-proved: ring 1, want 0).
|
||||
|
||||
## v0.153.0 — ring 0 stages exactly the told kernel (R-898; `09` §3 decision 176) (2026-10-07)
|
||||
|
||||
Released by `scripts/release-agent.sh`: binary sha256 `b204ebe6d65944f43b70dcfa4ae0dfb90da38c92e2ba6a38a1826e488eeb6630`, bundle
|
||||
|
||||
@@ -15,7 +15,7 @@ import (
|
||||
// Red-proof: drop the `b.Success &&` guard and the failed-backup sub-case fails; move the call after release()
|
||||
// and the gate sub-case fails.
|
||||
func TestAfterPrimaryBackup(t *testing.T) {
|
||||
run := func(t *testing.T, failErr string) (calls []int, gateHeld bool) {
|
||||
run := func(t *testing.T, failErr, path string) (calls []int, gateHeld bool) {
|
||||
gate := &backup.InFlight{}
|
||||
b := &fakeBackups{failErr: failErr}
|
||||
srv := newTestServerS(t, &fakeGuests{}, b, &fakeStore{}, nil)
|
||||
@@ -34,7 +34,7 @@ func TestAfterPrimaryBackup(t *testing.T) {
|
||||
done <- struct{}{}
|
||||
})
|
||||
h := srv.Handler()
|
||||
if do(t, h, "POST", "/backup", "A", "").Code != http.StatusAccepted {
|
||||
if do(t, h, "POST", path, "A", "").Code != http.StatusAccepted {
|
||||
t.Fatal("POST /backup not accepted")
|
||||
}
|
||||
select {
|
||||
@@ -47,7 +47,7 @@ func TestAfterPrimaryBackup(t *testing.T) {
|
||||
return calls, gateHeld
|
||||
}
|
||||
t.Run("success runs the leg under the gate", func(t *testing.T) {
|
||||
calls, held := run(t, "")
|
||||
calls, held := run(t, "", "/backup")
|
||||
if len(calls) != 1 {
|
||||
t.Fatalf("the leg ran %d time(s), want 1", len(calls))
|
||||
}
|
||||
@@ -56,8 +56,21 @@ func TestAfterPrimaryBackup(t *testing.T) {
|
||||
}
|
||||
})
|
||||
t.Run("a failed backup runs nothing", func(t *testing.T) {
|
||||
if calls, _ := run(t, "vzdump exploded"); len(calls) != 0 {
|
||||
if calls, _ := run(t, "vzdump exploded", "/backup"); len(calls) != 0 {
|
||||
t.Fatalf("the leg ran after a FAILED backup: %v", calls)
|
||||
}
|
||||
})
|
||||
// R-899: a household press is not the night's backup — no OS leg after it. Same fake, same successful backup as
|
||||
// the first sub-case; only the query differs. Red-proof: make handleBackup ignore `trigger` and this sub-case
|
||||
// fails with the leg run once.
|
||||
t.Run("a manual press runs nothing", func(t *testing.T) {
|
||||
if calls, _ := run(t, "", "/backup?trigger=manual"); len(calls) != 0 {
|
||||
t.Fatalf("the OS leg ran after a manual press: %v", calls)
|
||||
}
|
||||
})
|
||||
t.Run("the scheduled path with the new query still runs the leg", func(t *testing.T) {
|
||||
if calls, _ := run(t, "", "/backup?trigger=night"); len(calls) != 1 {
|
||||
t.Fatalf("the leg ran %d time(s) after a non-manual backup, want 1", len(calls))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -825,6 +825,10 @@ func (s *Server) handleBackup(w http.ResponseWriter, r *http.Request, vmid int)
|
||||
return
|
||||
}
|
||||
key := backupJobKey{vmid: vmid, target: tier.TargetID}
|
||||
// R-899 (operator ruling 2026-10-08): a household press („Mentés most") is not the night's backup. A controller
|
||||
// that knows sends `trigger=manual`; then the OS leg does not follow (it belongs to the night, after the night's
|
||||
// own copy). An older controller sends nothing and keeps the old behaviour.
|
||||
manual := r.URL.Query().Get("trigger") == "manual"
|
||||
|
||||
// ONE BACKUP AT A TIME PER GUEST, ACROSS ALL TIERS (operator ruling 2026-07-26: "other backup
|
||||
// shouldn't start until finished"). vzdump takes a guest lock, so a concurrent second backup
|
||||
@@ -926,7 +930,9 @@ func (s *Server) handleBackup(w http.ResponseWriter, r *http.Request, vmid int)
|
||||
}
|
||||
s.finishJob(key, jobID, b)
|
||||
// OS leg (agent v0.140.0): after the night's whole-guest copy exists, still holding the heavy-op gate.
|
||||
if b.Success && tier.Primary && s.afterPrimaryBackup != nil {
|
||||
if b.Success && tier.Primary && s.afterPrimaryBackup != nil && manual {
|
||||
s.logger.Info("local-api: no OS leg after a manual backup — it follows the night's own backup (R-899)", "vmid", vmid, "job", jobID)
|
||||
} else if b.Success && tier.Primary && s.afterPrimaryBackup != nil {
|
||||
s.afterPrimaryBackup(base, vmid)
|
||||
}
|
||||
}()
|
||||
|
||||
@@ -51,6 +51,24 @@ type KernelView struct {
|
||||
VMID int `json:"vmid"` // the customer guest the step was staged for (the health rule's guest)
|
||||
}
|
||||
|
||||
// kernelReportRing is the ring an after-boot report carries. In the first second after a boot the agent has not
|
||||
// fetched the hub's block yet, and Block() then answers ring 1 — so a ring-0 box's „judging" report said ring 1
|
||||
// (seen on demo-felhom, 2026-10-08 night, `audits/kernel-night-2026-10-07/readback/`). Order: the fetched block; the
|
||||
// block the daemon saved on disk before the reboot (R-866); ring 1 as before. A label only: the hub's approval reads its
|
||||
// own ring list. Pinned by TestKernelReportRing_BeforeFirstFetch.
|
||||
func (l *Leg) kernelReportRing() int {
|
||||
l.mu.Lock()
|
||||
fetched := l.block
|
||||
l.mu.Unlock()
|
||||
if fetched != nil {
|
||||
return fetched.Ring
|
||||
}
|
||||
if b, _, ok := LoadSavedBlock(l.planDir()); ok && b != nil {
|
||||
return b.Ring
|
||||
}
|
||||
return 1
|
||||
}
|
||||
|
||||
func parseKernel(raw json.RawMessage) KernelView {
|
||||
var v KernelView
|
||||
_ = json.Unmarshal(raw, &v)
|
||||
@@ -190,7 +208,7 @@ func (l *Leg) KernelAfterBoot(ctx context.Context, vmid int, j KernelJudge) Repo
|
||||
if vmid <= 0 {
|
||||
vmid = v.VMID // after a boot the guest may not run yet — the step's own record names it
|
||||
}
|
||||
rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, Mode: "kernel-boot",
|
||||
rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-boot",
|
||||
ReleaseID: v.To, Kernel: rawOrNil(wr.Kernel)}
|
||||
switch wr.KernelEvent {
|
||||
case "fell_back":
|
||||
@@ -240,7 +258,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV
|
||||
for {
|
||||
if !hubReached && l.Hub != nil {
|
||||
// the hub's reachability IS this report reaching it (and the operator sees the box is back on the new kernel)
|
||||
body, _ := json.Marshal(Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid,
|
||||
body, _ := json.Marshal(Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid,
|
||||
Mode: "kernel-boot", ReleaseID: v.To, Outcome: "judging", Kernel: mustRaw(v)})
|
||||
rctx, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
if err := l.Hub.PostOSReport(rctx, body); err == nil {
|
||||
@@ -280,7 +298,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV
|
||||
return Report{}
|
||||
}
|
||||
// not healthy by the deadline: tell the hub (best effort), then ONE self-revert into the old kernel
|
||||
rep := l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid,
|
||||
rep := l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid,
|
||||
Mode: "kernel-revert", ReleaseID: v.To, Outcome: "health_failed", HealthReason: why + " — reverting to " + v.From,
|
||||
Kernel: mustRaw(v)})
|
||||
lg.Error("osupdate: kernel step — the one-shot boot is NOT healthy; restarting ONCE into the old kernel", "reason", why,
|
||||
@@ -289,7 +307,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV
|
||||
if err != nil || wr.refused() || wr.failed() {
|
||||
lg.Error("osupdate: kernel self-revert did not start — the box stays on the new kernel; the operator decides",
|
||||
"err", err, "refused", string(firstRaw(wr.Refused, wr.Failed)))
|
||||
return l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid,
|
||||
return l.finish(ctx, lg, Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid,
|
||||
Mode: "kernel-revert", ReleaseID: v.To, Outcome: "revert_failed", Refused: firstRaw(wr.Refused, wr.Failed),
|
||||
HealthReason: "the self-revert did not start"})
|
||||
}
|
||||
@@ -298,7 +316,7 @@ func (l *Leg) judgeKernel(ctx context.Context, runID string, vmid int, v KernelV
|
||||
|
||||
func (l *Leg) kernelGood(ctx context.Context, runID string, vmid int, v KernelView, start time.Time, lg *slog.Logger) Report {
|
||||
wr, err := l.call(ctx, runID, kernelPlan("kernel-good", vmid, nil))
|
||||
rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.Block().Ring, VMID: vmid, Mode: "kernel-good",
|
||||
rep := Report{RunID: runID, Layer: LayerKernel, Trigger: "boot", Ring: l.kernelReportRing(), VMID: vmid, Mode: "kernel-good",
|
||||
ReleaseID: v.To}
|
||||
switch {
|
||||
case err != nil:
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
package osupdate
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
||||
)
|
||||
|
||||
// After a boot the agent has not fetched the hub's block yet; the after-boot kernel report must still carry the
|
||||
// box's real ring (seen 2026-10-08: a ring-0 box's „judging" report said ring 1). Red-proof: return Block().Ring
|
||||
// from kernelReportRing and the saved-block case fails with 1.
|
||||
func TestKernelReportRing_BeforeFirstFetch(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
l := &Leg{PlanDir: dir}
|
||||
if got := l.kernelReportRing(); got != 1 {
|
||||
t.Fatalf("nothing fetched, nothing saved: ring %d, want 1 (the old default)", got)
|
||||
}
|
||||
l.saveBlock(&hub.WireOSUpdate{Ring: 0, Enabled: true})
|
||||
l2 := &Leg{PlanDir: dir} // a fresh daemon after the reboot: no block fetched yet
|
||||
if got := l2.kernelReportRing(); got != 0 {
|
||||
t.Fatalf("ring-0 block saved before the reboot: ring %d, want 0", got)
|
||||
}
|
||||
l2.SetBlock(&hub.WireOSUpdate{Ring: 1, Enabled: true})
|
||||
if got := l2.kernelReportRing(); got != 1 {
|
||||
t.Fatalf("a fetched block wins over the saved one: ring %d, want 1", got)
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(dir, SavedBlockFile)); err != nil {
|
||||
t.Fatalf("saved block missing: %v", err)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user