OS updates, guest fast lane (11 §8 step 2): felhom-os-apply wrapper (R1-R13 refusals, repair first, snapshot.debian.org fallback), FELHOM_OSAPPLY sudoers, the OS leg after the primary backup, hub os_update block + os-report, --selftest=os-update
gates / gates (push) Successful in 18s

No automatic undo: a customer guest cannot be snapshotted (R-837, measured).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
2026-10-04 10:44:29 +02:00
parent 596238cc2e
commit 23a8ef3de4
13 changed files with 1877 additions and 1 deletions
+63
View File
@@ -0,0 +1,63 @@
package localapi
import (
"context"
"net/http"
"sync"
"testing"
"time"
"gitea.dooplex.hu/admin/felhom-agent/internal/backup"
)
// The OS leg (agent v0.140.0) runs after a SUCCESSFUL primary backup, and only then; and it runs BEFORE the
// host-wide heavy-op gate is released, so a restore-test cannot start in the middle of it (`11` C10).
// Red-proof: drop the `b.Success &&` guard and the failed-backup sub-case fails; move the call after release()
// and the gate sub-case fails.
func TestAfterPrimaryBackup(t *testing.T) {
run := func(t *testing.T, failErr string) (calls []int, gateHeld bool) {
gate := &backup.InFlight{}
b := &fakeBackups{failErr: failErr}
srv := newTestServerS(t, &fakeGuests{}, b, &fakeStore{}, nil)
srv.inFlight = gate
var mu sync.Mutex
done := make(chan struct{}, 1)
srv.SetAfterPrimaryBackup(func(_ context.Context, vmid int) {
rel, _, ok := gate.TryAcquire("probe")
mu.Lock()
calls = append(calls, vmid)
gateHeld = !ok
mu.Unlock()
if ok {
rel()
}
done <- struct{}{}
})
h := srv.Handler()
if do(t, h, "POST", "/backup", "A", "").Code != http.StatusAccepted {
t.Fatal("POST /backup not accepted")
}
select {
case <-done:
case <-time.After(500 * time.Millisecond):
}
time.Sleep(20 * time.Millisecond)
mu.Lock()
defer mu.Unlock()
return calls, gateHeld
}
t.Run("success runs the leg under the gate", func(t *testing.T) {
calls, held := run(t, "")
if len(calls) != 1 {
t.Fatalf("the leg ran %d time(s), want 1", len(calls))
}
if !held {
t.Fatal("the heavy-op gate was free while the leg ran — a restore-test could overlap it")
}
})
t.Run("a failed backup runs nothing", func(t *testing.T) {
if calls, _ := run(t, "vzdump exploded"); len(calls) != 0 {
t.Fatalf("the leg ran after a FAILED backup: %v", calls)
}
})
}
+13
View File
@@ -160,6 +160,10 @@ type Options struct {
// NetStorage is the privileged network-mount (NAS) surface (Part A1). OPTIONAL — when nil, the
// /netstorage endpoints report "not configured". Satisfied by *storage.SudoHostOps.
NetStorage NetworkStorageOps
// AfterPrimaryBackup (agent v0.140.0, `11-os-updates.md` §8 step 2) runs right after a SUCCESSFUL backup on the
// PRIMARY tier, inside the backup goroutine and BEFORE the host-wide heavy-op gate is released — so the OS leg
// that it starts can never overlap another backup or a restore-test (`11` C10). OPTIONAL — nil → nothing runs.
AfterPrimaryBackup func(ctx context.Context, vmid int)
// Privileged runs the fenced root wrappers (E-2a: felhom-backup-target-apply). OPTIONAL — when
// nil, POST /backup/target reports "not configured". Satisfied by *proxmox.ExecRunner.
Privileged PrivilegedRunner
@@ -276,6 +280,7 @@ type Server struct {
tiers []BackupTier
// inFlight (R-85) is shared with the restore-test scheduler so the two never run together.
inFlight *backup.InFlight
afterPrimaryBackup func(ctx context.Context, vmid int) // the OS leg (agent v0.140.0); nil = none
logger *slog.Logger
now func() time.Time
@@ -469,6 +474,7 @@ func NewServer(o Options) (*Server, error) {
// the primary is always first, because that is what the untargeted endpoints act on.
s.tiers = normalizeBackupTiers(o.BackupTiers, o.Backups, cadence)
s.inFlight = o.InFlight
s.afterPrimaryBackup = o.AfterPrimaryBackup
if s.backups == nil && len(s.tiers) > 0 {
s.backups = s.tiers[0].Service
}
@@ -894,6 +900,10 @@ func (s *Server) handleBackup(w http.ResponseWriter, r *http.Request, vmid int)
}
s.store.RecordBackup(b)
s.finishJob(key, jobID, b)
// OS leg (agent v0.140.0): after the night's whole-guest copy exists, still holding the heavy-op gate.
if b.Success && tier.Primary && s.afterPrimaryBackup != nil {
s.afterPrimaryBackup(base, vmid)
}
}()
writeStatus(w, http.StatusAccepted, true, BackupResponse{VMID: vmid, JobID: jobID, Phase: PhaseRunning}, "")
}
@@ -1465,3 +1475,6 @@ func writeStatus(w http.ResponseWriter, code int, ok bool, data any, errMsg stri
w.WriteHeader(code)
_ = json.NewEncoder(w).Encode(apiResponse{OK: ok, Data: data, Error: errMsg})
}
// SetAfterPrimaryBackup wires the hook that runs after a successful primary-tier backup (the OS leg, agent v0.140.0).
func (s *Server) SetAfterPrimaryBackup(fn func(ctx context.Context, vmid int)) { s.afterPrimaryBackup = fn }