56ef1d6655
gates / gates (push) Successful in 20s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
198 lines
7.3 KiB
Go
198 lines
7.3 KiB
Go
package osupdate
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"syscall"
|
|
"time"
|
|
|
|
"gitea.dooplex.hu/admin/felhom-agent/internal/hub"
|
|
)
|
|
|
|
// ── R-868 (v0.144.0): a pass whose agent was killed still reports ─────────────────────────────────────
|
|
//
|
|
// MEASURED 2026-10-05 02:57 UTC on demo-hp (night drill A5): the debug pass and the agent daemon were kill -9-ed while
|
|
// apt-get ran. The root wrapper (its own process under sudo) finished all six packages, but the agent that would have
|
|
// read its stdout and posted the report was gone; the hub never learned what the pass installed.
|
|
//
|
|
// THE MECHANISM: the wrapper writes its apply report to <plan dir>/report-<run>-<layer>-apply.json BEFORE printing
|
|
// it (configs/felhom-os-apply save_report). The agent deletes that copy once the hub has the report (finish). A copy
|
|
// still on disk is a report nobody sent: SendUnsent posts it — at the agent's start, and before every pass — and then
|
|
// deletes it. A pass lock (flock on <plan dir>/pass.lock, released by the kernel when a process dies) keeps the
|
|
// sender from picking up the copy of a pass that is still running, also across the daemon and a selftest process.
|
|
// Pinned by TestR868_* (unsent_test.go).
|
|
|
|
// lockPass takes the pass lock. block=false returns ok=false at once when another pass holds it. A lock that cannot
|
|
// be opened at all (no plan dir yet) does not stop a pass: the unlock is then a no-op.
|
|
func (l *Leg) lockPass(block bool) (unlock func()) {
|
|
u, _ := l.tryLockPass(block)
|
|
return u
|
|
}
|
|
|
|
func (l *Leg) tryLockPass(block bool) (unlock func(), ok bool) {
|
|
dir := l.planDir()
|
|
_ = os.MkdirAll(dir, 0o700)
|
|
f, err := os.OpenFile(filepath.Join(dir, "pass.lock"), os.O_CREATE|os.O_RDWR, 0o600)
|
|
if err != nil {
|
|
l.log().Warn("osupdate: pass lock unavailable — continuing without it", "err", err)
|
|
return func() {}, true
|
|
}
|
|
how := syscall.LOCK_EX
|
|
if !block {
|
|
how |= syscall.LOCK_NB
|
|
}
|
|
if err := syscall.Flock(int(f.Fd()), how); err != nil {
|
|
f.Close()
|
|
return func() {}, false
|
|
}
|
|
return func() { _ = syscall.Flock(int(f.Fd()), syscall.LOCK_UN); f.Close() }, true
|
|
}
|
|
|
|
// SendUnsent posts every report a pass kept on disk and nobody sent (R-868). It skips when a pass runs now (that
|
|
// pass sends them first). Called at the agent's start.
|
|
func (l *Leg) SendUnsent(ctx context.Context) int {
|
|
unlock, ok := l.tryLockPass(false)
|
|
if !ok {
|
|
l.log().Info("osupdate: a pass is running — its start sends any kept report")
|
|
return 0
|
|
}
|
|
defer unlock()
|
|
return l.sendUnsentLocked(ctx)
|
|
}
|
|
|
|
// SendUnsentLoop runs SendUnsent now and then every `every` until ctx ends (v0.144.1). MEASURED live on demo-hp
|
|
// 2026-10-05: after a kill -9 the daemon restarted in ~5 s, while the orphaned root wrapper was still installing — its
|
|
// copy appeared ~7 s AFTER the start-time sender had looked. One look at start is therefore not enough. A glob of the
|
|
// plan dir every few minutes costs nothing; the pass lock keeps it off a running pass. Pinned by
|
|
// TestR868_ACopyWrittenAfterTheStartIsSentByTheLoop.
|
|
func (l *Leg) SendUnsentLoop(ctx context.Context, every time.Duration, onSent func(int)) {
|
|
for {
|
|
if n := l.SendUnsent(ctx); n > 0 && onSent != nil {
|
|
onSent(n)
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-time.After(every):
|
|
}
|
|
}
|
|
}
|
|
|
|
func (l *Leg) sendUnsentLocked(ctx context.Context) int {
|
|
if l.Hub == nil {
|
|
return 0 // nobody to send to: keep the copies for a process that has the hub
|
|
}
|
|
files, _ := filepath.Glob(filepath.Join(l.planDir(), "report-*.json"))
|
|
sent := 0
|
|
for _, f := range files {
|
|
b, err := os.ReadFile(f)
|
|
if err != nil {
|
|
l.log().Warn("osupdate: a kept report cannot be read", "path", f, "err", err)
|
|
continue
|
|
}
|
|
var wr WrapperReport
|
|
if err := json.Unmarshal(b, &wr); err != nil || wr.Layer == "" {
|
|
l.log().Warn("osupdate: a kept report is not a report — moved aside", "path", f, "err", err)
|
|
_ = os.Rename(f, f+".bad")
|
|
continue
|
|
}
|
|
rep := l.reportFromKept(ctx, wr, f)
|
|
lg := l.log().With("run", rep.RunID, "layer", rep.Layer, "vmid", rep.VMID, "ring", rep.Ring, "trigger", rep.Trigger)
|
|
lg.Info("osupdate: sending a kept report late (the agent was stopped mid-pass or the hub was away, R-868)", "path", f)
|
|
before := rep.unsent
|
|
_ = l.finish(ctx, lg, rep)
|
|
if _, err := os.Stat(before); os.IsNotExist(err) {
|
|
sent++
|
|
// the pass's plan file is left behind too when the agent was killed inside call()
|
|
_ = os.Remove(filepath.Join(filepath.Dir(f), "plan-"+strings.TrimPrefix(filepath.Base(f), "report-")))
|
|
}
|
|
}
|
|
return sent
|
|
}
|
|
|
|
// reportFromKept builds the hub report from a kept wrapper report, as runLayer would have. Health: the copy's own
|
|
// before/after reading; when that is not healthy after an install, one fresh reading (services restart after an
|
|
// install, and the pass that would have waited for them is gone).
|
|
func (l *Leg) reportFromKept(ctx context.Context, wr WrapperReport, path string) Report {
|
|
ring := 1
|
|
if wr.Ring != nil {
|
|
ring = *wr.Ring
|
|
}
|
|
runID, trigger := wr.RunID, wr.Trigger
|
|
if runID == "" {
|
|
runID = strings.TrimSuffix(strings.TrimPrefix(filepath.Base(path), "report-"), ".json")
|
|
}
|
|
if trigger == "" {
|
|
trigger = "unknown"
|
|
}
|
|
rep := Report{RunID: runID, Layer: wr.Layer, Trigger: trigger, Mode: wr.Mode, Ring: ring, ReleaseID: wr.ReleaseID,
|
|
VMID: wr.VMID, unsent: path}
|
|
prefix := "sent late — kept on the box until the hub could take it (R-868, R-875)"
|
|
switch {
|
|
case wr.refused():
|
|
rep.Outcome, rep.Refused, rep.HealthReason = "refused", wr.Refused, prefix
|
|
return rep
|
|
case wr.failed():
|
|
rep.Outcome, rep.Refused = "failed", wr.Failed
|
|
case len(wr.Upgraded) == 0:
|
|
rep.Outcome = "nothing"
|
|
default:
|
|
rep.Outcome = "applied"
|
|
}
|
|
rep.Upgraded, rep.PassSeconds = wr.Upgraded, wr.PassSeconds
|
|
rep.DockerEngine, rep.Authority, rep.Undo = wr.DockerEngine, wr.Authority, wr.Undo
|
|
wantEngine := ""
|
|
for _, u := range wr.Upgraded {
|
|
if u.Name == "docker-ce" {
|
|
wantEngine = EngineOf(u.Version)
|
|
}
|
|
}
|
|
verdict := func(h *Health) (bool, string) {
|
|
switch wr.Layer {
|
|
case LayerDocker:
|
|
return DockerHealthVerdict(wr.HealthBefore, h, wantEngine, wr.DockerEngine)
|
|
case LayerHost:
|
|
t := hub.TunnelUnknown
|
|
if l.Tunnel != nil {
|
|
t, _ = l.Tunnel.Status(ctx)
|
|
}
|
|
return HostHealthVerdict(wr.HealthBefore, h, t)
|
|
}
|
|
return HealthVerdict(wr.HealthBefore, h)
|
|
}
|
|
ok, why := verdict(wr.HealthAfter)
|
|
if !ok && len(wr.Upgraded) > 0 && wr.VMID > 0 {
|
|
lane := "fast"
|
|
if wr.Layer == LayerDocker {
|
|
lane = "slow"
|
|
}
|
|
if hr, err := l.call(ctx, "kept-"+runID, map[string]any{"release_id": "kept", "layer": wr.Layer, "lane": lane,
|
|
"vmid": wr.VMID, "mode": "health", "packages": []Package{}}); err == nil && hr.Health != nil {
|
|
ok, why = verdict(hr.Health)
|
|
}
|
|
}
|
|
rep.Healthy, rep.HealthReason = ok, prefix
|
|
if why != "" {
|
|
rep.HealthReason = prefix + ": " + why
|
|
}
|
|
if !ok && rep.Outcome == "applied" {
|
|
rep.Outcome = "health_failed"
|
|
}
|
|
rep.Installed, rep.Pending = wr.Installed, wr.Pending
|
|
rep.RestartNeeded, rep.DockerRestartNeeded, rep.RebootNeeded = wr.RestartNeeded, wr.DockerRestartNeeded, wr.RebootNeeded
|
|
rep.RebootScanned = wr.RebootScanned
|
|
if wr.Layer == LayerDocker {
|
|
rep.Installed, rep.Pending = onlyDocker(wr.Installed), onlyDockerPending(wr.Pending)
|
|
} else {
|
|
planned := map[string]bool{}
|
|
for _, u := range wr.Upgraded {
|
|
planned[u.Name] = true
|
|
}
|
|
rep.NotCovered = notCovered(wr.Pending, ring, planned)
|
|
}
|
|
return rep
|
|
}
|