v0.145.0 code: the OS wrapper repairs dpkg's update journal by itself after a power cut (R-876); restore-test first check 30 min after start (R-874); neutral "sent late" text (R-875)
gates / gates (push) Successful in 20s
gates / gates (push) Successful in 20s
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0159rPz1ZhFKsS53msqPYxtS
This commit is contained in:
@@ -60,12 +60,21 @@ type Scheduler struct {
|
||||
|
||||
// R-85 tier rotation. All optional: without them the scheduler behaves exactly as before
|
||||
// (single tier via `pick`), which keeps every existing caller and test working untouched.
|
||||
tiers []string // configured tier target ids, primary first
|
||||
tierPick TierPicker // newest archive on a named tier
|
||||
rtState *RestoreTestState // persisted last-successful-per-tier (drives oldest-first)
|
||||
inFlight *InFlight // shared with the backup path — Scenario F
|
||||
tiers []string // configured tier target ids, primary first
|
||||
tierPick TierPicker // newest archive on a named tier
|
||||
rtState *RestoreTestState // persisted last-successful-per-tier (drives oldest-first)
|
||||
inFlight *InFlight // shared with the backup path — Scenario F
|
||||
firstEval time.Duration // R-874: the first evaluation after start
|
||||
}
|
||||
|
||||
// DefaultFirstEval (R-874): the first due-ness evaluation runs 30 minutes after the agent starts, then every
|
||||
// cadence. MEASURED need (2026-10-05 Part F spike): a box whose power-on sessions are all shorter than the 6 h
|
||||
// interval (Tester 2: ~1.5 h and ~5 min) NEVER evaluated, because the ticker restarts at each start. 30 minutes
|
||||
// keeps the earned restraint below — a crash-looping agent restarts far more often than that and still never
|
||||
// evaluates — while a box that stays on for half an hour gets its due test. Pinned by
|
||||
// TestR874_FirstEvaluationAfterStart and TestR874_CrashLoopNeverEvaluates.
|
||||
const DefaultFirstEval = 30 * time.Minute
|
||||
|
||||
// SchedulerOptions configures a Scheduler.
|
||||
type SchedulerOptions struct {
|
||||
Runner RestoreTestRunner
|
||||
@@ -81,6 +90,8 @@ type SchedulerOptions struct {
|
||||
// 0 → no settle requirement (any archive is a candidate).
|
||||
Settle time.Duration
|
||||
Logger *slog.Logger
|
||||
// FirstEval (R-874, v0.145.0) is when the FIRST evaluation runs after start; 0 → DefaultFirstEval.
|
||||
FirstEval time.Duration
|
||||
|
||||
// R-85 (all optional — omit for the pre-R-85 single-tier behaviour):
|
||||
// Tiers are the configured tier target ids (primary first); TierPick resolves an archive on a
|
||||
@@ -110,6 +121,12 @@ func NewScheduler(opts SchedulerOptions) *Scheduler {
|
||||
tierPick: opts.TierPick,
|
||||
rtState: opts.State,
|
||||
inFlight: opts.InFlight,
|
||||
firstEval: func() time.Duration {
|
||||
if opts.FirstEval > 0 {
|
||||
return opts.FirstEval
|
||||
}
|
||||
return DefaultFirstEval
|
||||
}(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -120,8 +137,9 @@ func NewScheduler(opts SchedulerOptions) *Scheduler {
|
||||
// trigger any more: its phase is the process's uptime, and agent deploys reset it, which is exactly
|
||||
// the defect R-86 removes. What decides that a test happens is `EvaluateDue`.
|
||||
//
|
||||
// It still does NOT evaluate immediately on start — the first evaluation is one interval in. That
|
||||
// is an EARNED restraint, kept deliberately: a restore is heavy, agent restarts are routine, and a
|
||||
// It still does NOT evaluate immediately on start. v0.145.0 (R-874): the first evaluation is
|
||||
// firstEval (30 min) in, then every interval — it was one full interval in, which a box with short
|
||||
// power-on sessions never reached. The restraint itself is EARNED and kept: a restore is heavy, agent restarts are routine, and a
|
||||
// crash-loop that evaluated at start would hammer a permanently-failing tier as fast as it could
|
||||
// restart. Due-ness does not expire while we wait, so the only cost is up to one interval of
|
||||
// latency on a tier that just became due. On-demand runs use `--selftest=restore-test`.
|
||||
@@ -135,6 +153,16 @@ func (s *Scheduler) Run(ctx context.Context) error {
|
||||
}
|
||||
s.logger.Info("backup: restore-test scheduler starting (per-archive due-check)",
|
||||
"eval_interval", s.cadence, "settle", s.settle)
|
||||
first := time.NewTimer(s.firstEval)
|
||||
defer first.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
s.logger.Info("backup: restore-test scheduler shutting down", "reason", ctx.Err())
|
||||
return nil
|
||||
case <-first.C:
|
||||
s.logger.Info("backup: restore-test first evaluation after start (R-874)", "after", s.firstEval)
|
||||
s.tick(ctx)
|
||||
}
|
||||
t := time.NewTicker(s.cadence)
|
||||
defer t.Stop()
|
||||
for {
|
||||
|
||||
Reference in New Issue
Block a user